diff --git a/scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py b/scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py new file mode 100644 index 00000000..16fc450e --- /dev/null +++ b/scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py @@ -0,0 +1,553 @@ +"""Evaluate the 2009 Korkontzelos–Manandhar Table 1 on the 225 development rows. + +The pass does not select the source, integrate it, or draw a measurement sample. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter, defaultdict +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.km_candidate_evaluation import align_item, join_synset, lookup_key +from hyperlexical.unbind_sense_screen_v1 import load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +PDF = LEDGER / "acquisition/sources/korkontzelos-manandhar-2009/P09-2017.pdf" +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +PROVENANCE_PATH = SOURCE / "KM_SOURCE_PROVENANCE.json" +LICENSE_PATH = SOURCE / "KM_LICENSE_RECEIPT.json" +INVENTORY_PATH = SOURCE / "KM_RAW_EVALUATION_INVENTORY.jsonl" +ALIGNMENT_PATH = SOURCE / "KM_PWN30_ALIGNMENT.jsonl" +SEMANTIC_PATH = SOURCE / "KM_HYPERLEX_SEMANTIC_EVIDENCE.jsonl" +EVALUATION_PATH = SOURCE / "KM_DEVELOPMENT_EVALUATION.json" +COMPARISON_PATH = SOURCE / "KM_MAGPIE_COMPARISON.json" +DECISION_PATH = SOURCE / "KM_CANDIDATE_DECISION.json" + +PUBLICATION = "P09-2017" +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +SEMANTIC_STATES = ("YES", "NO", "UNKNOWN") +ALIGNMENT_STATES = ( + "EXACT_SOURCE_ID", + "EXACT_UNIQUE_RECONSTRUCTION", + "AMBIGUOUS_MULTIPLE_SYNSETS", + "NO_PWN3_MATCH", + "VERSION_CONFLICT", + "UNKNOWN", +) +TABLE = ( + ("KM2009-NC-01", "NONCOMPOSITIONAL", "agony aunt"), + ("KM2009-NC-02", "NONCOMPOSITIONAL", "black maria"), + ("KM2009-NC-03", "NONCOMPOSITIONAL", "dead end"), + ("KM2009-NC-04", "NONCOMPOSITIONAL", "dutch oven"), + ("KM2009-NC-05", "NONCOMPOSITIONAL", "fish finger"), + ("KM2009-NC-06", "NONCOMPOSITIONAL", "fool\u2019s paradise"), + ("KM2009-NC-07", "NONCOMPOSITIONAL", "goat\u2019s rue"), + ("KM2009-NC-08", "NONCOMPOSITIONAL", "green light"), + ("KM2009-NC-09", "NONCOMPOSITIONAL", "high jump"), + ("KM2009-NC-10", "NONCOMPOSITIONAL", "joint chiefs"), + ("KM2009-NC-11", "NONCOMPOSITIONAL", "lip service"), + ("KM2009-NC-12", "NONCOMPOSITIONAL", "living rock"), + ("KM2009-NC-13", "NONCOMPOSITIONAL", "monkey puzzle"), + ("KM2009-NC-14", "NONCOMPOSITIONAL", "motor pool"), + ("KM2009-NC-15", "NONCOMPOSITIONAL", "prince Albert"), + ("KM2009-NC-16", "NONCOMPOSITIONAL", "stocking stuffer"), + ("KM2009-NC-17", "NONCOMPOSITIONAL", "sweet bay"), + ("KM2009-NC-18", "NONCOMPOSITIONAL", "teddy boy"), + ("KM2009-NC-19", "NONCOMPOSITIONAL", "think tank"), + ("KM2009-CO-01", "COMPOSITIONAL", "box white oak"), + ("KM2009-CO-02", "COMPOSITIONAL", "cartridge brass"), + ("KM2009-CO-03", "COMPOSITIONAL", "common iguana"), + ("KM2009-CO-04", "COMPOSITIONAL", "closed chain"), + ("KM2009-CO-05", "COMPOSITIONAL", "eastern pipistrel"), + ("KM2009-CO-06", "COMPOSITIONAL", "field mushroom"), + ("KM2009-CO-07", "COMPOSITIONAL", "hard candy"), + ("KM2009-CO-08", "COMPOSITIONAL", "king snake"), + ("KM2009-CO-09", "COMPOSITIONAL", "labor camp"), + ("KM2009-CO-10", "COMPOSITIONAL", "lemon tree"), + ("KM2009-CO-11", "COMPOSITIONAL", "life form"), + ("KM2009-CO-12", "COMPOSITIONAL", "parenthesis-free notation"), + ("KM2009-CO-13", "COMPOSITIONAL", "parking brake"), + ("KM2009-CO-14", "COMPOSITIONAL", "petit juror"), + ("KM2009-CO-15", "COMPOSITIONAL", "relational adjective"), + ("KM2009-CO-16", "COMPOSITIONAL", "taxonomic category"), + ("KM2009-CO-17", "COMPOSITIONAL", "telephone service"), + ("KM2009-CO-18", "COMPOSITIONAL", "tea table"), + ("KM2009-CO-19", "COMPOSITIONAL", "upland cotton"), +) + +EXPECTED = { + SENSE / "CLASSIFICATION_PROCEDURE.json": "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + SENSE / "CLASSIFICATION_PROCEDURE.v2.json": "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + SENSE / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + SENSE / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE_MANIFEST: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + SENSE / "development_replay_predictions.jsonl": "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + SENSE / "development_replay_report.json": "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + SENSE / "development_replay_v2_predictions.jsonl": "1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a", + SENSE / "development_replay_v2_report.json": "93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5", + SENSE / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + SENSE / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + SENSE / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + SENSE / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + SENSE / "V2_DEVELOPMENT_RESULT_REVIEW.json": "77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d", + SENSE / "WORDNET_STRUCTURAL_SOURCE_LIMITATION.json": "3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a", + SOURCE / "HYPOTHESIS.json": "39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787", + SOURCE / "ACCEPTANCE.json": "1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94", + SOURCE / "CANDIDATE_SOURCE_EVALUATION_PLAN.json": "472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a", + SOURCE / "SEMANTIC_COMPOSITIONALITY.architecture.json": "180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36", + SOURCE / "MAGPIE_SOURCE_PROVENANCE.json": "bf0dd1dd747a6423406d97393f375f99620894a20af6bd4758e2de738d5c82dd", + SOURCE / "MAGPIE_LICENSE_RECEIPT.json": "8813818aa3704ba1e764121d2f66ff1630862a66d6c0c0b959b6afa35e0c3972", + SOURCE / "MAGPIE_DEVELOPMENT_MATCHES.jsonl": "84c847cfa545883de5a31979133fed87b0cdf9a7d13074c74cf227d9bfcadc83", + SOURCE / "MAGPIE_SENSE_ALIGNMENT.jsonl": "037b0f4d96d463aa7c5fbecdbef06a530ffbf770735232c92bd6abd0dd71fc66", + SOURCE / "MAGPIE_SEMANTIC_EVIDENCE.jsonl": "89f7227e1098407c7aaae6d9876b1f5780dbfe5b6a2eb3c3b09357d57822b577", + SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json": "74e2174d15c486dc60e9ad6be338199ff66a2d0950ed268105119323711108ba", + SOURCE / "MAGPIE_CANDIDATE_DECISION.json": "6eaa968b6260946998dba13e5c423f178d3349cdfe06e5ea401717f5a9bcdd0d", + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7/measurement_error_analysis.json": "ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8", + EVENTS: "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + LEDGER_FILE: "77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0", + TRACKER: "c3fe6f2fd21d07b8b45d9f26ffeebbcfbe3cb68a1f5a7952dee94d9a2c995936", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v2.py": "4b6f125da435b365af143c187902093bee5c9502db5b813a3d2bba11f889fa2b", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", + PDF: "046da9fc26cfdf220e41ad914314f0703ea3c84cd86e045b20146b34189113d1", + WORDNET / "data.noun": "489f145e0f68877c0be5bd0eb4117adaaac52f38f6204eb8d85dbe2158b614cc", + WORDNET / "data.verb": "29cc96ed80c9f47d94fe75e332a9df80f4b1c737205f92d2f433d63c6da2ab51", + WORDNET / "data.adj": "f24b635368be441501c9b8001e9271fd3b30b203f00d91e332979e6f8fe35646", + WORDNET / "data.adv": "e66dbbda0e0359e41b7f225bff71dd0c263dc7c66c1b61abc9ba334973d92979", + WORDNET / "README": "adad8d28ddea1db05b67ba1ac23506b025d29e0bcbf23bb35dde346089d8808d", +} + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str: + return f"{numerator}/{denominator}" + + +def lemma_index(root: Path) -> dict[str, list[dict]]: + synsets, glosses = load_wordnet(root) + grouped: dict[str, dict[str, dict]] = defaultdict(dict) + for (pos, offset), synset in synsets.items(): + if pos not in {"noun", "verb", "adj", "adv"}: + continue + synset_id = f"{pos}:{offset}" + gloss = " ".join(glosses[(pos, offset)].split()) + bucket = grouped[synset_id] + if not bucket: + bucket["gloss"] = gloss + bucket["lemmas"] = [] + bucket["synset"] = synset_id + for lemma in synset.lemmas: + if lemma not in bucket["lemmas"]: + bucket["lemmas"].append(lemma) + by_key: dict[str, list[dict]] = defaultdict(list) + for record in grouped.values(): + record["lemmas"] = sorted(record["lemmas"]) + for lemma in record["lemmas"]: + by_key[lookup_key(lemma)].append(record) + for key, records in by_key.items(): + unique = {item["synset"]: item for item in records} + by_key[key] = [unique[name] for name in sorted(unique)] + return by_key + + +def candidate_status(yes_count: int, yes_exact: int) -> str: + if yes_count == 0: + return "CANDIDATE_INSUFFICIENT" + if yes_exact != yes_count: + return "CANDIDATE_REJECTED" + return "CANDIDATE_PROMISING" + + +def main() -> None: + magpie_module = HYPERLEX / "scripts/shadow/hyperlexical/magpie_candidate_evaluation.py" + if not magpie_module.is_file(): + refuse("MAGPIE evaluator is missing") + for path, expected in EXPECTED.items(): + if expected is None: + continue + if sha256(path) != expected: + refuse(f"sealed artifact changed before evaluation: {path}") + readme = (WORDNET / "README").read_text(encoding="utf-8", errors="replace") + if "WordNet 3.0" not in readme: + refuse("local WordNet README does not identify release 3.0") + labels = Counter(label for _item_id, label, _surface in TABLE) + if labels["NONCOMPOSITIONAL"] != 19 or labels["COMPOSITIONAL"] != 19 or len(TABLE) != 38: + refuse("recovered table does not match the published 19/19 split") + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + pdf_mtime = datetime.fromtimestamp(PDF.stat().st_mtime, timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + by_key = lemma_index(WORDNET) + inventory = [] + alignments = [] + for item_id, label, surface in TABLE: + records = by_key.get(lookup_key(surface), []) + synset_ids = [record["synset"] for record in records] + aligned = align_item(label, None, synset_ids) + inventory.append({ + "schema": "hyperlex.km_raw_evaluation_item.v1", + "source_context": None, + "source_item_id": item_id, + "source_label": label, + "source_page_or_location": "P09-2017 Table 1, proceedings page 67", + "surface": surface, + }) + alignments.append({ + "aligned_pwn30_synset": aligned["aligned_pwn30_synset"], + "alignment_status": aligned["alignment_status"], + "candidate_glosses": [record["gloss"] for record in records], + "candidate_lemmas": sorted({lemma for record in records for lemma in record["lemmas"]}), + "candidate_synsets": synset_ids, + "exact_sense_identity_preserved_by_source": False, + "json_schema_document": None, + "lookup_key": lookup_key(surface), + "primary_evidence_code": aligned["primary_evidence_code"], + "schema": "hyperlex.km_pwn30_alignment_row.v1", + "semantic_noncompositional": aligned["semantic_noncompositional"], + "source_gloss": None, + "source_item_id": item_id, + "source_label": label, + "source_lemma_key": None, + "source_pos": None, + "source_sense_key": None, + "source_synset_identifier": None, + "source_synset_offset": None, + "surface": surface, + "wordnet_version_in_source": "3.0", + }) + exact = {} + ambiguous: dict[str, list[str]] = defaultdict(list) + for row in alignments: + if row["alignment_status"] in {"EXACT_SOURCE_ID", "EXACT_UNIQUE_RECONSTRUCTION"}: + synset = row["aligned_pwn30_synset"] + if synset in exact: + refuse("two exact items share one synset") + exact[synset] = row + elif row["alignment_status"] == "AMBIGUOUS_MULTIPLE_SYNSETS": + for synset in row["candidate_synsets"]: + ambiguous[synset].append(row["source_item_id"]) + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + rows = manifest["rows"] + if len(rows) != 225 or any(row.get("sense_class") is not None for row in rows): + refuse("development manifest changed") + semantic_rows = [] + surface_key_hits = [] + inventory_keys = {lookup_key(surface) for _item_id, _label, surface in TABLE} + for row in rows: + synset = f"{row['synset_pos']}:{row['synset_offset']}" + joined = join_synset(synset, exact, ambiguous) + if lookup_key(row["surface"]) in inventory_keys: + surface_key_hits.append(row["row_id"]) + semantic_rows.append({ + "alignment_status": joined["alignment_status"], + "candidate_aligned_synset": joined["candidate_aligned_synset"], + "candidate_source": "KORKONTZELOS_MANANDHAR", + "candidate_source_item_id": joined["candidate_source_item_id"], + "hyperlex_synset": synset, + "json_schema_document": None, + "pos": row["pos"], + "primary_evidence_code": joined["primary_evidence_code"], + "provenance": { + "gloss_used_to_choose_synset": False, + "operator_label_used": False, + "publication": PUBLICATION, + "publication_pdf_sha256": EXPECTED[PDF], + "surface_join": False, + "wordnet_release": "3.0", + }, + "row_id": row["row_id"], + "schema": "hyperlex.km_hyperlex_semantic_evidence_row.v1", + "semantic_noncompositional": joined["semantic_noncompositional"], + "surface": row["surface"], + }) + alignment_counts = Counter(row["alignment_status"] for row in alignments) + label_counts = Counter(row["source_label"] for row in alignments) + inventory_sha = write_jsonl(INVENTORY_PATH, inventory) + alignment_sha = write_jsonl(ALIGNMENT_PATH, alignments) + semantic_sha = write_jsonl(SEMANTIC_PATH, semantic_rows) + frozen = [json.loads(line) for line in SEMANTIC_PATH.read_text(encoding="utf-8").splitlines() if line] + if any("operator_bucket" in row for row in frozen): + refuse("operator label entered semantic evidence") + by_operator = {name: Counter() for name in OPERATORS} + semantic_counts = Counter() + code_counts = Counter() + false_yes = [] + false_no = [] + for manifest_row, evidence_row in zip(rows, frozen): + operator = manifest_row["operator_bucket"] + semantic = evidence_row["semantic_noncompositional"] + if operator not in by_operator: + refuse(f"unexpected operator bucket: {operator}") + by_operator[operator][semantic] += 1 + semantic_counts[semantic] += 1 + code_counts[evidence_row["primary_evidence_code"]] += 1 + if semantic == "YES" and operator != "HIGH": + false_yes.append(evidence_row["row_id"]) + if semantic == "NO" and operator == "HIGH": + false_no.append(evidence_row["row_id"]) + yes_count = semantic_counts["YES"] + no_count = semantic_counts["NO"] + unknown_count = semantic_counts["UNKNOWN"] + yes_exact = sum(1 for row in frozen if row["semantic_noncompositional"] == "YES" and row["alignment_status"] in {"EXACT_SOURCE_ID", "EXACT_UNIQUE_RECONSTRUCTION"}) + sense_covered = sum(1 for row in frozen if row["alignment_status"] in {"EXACT_SOURCE_ID", "EXACT_UNIQUE_RECONSTRUCTION"}) + polysemous = alignment_counts["AMBIGUOUS_MULTIPLE_SYNSETS"] + unique = alignment_counts["EXACT_UNIQUE_RECONSTRUCTION"] + identity_not_preserved = unique == 0 and polysemous > (len(TABLE) / 2) + status = "CANDIDATE_INSUFFICIENT" if identity_not_preserved else candidate_status(yes_count, yes_exact) + provenance = { + "acquisition_method": "HTTPS GET of the ACL Anthology PDF", + "acquisition_timestamp_utc": pdf_mtime, + "acquisition_url": "https://aclanthology.org/P09-2017.pdf", + "authors": ["Ioannis Korkontzelos", "Suresh Manandhar"], + "canonical_publication_url": "https://aclanthology.org/P09-2017/", + "evaluated_at": when, + "historical_count_hint_not_used": {"compositional": 60, "noncompositional": 56}, + "json_schema_document": None, + "later_naacl_2010_sample_acquired": False, + "publication": "Detecting Compositionality in Multi-Word Expressions", + "publication_year": 2009, + "schema": "hyperlex.km_source_provenance.v1", + "source_artifact": "P09-2017.pdf", + "source_artifact_sha256": EXPECTED[PDF], + "source_name": "Korkontzelos-Manandhar WordNet-derived MWE compositionality evaluation set", + "source_preserves_synset_identifier": False, + "table": "Table 1, proceedings pages 65-68, table printed on page 67", + "venue": "ACL-IJCNLP 2009 short papers", + "verified_inventory_counts": {"COMPOSITIONAL": 19, "NONCOMPOSITIONAL": 19, "items": 38}, + "version": "ACL Anthology P09-2017", + "wordnet_data_sha256": { + "data.adj": EXPECTED[WORDNET / "data.adj"], + "data.adv": EXPECTED[WORDNET / "data.adv"], + "data.noun": EXPECTED[WORDNET / "data.noun"], + "data.verb": EXPECTED[WORDNET / "data.verb"], + }, + "wordnet_release_used_for_reconstruction": "Princeton WordNet 3.0", + "wordnet_version_stated_in_paper": "3.0", + } + receipt = { + "anthology_distribution_license": "CC-BY-NC-SA-3.0", + "anthology_distribution_license_basis": "ACL Anthology footer: materials prior to 2016 are licensed under CC-BY-NC-SA 3.0, and permission is granted to make copies for teaching and research.", + "code_license": None, + "code_license_note": "No code artifact was published with Table 1 and none was acquired.", + "data_license": None, + "data_license_note": "No separate dataset license exists. The evaluation list is Table 1 of the paper.", + "json_schema_document": None, + "license_status": "ESTABLISHED_FOR_RESEARCH_EXTRACTION", + "pdf_copyright_notice": "c\u00a92009 ACL and AFNLP", + "publication_license": "CC-BY-NC-SA-3.0", + "schema": "hyperlex.km_license_receipt.v1", + "source_name": "KORKONTZELOS_MANANDHAR", + "source_version": "P09-2017", + "used": "Table 1 surfaces and the two section labels only", + "not_used_from_table_typography": "Bold, underline, and italic marks report system detections, not gold labels.", + } + provenance_sha = write_json(PROVENANCE_PATH, provenance) + receipt_sha = write_json(LICENSE_PATH, receipt) + evaluation = { + "abstention_count": unknown_count, + "abstention_rate_fraction": fraction(unknown_count, 225), + "alignment_counts": {name: alignment_counts[name] for name in ALIGNMENT_STATES}, + "candidate": "KORKONTZELOS_MANANDHAR", + "candidate_inventory_size": 38, + "compositional_source_count": label_counts["COMPOSITIONAL"], + "development_manifest_sha256": EXPECTED[EVIDENCE_MANIFEST], + "development_rows": 225, + "diagnostics_are_descriptive": True, + "evidence_code_counts": dict(sorted(code_counts.items())), + "false_no_count": len(false_no), + "false_yes_count": len(false_yes), + "inventory_sha256": inventory_sha, + "json_schema_document": None, + "mapping_assumptions": { + "false_no": "A semantic NO whose operator bucket is HIGH. Operator HIGH is not gold semantic YES.", + "false_yes": "A semantic YES whose operator bucket is not HIGH. Operator REJECT is not compositional evidence.", + "no_precision": "Among semantic NO rows, the fraction whose operator bucket is SECONDARY. SECONDARY is not defined as compositional NO.", + "operator_quarantine": "Quarantine is not compositionality evidence.", + "operator_reject": "Reject stays in its own joint-table row. It is a different axis from compositionality.", + "yes_precision": "Among semantic YES rows, the fraction whose operator bucket is HIGH. The correlation is a proxy, not an identity.", + }, + "no_precision": "NOT_COMPUTABLE" if no_count == 0 else fraction(by_operator["SECONDARY"]["NO"], no_count), + "no_support": no_count, + "noncompositional_source_count": label_counts["NONCOMPOSITIONAL"], + "operator_by_semantic": { + name: {state: by_operator[name][state] for state in SEMANTIC_STATES} + for name in OPERATORS + }, + "operator_labels_joined_after_semantic_evidence_was_written": True, + "operator_labels_used_as_runtime_evidence": False, + "pwn30_alignment_sha256": alignment_sha, + "schema": "hyperlex.km_development_evaluation.v1", + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_aligned_coverage_count": sense_covered, + "sense_aligned_coverage_fraction": fraction(sense_covered, 225), + "source_version": "P09-2017", + "surface_key_overlap_count": len(surface_key_hits), + "surface_key_overlap_not_used_as_join": True, + "unknown_support": unknown_count, + "yes_precision": "NOT_COMPUTABLE" if yes_count == 0 else fraction(by_operator["HIGH"]["YES"], yes_count), + "yes_support": yes_count, + } + evaluation_sha = write_json(EVALUATION_PATH, evaluation) + magpie_path = SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json" + magpie = json.loads(magpie_path.read_text(encoding="utf-8")) + comparison = { + "json_schema_document": None, + "korkontzelos_manandhar": { + "abstention": evaluation["abstention_rate_fraction"], + "no_support": no_count, + "sense_aligned_coverage": evaluation["sense_aligned_coverage_fraction"], + "surface_coverage": fraction(len(surface_key_hits), 225), + "target_quality": "Human compositional versus noncompositional labels on a WordNet 3.0 MWE sample. The paper stores no synset id. A unique PWN 3.0 lemma reconstructs one synset for monosemous items.", + "yes_support": yes_count, + }, + "magpie": { + "abstention": magpie["abstention_rate_fraction"], + "development_evaluation_sha256": EXPECTED[magpie_path], + "no_support": magpie["no_support"], + "sense_aligned_coverage": magpie["sense_alignment_coverage_fraction"], + "surface_coverage": magpie["surface_coverage_fraction"], + "target_quality": "Contextual literal versus idiomatic labels. The artifact stores no WordNet sense identifier.", + "yes_support": magpie["yes_support"], + }, + "purpose": "Compare sense-aligned support. Raw surface coverage is not the ranking.", + "schema": "hyperlex.km_magpie_comparison.v1", + "sense_alignment_barrier": { + "korkontzelos_manandhar_inventory_unique_reconstructions": unique, + "korkontzelos_manandhar_hyperlex_sense_aligned_rows": sense_covered, + "magpie_hyperlex_sense_aligned_rows": 0, + }, + } + comparison_sha = write_json(COMPARISON_PATH, comparison) + decision = { + "candidate": "KORKONTZELOS_MANANDHAR", + "comparison_sha256": comparison_sha, + "evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "inventory_sha256": inventory_sha, + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + "next_transition_authorized": False, + "pwn30_alignment_sha256": alignment_sha, + "readiness_question_met": yes_count > 0 and yes_exact == yes_count, + "runtime_integration": False, + "schema": "hyperlex.km_candidate_decision.v1", + "select_005_authorized": False, + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_identity_not_preserved_finding": identity_not_preserved, + "sense_identity_not_preserved_finding_name": "WORDNET_DERIVED_BUT_SENSE_IDENTITY_NOT_PRESERVED" if identity_not_preserved else None, + "source_provenance_sha256": provenance_sha, + "source_version": "P09-2017", + "state": "CANDIDATE_SOURCE_EVALUATED", + "why": "Table 1 stores surfaces and labels, not synset ids. Unique PWN 3.0 lemmas reconstruct 30 synsets, and 8 items are polysemous and stay UNKNOWN. None of the reconstructed synsets occur in the 225 development rows, so Hyperlex YES support is 0.", + } + decision_sha = write_json(DECISION_PATH, decision) + for path, expected in EXPECTED.items(): + if path == TRACKER or expected is None: + continue + if sha256(path) != expected: + refuse(f"evaluation mutated {path}") + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["evaluated_candidates"] = ["MAGPIE", "KORKONTZELOS_MANANDHAR"] + tracker["magpie_evaluation_status"] = "CANDIDATE_INSUFFICIENT" + tracker["km_evaluation_status"] = status + tracker["latest_evaluated_candidate"] = "KORKONTZELOS_MANANDHAR" + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["km_source_provenance_sha256"] = provenance_sha + tracker["km_license_receipt_sha256"] = receipt_sha + tracker["km_raw_inventory_sha256"] = inventory_sha + tracker["km_pwn30_alignment_sha256"] = alignment_sha + tracker["km_hyperlex_semantic_evidence_sha256"] = semantic_sha + tracker["km_development_evaluation_sha256"] = evaluation_sha + tracker["km_magpie_comparison_sha256"] = comparison_sha + tracker["km_candidate_decision_sha256"] = decision_sha + tracker["km_source_version"] = "P09-2017" + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION" + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + for path, expected in EXPECTED.items(): + if path == TRACKER or expected is None: + continue + if sha256(path) != expected: + refuse(f"tracker update mutated {path}") + print(json.dumps({ + "alignment_counts": evaluation["alignment_counts"], + "candidate_decision_sha256": decision_sha, + "comparison_sha256": comparison_sha, + "development_evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "inventory_sha256": inventory_sha, + "license_receipt_sha256": receipt_sha, + "no_support": no_count, + "operator_by_semantic": evaluation["operator_by_semantic"], + "pwn30_alignment_sha256": alignment_sha, + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_aligned_coverage_fraction": evaluation["sense_aligned_coverage_fraction"], + "sense_identity_not_preserved_finding": identity_not_preserved, + "source_provenance_sha256": provenance_sha, + "tracker_sha256": tracker_sha, + "unknown_support": unknown_count, + "yes_support": yes_count, + "label_counts": dict(label_counts), + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py b/scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py new file mode 100644 index 00000000..30c64e18 --- /dev/null +++ b/scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py @@ -0,0 +1,576 @@ +"""Evaluate the pinned MAGPIE author corpus on the 225 development rows. + +This is candidate-source evidence research. It does not select MAGPIE, +integrate it, draw a measurement sample, or authorize SELECT-005. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from datetime import datetime, timezone +from decimal import Decimal +from pathlib import Path + +from hyperlexical.magpie_candidate_evaluation import build_index, evaluate_row + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +MAGPIE = ( + LEDGER + / "acquisition/sources/magpie-corpus/7fa677b82b9a772dfa54bbdd0fb414412d73db3b" +) +COMMIT = "7fa677b82b9a772dfa54bbdd0fb414412d73db3b" +UNFILTERED = MAGPIE / "MAGPIE_unfiltered.jsonl" +LICENSE = MAGPIE / "LICENSE" +README = MAGPIE / "README.md" +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +PROVENANCE_PATH = SOURCE / "MAGPIE_SOURCE_PROVENANCE.json" +LICENSE_PATH = SOURCE / "MAGPIE_LICENSE_RECEIPT.json" +MATCHES_PATH = SOURCE / "MAGPIE_DEVELOPMENT_MATCHES.jsonl" +ALIGNMENT_PATH = SOURCE / "MAGPIE_SENSE_ALIGNMENT.jsonl" +SEMANTIC_PATH = SOURCE / "MAGPIE_SEMANTIC_EVIDENCE.jsonl" +EVALUATION_PATH = SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json" +DECISION_PATH = SOURCE / "MAGPIE_CANDIDATE_DECISION.json" + +EXPECTED = { + SENSE / "CLASSIFICATION_PROCEDURE.json": "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + SENSE / "CLASSIFICATION_PROCEDURE.v2.json": "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + SENSE / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + SENSE / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE_MANIFEST: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + SENSE / "development_replay_predictions.jsonl": "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + SENSE / "development_replay_report.json": "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + SENSE / "development_replay_v2_predictions.jsonl": "1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a", + SENSE / "development_replay_v2_report.json": "93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5", + SENSE / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + SENSE / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + SENSE / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + SENSE / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + SENSE / "V2_DEVELOPMENT_RESULT_REVIEW.json": "77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d", + SENSE / "WORDNET_STRUCTURAL_SOURCE_LIMITATION.json": "3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a", + SOURCE / "HYPOTHESIS.json": "39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787", + SOURCE / "ACCEPTANCE.json": "1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94", + SOURCE / "CANDIDATE_SOURCE_EVALUATION_PLAN.json": "472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a", + SOURCE / "SEMANTIC_COMPOSITIONALITY.architecture.json": "180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36", + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7/measurement_error_analysis.json": "ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8", + EVENTS: "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + LEDGER_FILE: "77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0", + TRACKER: "521eccb8bd068d4697a3ffdc06e3b44feb4eec3c8a3bd6aebcea11dd288b7dfc", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v2.py": "4b6f125da435b365af143c187902093bee5c9502db5b813a3d2bba11f889fa2b", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", + UNFILTERED: "541ee535e93d71eff85351351665115e2a9f22ad736423881da5774a93bc880e", + LICENSE: "05ab88f3f9da1d05f9c5bf0a7c45c49a9007f877dd9c237a5bf668276fe04c3b", + README: "bc9d281e2348a780b91929e44de0e58bc26a822d69835b0e6dd8a666754af6db", +} +BLOB = { + UNFILTERED: "24a6ef5cc6b226859956a903395caddb81c09493", + LICENSE: "56902ed9fa665f4aeebdb8cc3df5743a6de02542", + README: "f20506c021aa1484ea3ece67dedde20c202bbf53", +} +EXPECTED_INSTANCES = 56622 +EXPECTED_TYPES = 1756 +EXPECTED_LABELS = {"?": 7, "i": 40011, "l": 16168, "o": 436} +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +SEMANTIC_STATES = ("YES", "NO", "UNKNOWN") +SURFACE_STATES = ("EXACT", "NORMALIZED", "VARIANT", "NONE", "AMBIGUOUS") +ALIGNMENTS = ("ALIGNED_IDIOMATIC", "ALIGNED_LITERAL", "MIXED", "CONFLICT", "UNKNOWN") + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def git_blob_sha1(path: Path) -> str: + data = path.read_bytes() + return hashlib.sha1(f"blob {len(data)}\0".encode() + data).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def iso_mtime(path: Path) -> str: + stamp = datetime.fromtimestamp(path.stat().st_mtime, timezone.utc) + return stamp.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str: + return f"{numerator}/{denominator}" + + +def load_instances(path: Path) -> tuple[list[dict], Counter]: + rows = [] + labels: Counter[str] = Counter() + with path.open(encoding="utf-8") as handle: + for line in handle: + if not line.strip(): + continue + instance = json.loads(line, parse_float=Decimal) + rows.append(instance) + labels[str(instance.get("label"))] += 1 + return rows, labels + + +def license_established(text: str) -> bool: + if "Attribution 4.0 International" not in text: + return False + if "NonCommercial" in text or "Non-Commercial" in text: + return False + return True + + +def project_match(row: dict, found: dict) -> dict: + return { + "ambiguous_candidates": found["ambiguous_candidates"], + "annotation_confidence": found["annotation_confidence"], + "idiomatic_instance_count": found["idiomatic_instance_count"], + "literal_instance_count": found["literal_instance_count"], + "matched_magpie_expression": found["matched_magpie_expression"], + "pos": row["pos"], + "row_id": row["row_id"], + "schema": "hyperlex.magpie_development_match_row.v1", + "source_instance_ids": found["source_instance_ids"], + "source_name": "MAGPIE", + "surface": row["surface"], + "surface_match": found["surface_match"], + "synset": found["synset"], + "unresolved_instance_count": found["unresolved_instance_count"], + "variant_types": found["variant_types"], + } + + +def project_alignment(row: dict, found: dict) -> dict: + return { + "alignment_basis": found["alignment_basis"], + "gloss_used": False, + "json_schema_document": None, + "matched_magpie_expression": found["matched_magpie_expression"], + "operator_label_used": False, + "pos": row["pos"], + "row_id": row["row_id"], + "schema": "hyperlex.magpie_sense_alignment_row.v1", + "sense_alignment": found["sense_alignment"], + "sense_alignment_rule": "identifier_equality_only", + "surface": row["surface"], + "surface_match": found["surface_match"], + "synset": found["synset"], + } + + +def project_evidence(row: dict, found: dict, provenance: dict) -> dict: + return { + "idiomatic_instance_count": found["idiomatic_instance_count"], + "json_schema_document": None, + "literal_instance_count": found["literal_instance_count"], + "matched_magpie_expression": found["matched_magpie_expression"], + "pos": row["pos"], + "primary_evidence_code": found["primary_evidence_code"], + "provenance": provenance, + "row_id": row["row_id"], + "schema": "hyperlex.magpie_semantic_evidence_row.v1", + "semantic_noncompositional": found["semantic_noncompositional"], + "sense_alignment": found["sense_alignment"], + "source_artifact_hash": found["source_artifact_hash"], + "source_instance_ids": found["source_instance_ids"], + "source_name": "MAGPIE", + "source_version": found["source_version"], + "surface": row["surface"], + "surface_match": found["surface_match"], + "synset": found["synset"], + "unresolved_instance_count": found["unresolved_instance_count"], + } + + +def candidate_status(yes_count: int, yes_aligned: int) -> str: + if yes_count == 0: + return "CANDIDATE_INSUFFICIENT" + if yes_aligned != yes_count: + return "CANDIDATE_REJECTED" + return "CANDIDATE_PROMISING" + + +def main() -> None: + for path, expected in EXPECTED.items(): + if sha256(path) != expected: + refuse(f"sealed artifact changed before evaluation: {path}") + for path, expected in BLOB.items(): + if git_blob_sha1(path) != expected: + refuse(f"git blob does not match hslh/magpie-corpus: {path.name}") + license_text = LICENSE.read_text(encoding="utf-8") + established = license_established(license_text) + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + artifact_hash = EXPECTED[UNFILTERED] + publication_pdf_sha = "8247c926909772ce317d5f33cddb83caa51969eb5ef928bcbadbbcf05e39c979" + provenance = { + "acquisition_timestamp_utc": { + "LICENSE": iso_mtime(LICENSE), + "MAGPIE_unfiltered.jsonl": iso_mtime(UNFILTERED), + "README.md": iso_mtime(README), + }, + "acquisition_urls": { + "LICENSE": f"https://raw.githubusercontent.com/hslh/magpie-corpus/{COMMIT}/LICENSE", + "MAGPIE_unfiltered.jsonl": f"https://raw.githubusercontent.com/hslh/magpie-corpus/{COMMIT}/MAGPIE_unfiltered.jsonl", + "README.md": f"https://raw.githubusercontent.com/hslh/magpie-corpus/{COMMIT}/README.md", + }, + "artifact_filenames": ["LICENSE", "MAGPIE_unfiltered.jsonl", "README.md"], + "artifact_hashes_sha256": { + "LICENSE": EXPECTED[LICENSE], + "MAGPIE_unfiltered.jsonl": artifact_hash, + "README.md": EXPECTED[README], + }, + "canonical_source_repository": "https://github.com/hslh/magpie-corpus", + "dataset_variant": "author corpus, MAGPIE_unfiltered.jsonl", + "filtered_splits_acquired": False, + "filtered_splits_not_used": { + "MAGPIE_filtered_split_random.jsonl": { + "acquired": False, + "git_blob_sha1": "59c24bffbccb38d912930d761abc0e8d7846e128", + "sha256": None, + }, + "MAGPIE_filtered_split_typebased.jsonl": { + "acquired": False, + "git_blob_sha1": "722546a2df509b5f50556445fdabb13179019b85", + "sha256": None, + }, + }, + "git_blob_sha1": {path.name: digest for path, digest in BLOB.items()}, + "hugging_face_package_used": False, + "publication": { + "anthology_url": "https://aclanthology.org/2020.lrec-1.35/", + "authors": ["Hessel Haagsma", "Johan Bos", "Malvina Nissim"], + "pdf_sha256": publication_pdf_sha, + "pdf_url": "https://aclanthology.org/2020.lrec-1.35.pdf", + "title": "MAGPIE: A Large Corpus of Potentially Idiomatic Expressions", + "venue": "LREC 2020", + }, + "repository_full_name": "hslh/magpie-corpus", + "schema": "hyperlex.magpie_source_provenance.v1", + "source_name": "MAGPIE — A Large Corpus of Potentially Idiomatic Expressions", + "unrelated_magpie_repository_used": False, + "version_or_commit": COMMIT, + "version_recorded_at": "2020-06-07T09:57:12Z", + } + receipt = { + "code_license": "CC-BY-4.0", + "code_license_note": "The pinned commit has one LICENSE file and no separate software license.", + "dataset_artifact_license": "CC-BY-4.0" if established else "SOURCE_LICENSE_UNRESOLVED", + "dataset_license_basis": "LICENSE file in hslh/magpie-corpus at the pinned commit, corroborated by the GitHub license API SPDX CC-BY-4.0. The jsonl itself has no license field.", + "evaluated_artifact": "MAGPIE_unfiltered.jsonl", + "hugging_face_dataset_card_consulted": False, + "json_schema_document": None, + "license_status": "ESTABLISHED" if established else "SOURCE_LICENSE_UNRESOLVED", + "publication_license": "CC-BY-NC", + "publication_license_basis": "Page 1 of the anthology PDF states that the ELRA proceedings text is licensed under CC-BY-NC. That statement is not the dataset license.", + "publication_pdf_sha256": publication_pdf_sha, + "schema": "hyperlex.magpie_license_receipt.v1", + "source_name": "MAGPIE", + "source_version": COMMIT, + } + if not established: + provenance_sha = write_json(PROVENANCE_PATH, provenance) + receipt_sha = write_json(LICENSE_PATH, receipt) + decision = { + "candidate": "MAGPIE", + "evaluation_status": "SOURCE_LICENSE_UNRESOLVED", + "license_receipt_sha256": receipt_sha, + "runtime_integration": False, + "schema": "hyperlex.magpie_candidate_decision.v1", + "selected_source": "none", + "source_provenance_sha256": provenance_sha, + } + write_json(DECISION_PATH, decision) + refuse("SOURCE_LICENSE_UNRESOLVED") + + instances, labels = load_instances(UNFILTERED) + if labels != Counter(EXPECTED_LABELS): + refuse(f"label census does not match the pinned corpus: {dict(labels)}") + index = build_index(instances) + if index.instance_count != EXPECTED_INSTANCES or index.type_count != EXPECTED_TYPES: + refuse("instance or type count does not match the pinned corpus") + if index.records_bound_sense: + refuse("pinned corpus unexpectedly carries a sense identifier") + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + rows = manifest["rows"] + if len(rows) != 225: + refuse("development manifest is not 225 rows") + if any(row.get("sense_class") is not None for row in rows): + refuse("development manifest already has a sense class") + if manifest.get("sense_classes_assigned") is not False: + refuse("development manifest sense_classes_assigned flag changed") + + row_provenance = { + "alignment_basis_field": "alignment_basis", + "artifact": "MAGPIE_unfiltered.jsonl", + "artifact_sha256": artifact_hash, + "commit": COMMIT, + "gloss_used": False, + "operator_label_used": False, + "repository": "https://github.com/hslh/magpie-corpus", + "sense_alignment_rule": "identifier_equality_only", + "surface_match_assigns_semantic_yes": False, + "variant_type_is_not_an_alternate_expression": True, + } + matches = [] + alignments = [] + semantic_rows = [] + for row in rows: + synset = f"{row['synset_pos']}:{row['synset_offset']}" + found = evaluate_row(row["surface"], synset, index, COMMIT, artifact_hash) + found["synset"] = synset + item_provenance = dict(row_provenance) + item_provenance["alignment_basis"] = found["alignment_basis"] + matches.append(project_match(row, found)) + alignments.append(project_alignment(row, found)) + semantic_rows.append(project_evidence(row, found, item_provenance)) + + provenance["evaluated_at"] = when + provenance["instance_count"] = index.instance_count + provenance["label_counts"] = dict(sorted(labels.items())) + provenance["records_bound_sense_identifier"] = False + provenance["type_count"] = index.type_count + provenance["variant_match_available"] = False + provenance_sha = write_json(PROVENANCE_PATH, provenance) + receipt_sha = write_json(LICENSE_PATH, receipt) + matches_sha = write_jsonl(MATCHES_PATH, matches) + alignment_sha = write_jsonl(ALIGNMENT_PATH, alignments) + semantic_sha = write_jsonl(SEMANTIC_PATH, semantic_rows) + + frozen_semantic = [ + json.loads(line) + for line in SEMANTIC_PATH.read_text(encoding="utf-8").splitlines() + if line + ] + if [row["row_id"] for row in frozen_semantic] != [row["row_id"] for row in rows]: + refuse("semantic evidence row order drifted") + if any("operator_bucket" in row for row in frozen_semantic): + refuse("operator label entered the semantic evidence") + + by_operator = {name: Counter() for name in OPERATORS} + semantic_counts = Counter() + surface_counts = Counter() + alignment_counts = Counter() + code_counts = Counter() + false_yes = [] + false_no = [] + mixed_usage_rows = [] + mixed_expressions = set() + sense_conflict_rows = [] + for manifest_row, evidence_row in zip(rows, frozen_semantic): + operator = manifest_row["operator_bucket"] + semantic = evidence_row["semantic_noncompositional"] + if operator not in by_operator: + refuse(f"unexpected operator bucket: {operator}") + by_operator[operator][semantic] += 1 + semantic_counts[semantic] += 1 + surface_counts[evidence_row["surface_match"]] += 1 + alignment_counts[evidence_row["sense_alignment"]] += 1 + code_counts[evidence_row["primary_evidence_code"]] += 1 + if semantic == "YES" and operator != "HIGH": + false_yes.append(evidence_row["row_id"]) + if semantic == "NO" and operator == "HIGH": + false_no.append(evidence_row["row_id"]) + literal = evidence_row["literal_instance_count"] + idiomatic = evidence_row["idiomatic_instance_count"] + if literal and idiomatic: + mixed_usage_rows.append(evidence_row["row_id"]) + mixed_expressions.add(evidence_row["matched_magpie_expression"]) + if evidence_row["sense_alignment"] == "CONFLICT": + sense_conflict_rows.append(evidence_row["row_id"]) + + yes_count = semantic_counts["YES"] + no_count = semantic_counts["NO"] + unknown_count = semantic_counts["UNKNOWN"] + yes_aligned = sum( + 1 + for row in frozen_semantic + if row["semantic_noncompositional"] == "YES" and row["sense_alignment"] == "ALIGNED_IDIOMATIC" + ) + covered = surface_counts["EXACT"] + surface_counts["NORMALIZED"] + surface_counts["VARIANT"] + sense_covered = sum(alignment_counts[name] for name in ALIGNMENTS if name != "UNKNOWN") + status = candidate_status(yes_count, yes_aligned) + mapping_assumptions = { + "false_no": "A semantic NO whose operator bucket is HIGH. This is a descriptive disagreement under an explicit proxy. Operator HIGH is not gold semantic YES.", + "false_yes": "A semantic YES whose operator bucket is not HIGH. This is a descriptive disagreement under an explicit proxy. It is not computed by treating REJECT as compositional.", + "no_precision": "Among semantic NO rows, the fraction whose operator bucket is SECONDARY. SECONDARY is not defined as compositional NO.", + "operator_reject": "Operator REJECT is reported as its own row in the joint table. It is not mapped to semantic YES or NO.", + "yes_precision": "Among semantic YES rows, the fraction whose operator bucket is HIGH. Operator HIGH is not defined as semantic noncompositionality.", + } + evaluation = { + "abstention_count": unknown_count, + "abstention_rate_fraction": fraction(unknown_count, 225), + "candidate": "MAGPIE", + "coverage_is_not_success": True, + "development_manifest_sha256": EXPECTED[EVIDENCE_MANIFEST], + "development_rows": 225, + "diagnostics_are_descriptive": True, + "evidence_code_counts": dict(sorted(code_counts.items())), + "false_no_count": len(false_no), + "false_no_row_ids": false_no, + "false_yes_count": len(false_yes), + "false_yes_row_ids": false_yes, + "high_unknown_rate_is_acceptable": True, + "json_schema_document": None, + "mapping_assumptions": mapping_assumptions, + "mixed_usage_expression_count": len(mixed_expressions), + "mixed_usage_is_not_sense_alignment_mixed": True, + "mixed_usage_row_count": len(mixed_usage_rows), + "no_precision": "NOT_COMPUTABLE" if no_count == 0 else fraction(by_operator["SECONDARY"]["NO"], no_count), + "no_support": no_count, + "operator_by_semantic": { + name: {state: by_operator[name][state] for state in SEMANTIC_STATES} + for name in OPERATORS + }, + "operator_labels_joined_after_semantic_evidence_was_written": True, + "operator_labels_used_as_runtime_evidence": False, + "schema": "hyperlex.magpie_development_evaluation.v1", + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_alignment_counts": {name: alignment_counts[name] for name in ALIGNMENTS}, + "sense_alignment_coverage_count": sense_covered, + "sense_alignment_coverage_fraction": fraction(sense_covered, 225), + "sense_conflict_count": len(sense_conflict_rows), + "source_artifact_hash": artifact_hash, + "source_version": COMMIT, + "surface_ambiguous_count": surface_counts["AMBIGUOUS"], + "surface_coverage_count": covered, + "surface_coverage_excludes_ambiguous": True, + "surface_coverage_fraction": fraction(covered, 225), + "surface_match_counts": {name: surface_counts[name] for name in SURFACE_STATES}, + "surface_none_count": surface_counts["NONE"], + "unknown_support": unknown_count, + "yes_precision": "NOT_COMPUTABLE" if yes_count == 0 else fraction(by_operator["HIGH"]["YES"], yes_count), + "yes_support": yes_count, + "yes_support_with_aligned_idiomatic": yes_aligned, + } + evaluation_sha = write_json(EVALUATION_PATH, evaluation) + decision = { + "candidate": "MAGPIE", + "candidate_family": "C_curated_linguistic_resource", + "coverage_was_not_treated_as_success": True, + "evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + "next_transition_authorized": False, + "not_selected_reason": "Semantic YES support is zero. MAGPIE annotates idiomatic and literal uses of a potentially idiomatic expression. The pinned unfiltered artifact has no WordNet synset, sense key, or gloss, so the supplied Hyperlex sense cannot be identified by equality. Surface coverage is not sense alignment.", + "readiness_question_met": False, + "recommended_next_candidate": "A later authorization may name one curated resource whose entries carry a WordNet synset offset or sense key for the expression sense. PARSEME, STREUSLE, PIE, EPIE, and NCS were not downloaded and are not selected.", + "runtime_integration": False, + "schema": "hyperlex.magpie_candidate_decision.v1", + "select_005_authorized": False, + "selected_source": "none", + "sense_alignment_sha256": alignment_sha, + "source_provenance_sha256": provenance_sha, + "source_version": COMMIT, + "state": "CANDIDATE_SOURCE_EVALUATED", + } + decision_sha = write_json(DECISION_PATH, decision) + + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("evaluation wrote into the development manifest") + reloaded = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if any(row.get("sense_class") is not None for row in reloaded["rows"]): + refuse("evaluation wrote a sense class") + for path, expected in EXPECTED.items(): + if path == TRACKER: + continue + if sha256(path) != expected: + refuse(f"evaluation mutated {path}") + + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = "SEMANTIC_EVIDENCE_SOURCE_SPEC_FROZEN" + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_candidate"] = "MAGPIE" + tracker["semantic_evidence_source_evaluation_status"] = status + tracker["semantic_evidence_source_selected"] = "none" + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["magpie_source_provenance_sha256"] = provenance_sha + tracker["magpie_license_receipt_sha256"] = receipt_sha + tracker["magpie_development_matches_sha256"] = matches_sha + tracker["magpie_sense_alignment_sha256"] = alignment_sha + tracker["magpie_semantic_evidence_sha256"] = semantic_sha + tracker["magpie_development_evaluation_sha256"] = evaluation_sha + tracker["magpie_candidate_decision_sha256"] = decision_sha + tracker["magpie_source_version"] = COMMIT + tracker["magpie_source_artifact_sha256"] = artifact_hash + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION" + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + for path, expected in EXPECTED.items(): + if path == TRACKER: + continue + if sha256(path) != expected: + refuse(f"tracker update mutated {path}") + print(json.dumps({ + "candidate_decision_sha256": decision_sha, + "development_evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "license_receipt_sha256": receipt_sha, + "no_support": no_count, + "operator_by_semantic": evaluation["operator_by_semantic"], + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_alignment_coverage_fraction": evaluation["sense_alignment_coverage_fraction"], + "sense_alignment_sha256": alignment_sha, + "source_provenance_sha256": provenance_sha, + "surface_coverage_fraction": evaluation["surface_coverage_fraction"], + "surface_match_counts": evaluation["surface_match_counts"], + "tracker_sha256": tracker_sha, + "unknown_support": unknown_count, + "yes_support": yes_count, + "false_no_count": len(false_no), + "false_yes_count": len(false_yes), + "mixed_usage_expression_count": len(mixed_expressions), + "mixed_usage_row_count": len(mixed_usage_rows), + "evidence_code_counts": evaluation["evidence_code_counts"], + "matches_sha256": matches_sha, + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py new file mode 100644 index 00000000..8259c276 --- /dev/null +++ b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py @@ -0,0 +1,498 @@ +"""Tier-3 model decision for constituent senses that Extended Lesk left tied. + +The function selects only among supplied PWN 3.0 candidates, or it abstains. +It does not see operator labels, residual scores, or a residual embedding. +""" + +from __future__ import annotations + +from decimal import Decimal, ROUND_HALF_EVEN + +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + STRUCTURAL_TOKENS, + extract_constituents, + neighbor_keys, +) + +RULE_VERSION = "MODEL_BASED_WSD_CANDIDATE_V1" +MODEL_NAME = "kanishka/GlossBERT" +MODEL_FAMILY = "GlossBERT" +MODEL_REVISION = "0cc3b83af5496e27ebcc95ef0cf37ea0a9281a7a" +CANONICAL_SOURCE = "https://huggingface.co/kanishka/GlossBERT" +LICENSE_NAME = "MIT" +POSITIVE_CLASS_INDEX = 1 +MAX_TOKENS = 512 +SCORE_QUANTUM = Decimal("0.000001") +MINIMUM_CONFIDENCE = Decimal("0.50") +MINIMUM_MARGIN = Decimal("0.10") +BASELINE_HIGH_READY = 4 +BASELINE_SECONDARY_READY = 4 +BASELINE_TOTAL_READY = 20 +RESOLVED = "RESOLVED" +AMBIGUOUS = "AMBIGUOUS" +INVALID = "INVALID" +ERROR = "ERROR" +EXACT = "EXACT" +LESK_RESOLVED = "LESK_RESOLVED" +MODEL_RESOLVED = "MODEL_RESOLVED" +UNRESOLVED = "UNRESOLVED" +RESIDUAL_READY = "RESIDUAL_READY" +ROW_UNKNOWN = "UNKNOWN" +IDENTICAL = "IDENTICAL" +PROMISING = "CANDIDATE_PROMISING" +INSUFFICIENT = "CANDIDATE_INSUFFICIENT" +REJECTED = "CANDIDATE_REJECTED" +NOT_DETERMINISTIC = "NOT_DETERMINISTIC" +NOT_COMPUTABLE = "NOT_COMPUTABLE" +LICENSE_UNRESOLVED = "SOURCE_LICENSE_UNRESOLVED" +READY_CONSTITUENT = frozenset({EXACT, LESK_RESOLVED, MODEL_RESOLVED}) +NEXT_PROMISING = "RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION" +NEXT_REVISION = "MODEL_BASED_WSD_CANDIDATE_REVISION_AUTHORIZATION" +NEXT_DETERMINISM = "MODEL_BASED_WSD_DETERMINISM_REVIEW_AUTHORIZATION" +NEXT_RUNTIME = "MODEL_BASED_WSD_RUNTIME_REVIEW_AUTHORIZATION" +NEXT_LICENSE = "MODEL_BASED_WSD_LICENSE_REVIEW_AUTHORIZATION" + +_RESEARCH_QUESTION = ( + "Can one pinned WordNet-native WSD model resolve enough Extended Lesk ties " + "to make a later semantic-residual replay meaningful, without claiming accuracy?" +) +_INPUT_RULE = ( + "Sent-CLS. The context is the parent surface with the target content token in " + "double quotes. The paired text is the matched lemma, a colon, and the first " + "PWN 3.0 gloss clause. The parent gloss is not appended. Structural tokens " + "locate the quoted token and are not a score." +) +_CANDIDATE_RULE = ( + "Candidates are the frozen Extended Lesk synsets for that constituent. " + "Each gloss uses one matching lemma. Sense keys come from the local index.sense." +) +_SCORE_RULE = ( + "The score is the softmax probability of class index 1, quantized to six " + "decimal places, half even. Class index 1 is the gloss-fits class." +) +_SELECTION_RULE = ( + "Select the unique highest score among the candidate synsets. Do not break a " + "tie by synset order. A selected synset must be one of the candidates." +) +_CONFIDENCE_RULE = ( + "Confidence is the quantized top probability. It is recorded when the model " + "returns scores. Overflow and execution errors leave it empty." +) +_ABSTENTION_RULE = ( + "Abstain unless the top probability is at least one half and the top-versus-second " + "margin is at least one tenth. Do not truncate a pair longer than 512 tokens. " + "If any pair overflows, abstain the constituent." +) +_TIE_RULE = ( + "Equal quantized top scores are AMBIGUOUS. Near ties follow the margin rule. " + "No sense is chosen by list order." +) +_MAPPING_RULE = ( + "The checkpoint is SemCor 3.0 and WordNet 3.0. There is no cross-version map. " + "A winner with zero or several matching sense keys is AMBIGUOUS. Do not migrate a sense by hand." +) +_CIRCULARITY_RULE = ( + "Do not choose a constituent sense because a residual embedding is near the " + "parent gloss. This candidate does not read residual scores or residual vectors." +) +_POOL_RULE = ( + "Tier 3 sees only constituents that Extended Lesk v1 left AMBIGUOUS. " + "It does not replace structural exact, unique lemma, or a Lesk decision that cleared its margin." +) +_GATE_RULE = ( + "Promising only when ready HIGH rows, ready SECONDARY rows, and ready rows " + "are each strictly above the frozen lexical baseline, with no invalid output, " + "no execution error, and identical repeats. This is coverage, not accuracy." +) +_READY_RULE = ( + "A row is projected ready only when at least two content constituents exist " + "and every one is exact, Lesk-resolved, or model-resolved." +) +_NO_GOLD_RULE = ( + "There is no independent constituent-sense gold. Do not report accuracy, " + "precision, recall, or F1." +) + + +def candidate_policy() -> dict: + """Return the preregistered model rule. Outcome counts are not included.""" + return { + "abstention": { + "minimum_margin": "0.10", + "minimum_positive_probability": "0.50", + "on_overflow": "ABSTAIN", + "prose": _ABSTENTION_RULE, + "truncate": False, + }, + "applied_to_hyperlex_at_freeze": False, + "candidate_senses": _CANDIDATE_RULE, + "canonical_source": CANONICAL_SOURCE, + "circularity": _CIRCULARITY_RULE, + "confidence": _CONFIDENCE_RULE, + "determinism": { + "batch_size": 1, + "device": "cpu", + "dtype": "float32", + "eval_mode": True, + "inference_mode": True, + "inter_op_threads": 1, + "intra_op_threads": 1, + "manual_seed": 0, + "mkldnn": False, + "score_quantum": "0.000001", + "use_deterministic_algorithms": False, + }, + "device": "cpu", + "dtype": "float32", + "encoded_at_freeze": False, + "evaluation_only": True, + "forbidden_inputs": [ + "idiomaticity_expectation", + "korkontzelos_manandhar_labels", + "magpie_labels", + "measurement_labels", + "operator_labels", + "residual_embeddings", + "residual_scores", + "semantic_yes_no", + "unbind_bucket", + ], + "input_construction": _INPUT_RULE, + "license": LICENSE_NAME, + "max_tokens": MAX_TOKENS, + "model_family": MODEL_FAMILY, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "no_accuracy_claim": _NO_GOLD_RULE, + "no_gold_constituent_senses": True, + "pool": _POOL_RULE, + "positive_class_index": POSITIVE_CLASS_INDEX, + "pwn_mapping": { + "cross_version_map": False, + "manual_migration": False, + "nonunique_sense_key": AMBIGUOUS, + "prose": _MAPPING_RULE, + "source_wordnet": "PWN3.0", + "target_wordnet": "PWN3.0", + }, + "readiness": _READY_RULE, + "readiness_gate": { + "baseline_high_ready": BASELINE_HIGH_READY, + "baseline_secondary_ready": BASELINE_SECONDARY_READY, + "baseline_total_ready": BASELINE_TOTAL_READY, + "comparison": "strictly_greater", + "error_count_must_be": 0, + "invalid_output_count_must_be": 0, + "prose": _GATE_RULE, + "required_determinism": IDENTICAL, + }, + "research_question": _RESEARCH_QUESTION, + "residual_replay_authorized": False, + "rule": RULE_VERSION, + "runtime_integration": False, + "scoring": _SCORE_RULE, + "selected_source": "none", + "selection": _SELECTION_RULE, + "state_at_freeze": "SPEC_FROZEN", + "tie": { + "equal_top_scores": AMBIGUOUS, + "order_break": False, + "prose": _TIE_RULE, + }, + "tokenizer_revision": MODEL_REVISION, + } + + +def format_probability(value: float) -> str: + """Quantize one positive-class probability at the frozen quantum.""" + if value != value or value in {float("inf"), float("-inf")}: + raise RuntimeError("probability is not finite") + number = Decimal(value).quantize(SCORE_QUANTUM, rounding=ROUND_HALF_EVEN) + if number < 0 or number > 1: + raise RuntimeError("probability is outside the unit interval") + return format(number, "f") + + +def quoted_context(surface: str, constituent_index: int, constituent_surface: str) -> str: + """Quote the content token Extended Lesk already indexed. Structural tokens stay bare.""" + extraction = extract_constituents(surface) + content = extraction["content_constituents"] + if constituent_index < 0 or constituent_index >= len(content): + raise RuntimeError("constituent index is outside the content list") + if content[constituent_index] != constituent_surface: + raise RuntimeError("constituent surface does not match the frozen index") + seen = -1 + tokens = [] + for token in extraction["surface_tokens"]: + if lookup_key(token) in STRUCTURAL_TOKENS: + tokens.append(token) + continue + seen += 1 + if seen == constituent_index: + tokens.append('"' + token + '"') + else: + tokens.append(token) + if seen != len(content) - 1: + raise RuntimeError("quoted context did not land on the constituent") + return " ".join(tokens) + + +def gloss_lemma(lemmas: list[str], constituent: str, exceptions: dict[str, set[str]]) -> str | None: + """Pick the matching WordNet lemma. Display underscores as spaces.""" + keys = neighbor_keys(constituent, exceptions) + matched = [lemma for lemma in lemmas if lookup_key(lemma) in keys] + if not matched: + return None + own = lookup_key(constituent) + preferred = [lemma for lemma in matched if lookup_key(lemma) == own] or matched + chosen = sorted(preferred, key=lambda lemma: (lookup_key(lemma), lemma))[0] + return chosen.replace("_", " ") + + +def candidate_gloss_text(lemma: str, gloss_clause: str) -> str: + """Build the gloss side of a Sent-CLS pair.""" + return lemma + ": " + gloss_clause + + +def _decimal_score(text: str) -> Decimal: + value = Decimal(text) + if value != value.quantize(SCORE_QUANTUM): + raise RuntimeError("score is not at the frozen precision") + if value < 0 or value > 1: + raise RuntimeError("score is outside the unit interval") + return value + + +def _blank(status: str, code: str, scores: list[dict], confidence: str | None, margin: str | None) -> dict: + return { + "model_candidate_scores": scores, + "model_confidence": confidence, + "model_margin": margin, + "model_resolution_status": status, + "primary_evidence_code": code, + "selected_sense_key": None, + "selected_synset": None, + } + + +def resolve_model_scores( + candidate_synsets: list[str], + candidate_sense_keys: dict[str, list[str]], + positive_probabilities: dict[str, str] | None, + *, + overflow: bool = False, + error: str | None = None, +) -> dict: + """Apply the frozen abstention rule. The arguments are scores, not labels.""" + candidates = list(candidate_synsets) + if error: + return _blank(ERROR, "model_error", [], None, None) + if overflow: + return _blank(AMBIGUOUS, "context_overflow", [], None, None) + if len(candidates) < 2 or len(set(candidates)) != len(candidates): + return _blank(INVALID, "candidate_inventory_invalid", [], None, None) + if positive_probabilities is None or set(positive_probabilities) != set(candidates): + return _blank(INVALID, "invalid_model_output", [], None, None) + scored = {synset_id: _decimal_score(positive_probabilities[synset_id]) for synset_id in candidates} + rows = [] + for synset_id in sorted(scored): + keys = list(candidate_sense_keys.get(synset_id, [])) + rows.append( + { + "positive_probability": format(scored[synset_id], "f"), + "sense_keys": sorted(keys), + "synset": synset_id, + } + ) + ordered = sorted(scored.values(), reverse=True) + top = ordered[0] + second = ordered[1] + margin = top - second + confidence = format(top, "f") + margin_text = format(margin, "f") + winners = [synset_id for synset_id, score in scored.items() if score == top] + if len(winners) != 1: + return _blank(AMBIGUOUS, "model_score_tie", rows, confidence, margin_text) + if top < MINIMUM_CONFIDENCE or margin < MINIMUM_MARGIN: + return _blank(AMBIGUOUS, "model_abstention", rows, confidence, margin_text) + selected = winners[0] + if selected not in candidates: + return _blank(INVALID, "invalid_model_output", rows, confidence, margin_text) + keys = sorted(candidate_sense_keys.get(selected, [])) + if len(keys) != 1: + return _blank(AMBIGUOUS, "pwn30_sense_key_not_unique", rows, confidence, margin_text) + return { + "model_candidate_scores": rows, + "model_confidence": confidence, + "model_margin": margin_text, + "model_resolution_status": RESOLVED, + "primary_evidence_code": "model_margin", + "selected_sense_key": keys[0], + "selected_synset": selected, + } + + +def overlay_status(frozen_status: str, frozen_method: str, model_status: str | None) -> str: + """Combine one frozen tier with an optional tier-3 status. Tier 3 cannot replace a decision.""" + if frozen_status == EXACT: + if model_status is not None: + raise RuntimeError("tier 3 overrode an exact constituent") + return EXACT + if frozen_status == RESOLVED: + if model_status is not None: + raise RuntimeError("tier 3 overrode a resolved constituent") + return LESK_RESOLVED + if frozen_status == UNRESOLVED: + if model_status is not None: + raise RuntimeError("tier 3 overrode an unresolved constituent") + return UNRESOLVED + if frozen_status != AMBIGUOUS: + raise RuntimeError("unknown frozen constituent status") + if frozen_method != "EXTENDED_LESK_V1": + if model_status is not None: + raise RuntimeError("tier 3 overrode a non-lesk constituent") + return AMBIGUOUS + if model_status is None: + raise RuntimeError("ambiguous lesk constituent has no tier 3 result") + if model_status == RESOLVED: + return MODEL_RESOLVED + if model_status in {AMBIGUOUS, INVALID, ERROR}: + return model_status + raise RuntimeError("unknown model status") + + +def project_row_status(statuses: list[str]) -> str: + """Project one row. Fewer than two content constituents stays unknown.""" + if len(statuses) < 2 or any(status not in READY_CONSTITUENT for status in statuses): + return ROW_UNKNOWN + return RESIDUAL_READY + + +def coverage_gate( + *, + high_ready: int, + secondary_ready: int, + total_ready: int, + invalid_output_count: int, + error_count: int, + determinism: str, +) -> dict: + """Apply the preregistered coverage gate. The thresholds are not fit to this run.""" + counts = (high_ready, secondary_ready, total_ready, invalid_output_count, error_count) + if min(counts) < 0: + raise RuntimeError("negative coverage count") + if determinism != IDENTICAL: + status = NOT_DETERMINISTIC + transition = NEXT_DETERMINISM + elif invalid_output_count > 0 or error_count > 0: + status = REJECTED + transition = NEXT_REVISION + elif ( + high_ready > BASELINE_HIGH_READY + and secondary_ready > BASELINE_SECONDARY_READY + and total_ready > BASELINE_TOTAL_READY + and invalid_output_count == 0 + and error_count == 0 + and determinism == IDENTICAL + ): + status = PROMISING + transition = NEXT_PROMISING + else: + status = INSUFFICIENT + transition = NEXT_REVISION + return { + "candidate_status": status, + "next_legal_transition": transition, + "next_transition_authorized": False, + "state": "CANDIDATE_EVALUATED", + } + + +def _confidence_bin(value: Decimal) -> str: + if value < Decimal("0.50"): + return "below_0.50" + if value < Decimal("0.60"): + return "0.50_to_0.60" + if value < Decimal("0.70"): + return "0.60_to_0.70" + if value < Decimal("0.80"): + return "0.70_to_0.80" + if value < Decimal("0.90"): + return "0.80_to_0.90" + return "0.90_to_1.00" + + +def _margin_bin(value: Decimal) -> str: + if value == 0: + return "exact_tie" + if value < Decimal("0.10"): + return "below_0.10" + if value < Decimal("0.25"): + return "0.10_to_0.25" + if value < Decimal("0.50"): + return "0.25_to_0.50" + return "0.50_or_more" + + +def summarize_confidence(rows: list[dict]) -> dict: + """Summarize tier-3 scores. Operator buckets are not an argument.""" + confidence_bins = { + "0.50_to_0.60": 0, + "0.60_to_0.70": 0, + "0.70_to_0.80": 0, + "0.80_to_0.90": 0, + "0.90_to_1.00": 0, + "below_0.50": 0, + } + margin_bins = { + "0.10_to_0.25": 0, + "0.25_to_0.50": 0, + "0.50_or_more": 0, + "below_0.10": 0, + "exact_tie": 0, + } + by_count: dict[str, dict[str, int]] = {} + resolved = 0 + abstained = 0 + invalid = 0 + errors = 0 + for row in rows: + status = row["model_resolution_status"] + if status == RESOLVED: + resolved += 1 + elif status == AMBIGUOUS: + abstained += 1 + elif status == INVALID: + invalid += 1 + elif status == ERROR: + errors += 1 + else: + raise RuntimeError("unknown model status in the confidence summary") + key = str(len(row["candidate_pwn30_synsets"])) + bucket = by_count.setdefault( + key, + {"abstained": 0, "attempts": 0, "error": 0, "invalid": 0, "resolved": 0}, + ) + bucket["attempts"] += 1 + if status == RESOLVED: + bucket["resolved"] += 1 + elif status == AMBIGUOUS: + bucket["abstained"] += 1 + elif status == INVALID: + bucket["invalid"] += 1 + else: + bucket["error"] += 1 + if row["model_confidence"] is not None: + confidence_bins[_confidence_bin(Decimal(row["model_confidence"]))] += 1 + if row["model_margin"] is not None: + margin_bins[_margin_bin(Decimal(row["model_margin"]))] += 1 + return { + "abstained": abstained, + "by_candidate_count": dict(sorted(by_count.items(), key=lambda item: int(item[0]))), + "confidence_bins": confidence_bins, + "error": errors, + "invalid": invalid, + "margin_bins": margin_bins, + "resolved": resolved, + } diff --git a/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1.py b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1.py new file mode 100644 index 00000000..ae1ff1a7 --- /dev/null +++ b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1.py @@ -0,0 +1,578 @@ +"""Development comparison for one frozen residual replay. + +The functions do not encode text, do not read operator labels, and do not +choose a threshold. Scores enter only after the replay has hashed them. +""" + +from __future__ import annotations + +import random +from decimal import Decimal, ROUND_HALF_EVEN + +from hyperlexical.model_based_wsd_candidate_v1 import project_row_status +from hyperlexical.semantic_compositionality_residual import ( + RESIDUAL_QUANTUM, + decimal_mean, + percentile, +) + +EXACT = "EXACT" +RESOLVED = "RESOLVED" +AMBIGUOUS = "AMBIGUOUS" +UNRESOLVED = "UNRESOLVED" +TIER1_STRUCTURAL = "TIER1_STRUCTURAL" +TIER1_UNIQUE = "TIER1_UNIQUE_LEMMA" +TIER2_LESK = "TIER2_EXTENDED_LESK" +TIER3_GLOSSBERT = "TIER3_GLOSSBERT" +TIER_UNRESOLVED = "UNRESOLVED" +READY = "RESIDUAL_READY" +ROW_UNKNOWN = "UNKNOWN" +SUPPORTED = "SUPPORTED_DIRECTION" +NO_SEPARATION = "NO_DIRECTIONAL_SEPARATION" +INVERTED = "INVERTED_DIRECTION" +PROMISING = "CANDIDATE_PROMISING" +INSUFFICIENT = "CANDIDATE_INSUFFICIENT" +NOT_COMPUTABLE = "NOT_COMPUTABLE" +NOT_DETERMINISTIC = "NOT_DETERMINISTIC" +BOOTSTRAP_SEED = 0 +BOOTSTRAP_RESAMPLES = 10000 +TIER3_MIN_CELL = 3 +EFFECT_QUANTUM = Decimal("0.000001") +INTERVAL_LOW = Decimal("2.5") +INTERVAL_HIGH = Decimal("97.5") + +_DIRECTION_RULE = ( + "The hypothesized direction is a larger HIGH residual than a SECONDARY residual. " + "Supported requires a higher HIGH median, a positive rank-biserial, and an AUC above one half." +) +_AUC_RULE = ( + "AUC is the share of HIGH-versus-SECONDARY pairs in which the HIGH residual is larger. " + "Ties count one half. HIGH is the positive class. SECONDARY is the negative class. " + "REJECT is excluded." +) +_EFFECT_RULE = ( + "The rank-biserial is favorable pairs minus unfavorable pairs, divided by the " + "number of HIGH-SECONDARY pairs. A favorable pair has the larger HIGH residual." +) +_BOOTSTRAP_RULE = ( + "Resample each class with replacement. The interval is the interpolated 2.5 and 97.5 " + "percentiles. The seed and the resample count are fixed before the scores are joined to labels." +) +_EXTREME_RULE = ( + "Drop the largest HIGH residual and the smallest SECONDARY residual. " + "The result is extreme-driven when the HIGH median is no longer larger." +) +_TIER3_RULE = ( + "Tier 3 concentration requires both the no-tier-3 group and the tier-3 group " + "to have at least three HIGH rows and three SECONDARY rows, the no-tier-3 group " + "to lack the hypothesized direction, and the tier-3 group to show it." +) +_NO_THRESHOLD = "No residual threshold is selected. No row receives a semantic yes or no." + + +def analysis_plan() -> dict: + """Return the preregistered comparison. It contains no residual scores.""" + return { + "auc": _AUC_RULE, + "bootstrap_resamples": BOOTSTRAP_RESAMPLES, + "bootstrap_rule": _BOOTSTRAP_RULE, + "bootstrap_seed": BOOTSTRAP_SEED, + "direction": _DIRECTION_RULE, + "effect": _EFFECT_RULE, + "emits_yes_no": False, + "extreme_rule": _EXTREME_RULE, + "high_positive_class": "HIGH", + "json_schema_document": None, + "reject_excluded_from_auc": True, + "secondary_negative_class": "SECONDARY", + "semantic_noncompositionality_threshold": None, + "threshold_eligible": False, + "threshold_rule": _NO_THRESHOLD, + "tier3_min_cell": TIER3_MIN_CELL, + "tier3_rule": _TIER3_RULE, + } + + +def integrate_constituent(resolver_row: dict, model_row: dict | None) -> dict: + """Copy one frozen constituent decision. The model row is consulted only for a Lesk tie.""" + method = resolver_row["resolution_method"] + status = resolver_row["resolution_status"] + if method == "STRUCTURAL_EXACT" and status == EXACT: + _reject_model(model_row) + return _copy(resolver_row, TIER1_STRUCTURAL, EXACT, resolver_row["selected_synset"], None) + if method == "UNIQUE_LEMMA" and status == EXACT: + _reject_model(model_row) + return _copy(resolver_row, TIER1_UNIQUE, EXACT, resolver_row["selected_synset"], None) + if method == "EXTENDED_LESK_V1" and status == RESOLVED: + _reject_model(model_row) + return _copy(resolver_row, TIER2_LESK, RESOLVED, resolver_row["selected_synset"], None) + if method == "NONE" and status == UNRESOLVED: + _reject_model(model_row) + return _copy(resolver_row, TIER_UNRESOLVED, UNRESOLVED, None, None) + if method == "STRUCTURAL_EXACT" and status == AMBIGUOUS: + _reject_model(model_row) + return _copy(resolver_row, TIER1_STRUCTURAL, AMBIGUOUS, None, None) + if method != "EXTENDED_LESK_V1" or status != AMBIGUOUS: + raise RuntimeError("constituent decision is outside the frozen stack") + if model_row is None: + raise RuntimeError("lesk tie has no frozen model row") + if model_row["prior_resolution_status"] != AMBIGUOUS: + raise RuntimeError("model row is not a lesk tie") + if model_row["model_resolution_status"] == RESOLVED: + selected = model_row["selected_synset"] + if selected not in resolver_row["candidate_synsets"]: + raise RuntimeError("model synset is outside the frozen candidates") + return _copy( + resolver_row, + TIER3_GLOSSBERT, + RESOLVED, + selected, + model_row["selected_sense_key"], + model_row["primary_evidence_code"], + ) + if model_row["model_resolution_status"] == AMBIGUOUS: + if model_row["selected_synset"] is not None: + raise RuntimeError("abstaining model row names a synset") + return _copy(resolver_row, TIER3_GLOSSBERT, AMBIGUOUS, None, None, model_row["primary_evidence_code"]) + raise RuntimeError("model row is not a frozen resolved or abstaining decision") + + +def _reject_model(model_row: dict | None) -> None: + if model_row is not None: + raise RuntimeError("model row overrides a frozen non-tie") + + +def _copy(resolver_row, tier, status, synset, sense_key, provenance=None) -> dict: + return { + "constituent_index": resolver_row["constituent_index"], + "constituent_pos": resolver_row["constituent_pos"], + "constituent_surface": resolver_row["constituent_surface"], + "parent_row_id": resolver_row["parent_row_id"], + "parent_surface": resolver_row["parent_surface"], + "parent_synset": resolver_row["parent_synset"], + "resolution_provenance": provenance or resolver_row["primary_evidence_code"], + "resolution_status": status, + "resolution_tier": tier, + "selected_pwn30_synset": synset, + "selected_sense_key_if_available": sense_key, + } + + +def projection_token(tier: str, status: str) -> str: + """Map an integrated constituent onto the frozen projection vocabulary.""" + if status == EXACT: + return EXACT + if tier == TIER2_LESK and status == RESOLVED: + return "LESK_RESOLVED" + if tier == TIER3_GLOSSBERT and status == RESOLVED: + return "MODEL_RESOLVED" + if status == UNRESOLVED: + return UNRESOLVED + if status == AMBIGUOUS: + return AMBIGUOUS + raise RuntimeError("integrated constituent has no projection token") + + +def row_projection(statuses: list[str]) -> str: + return project_row_status(statuses) + + +def abstention_reason(statuses: list[str]) -> str | None: + """Return the leftmost frozen abstention. A ready row returns None.""" + if len(statuses) < 2: + return "fewer_than_two_content_constituents" + for status in statuses: + if status == AMBIGUOUS: + return "ambiguous_content_constituent" + if status == UNRESOLVED: + return "unresolved_content_constituent" + if status not in {EXACT, RESOLVED}: + raise RuntimeError("unknown integrated status") + return None + + +def _ordered(scores: list[str]) -> list[str]: + return sorted(scores, key=Decimal) + + +def sample_std(scores: list[str]) -> str | None: + """Sample standard deviation at the residual quantum. One row has no dispersion.""" + if len(scores) < 2: + return None + mean = sum((Decimal(score) for score in scores), start=Decimal(0)) / Decimal(len(scores)) + dispersion = sum((Decimal(score) - mean) ** 2 for score in scores) / Decimal(len(scores) - 1) + return str(dispersion.sqrt().quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def full_distribution(scores: list[str]) -> dict: + if not scores: + return { + "count": 0, + "max": None, + "mean": None, + "median": None, + "min": None, + "p10": None, + "p25": None, + "p75": None, + "p90": None, + "status": NOT_COMPUTABLE, + "std": None, + } + ordered = _ordered(scores) + return { + "count": len(ordered), + "max": ordered[-1], + "mean": decimal_mean(ordered), + "median": percentile(ordered, 50), + "min": ordered[0], + "p10": percentile(ordered, 10), + "p25": percentile(ordered, 25), + "p75": percentile(ordered, 75), + "p90": percentile(ordered, 90), + "status": "DESCRIPTIVE", + "std": sample_std(ordered), + } + + +def _quantize_effect(value: Decimal) -> str: + return str(value.quantize(EFFECT_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def _quantize_residual(value: Decimal) -> str: + return str(value.quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def pair_comparison(high_scores: list[str], secondary_scores: list[str]) -> dict: + """Compare HIGH with SECONDARY. REJECT is not an argument.""" + if not high_scores or not secondary_scores: + return {"status": NOT_COMPUTABLE} + favorable = 0 + unfavorable = 0 + ties = 0 + for high in high_scores: + high_value = Decimal(high) + for secondary in secondary_scores: + secondary_value = Decimal(secondary) + if high_value > secondary_value: + favorable += 1 + elif high_value < secondary_value: + unfavorable += 1 + else: + ties += 1 + pairs = Decimal(len(high_scores) * len(secondary_scores)) + u_high = Decimal(favorable) + (Decimal(ties) / Decimal(2)) + u_secondary = Decimal(unfavorable) + (Decimal(ties) / Decimal(2)) + high_dist = full_distribution(high_scores) + secondary_dist = full_distribution(secondary_scores) + mean_gap = Decimal(high_dist["mean"]) - Decimal(secondary_dist["mean"]) + median_gap = Decimal(high_dist["median"]) - Decimal(secondary_dist["median"]) + auc = u_high / pairs + effect = (Decimal(favorable) - Decimal(unfavorable)) / pairs + return { + "auc": _quantize_effect(auc), + "favorable_pairs": favorable, + "high": high_dist, + "mean_difference": _quantize_residual(mean_gap), + "median_difference": _quantize_residual(median_gap), + "rank_biserial": _quantize_effect(effect), + "secondary": secondary_dist, + "status": "DESCRIPTIVE", + "ties": ties, + "u_high": _quantize_effect(u_high), + "u_secondary": _quantize_effect(u_secondary), + "unfavorable_pairs": unfavorable, + } + + +def direction_result(comparison: dict) -> str: + if comparison.get("status") == NOT_COMPUTABLE: + return NOT_COMPUTABLE + median_gap = Decimal(comparison["median_difference"]) + effect = Decimal(comparison["rank_biserial"]) + auc = Decimal(comparison["auc"]) + if median_gap > 0 and effect > 0 and auc > Decimal("0.5"): + return SUPPORTED + if median_gap < 0 and effect < 0 and auc < Decimal("0.5"): + return INVERTED + return NO_SEPARATION + + +def _supports(scores_high: list[str], scores_secondary: list[str]) -> bool: + if not scores_high or not scores_secondary: + return False + return direction_result(pair_comparison(scores_high, scores_secondary)) == SUPPORTED + + +def _interpolated(sorted_values: list[Decimal], percent: Decimal) -> Decimal: + count = len(sorted_values) + if count == 1: + return sorted_values[0] + rank = Decimal(count - 1) * (percent / Decimal(100)) + low = int(rank) + high = min(low + 1, count - 1) + weight = rank - Decimal(low) + return sorted_values[low] + (sorted_values[high] - sorted_values[low]) * weight + + +def bootstrap_intervals( + high_scores: list[str], + secondary_scores: list[str], + *, + seed: int = BOOTSTRAP_SEED, + resamples: int = BOOTSTRAP_RESAMPLES, +) -> dict: + """Deterministic percentile intervals. The seed and count are arguments, not a search.""" + if seed != BOOTSTRAP_SEED or resamples != BOOTSTRAP_RESAMPLES: + if resamples < 1 or seed < 0: + raise RuntimeError("bootstrap settings are invalid") + if not high_scores or not secondary_scores: + return {"status": NOT_COMPUTABLE} + generator = random.Random(seed) + high_values = [Decimal(score) for score in high_scores] + secondary_values = [Decimal(score) for score in secondary_scores] + mean_gaps = [] + median_gaps = [] + aucs = [] + for _ in range(resamples): + high_draw = [high_values[generator.randrange(len(high_values))] for _ in high_values] + secondary_draw = [secondary_values[generator.randrange(len(secondary_values))] for _ in secondary_values] + high_text = [format(value, "f") for value in high_draw] + secondary_text = [format(value, "f") for value in secondary_draw] + comparison = pair_comparison(high_text, secondary_text) + mean_gaps.append(Decimal(comparison["mean_difference"])) + median_gaps.append(Decimal(comparison["median_difference"])) + aucs.append(Decimal(comparison["auc"])) + return { + "auc": _interval(aucs, EFFECT_QUANTUM), + "mean_difference": _interval(mean_gaps, RESIDUAL_QUANTUM), + "median_difference": _interval(median_gaps, RESIDUAL_QUANTUM), + "resamples": resamples, + "seed": seed, + "status": "DESCRIPTIVE", + } + + +def _interval(values: list[Decimal], quantum: Decimal) -> dict: + ordered = sorted(values) + low = _interpolated(ordered, INTERVAL_LOW).quantize(quantum, rounding=ROUND_HALF_EVEN) + high = _interpolated(ordered, INTERVAL_HIGH).quantize(quantum, rounding=ROUND_HALF_EVEN) + return {"high": str(high), "low": str(low)} + + +def _cell(rows: list[dict], bucket: str) -> list[str]: + return [row["residual_score"] for row in rows if row["bucket"] == bucket] + + +def tier3_concentration(rows: list[dict]) -> dict: + """Return whether the direction exists only inside the Tier 3 subgroup.""" + groups = {} + concentrated = False + flags = {} + for name, uses_tier3 in (("no_tier3", False), ("with_tier3", True)): + chosen = [row for row in rows if row["uses_tier3"] is uses_tier3 and row["bucket"] in {"HIGH", "SECONDARY"}] + high = _cell(chosen, "HIGH") + secondary = _cell(chosen, "SECONDARY") + computable = len(high) >= TIER3_MIN_CELL and len(secondary) >= TIER3_MIN_CELL + if not computable: + groups[name] = { + "high_count": len(high), + "secondary_count": len(secondary), + "status": NOT_COMPUTABLE, + } + flags[name] = None + continue + comparison = pair_comparison(high, secondary) + groups[name] = { + "comparison": comparison, + "direction": direction_result(comparison), + "high_count": len(high), + "secondary_count": len(secondary), + "status": "DESCRIPTIVE", + } + flags[name] = groups[name]["direction"] == SUPPORTED + if flags["no_tier3"] is False and flags["with_tier3"] is True: + concentrated = True + return {"concentrated": concentrated, "groups": groups} + + +def extreme_driven(high_scores: list[str], secondary_scores: list[str]) -> bool: + """True when dropping the most favorable extreme on each side removes the median gap.""" + if len(high_scores) < TIER3_MIN_CELL or len(secondary_scores) < TIER3_MIN_CELL: + return False + if not _supports(high_scores, secondary_scores): + return False + kept_high = list(high_scores) + kept_secondary = list(secondary_scores) + kept_high.remove(max(kept_high, key=Decimal)) + kept_secondary.remove(min(kept_secondary, key=Decimal)) + return not _supports(kept_high, kept_secondary) + + +def pos_partitioned(rows: list[dict]) -> bool: + high = {row["pos"] for row in rows if row["bucket"] == "HIGH"} + secondary = {row["pos"] for row in rows if row["bucket"] == "SECONDARY"} + return bool(high) and bool(secondary) and high.isdisjoint(secondary) + + +def replay_decision( + *, + readiness_reproduced: bool, + determinism: str, + direction: str, + tier3_concentrated: bool, + extremes: bool, + pos_split: bool, +) -> dict: + """Apply the preregistered status rule. The flags are not a threshold search.""" + if not readiness_reproduced: + status = NOT_COMPUTABLE + transition = "RESIDUAL_REPLAY_READINESS_REVIEW_AUTHORIZATION" + elif determinism != "IDENTICAL": + status = NOT_DETERMINISTIC + transition = "RESIDUAL_REPLAY_DETERMINISM_REVIEW_AUTHORIZATION" + elif direction == NOT_COMPUTABLE: + status = NOT_COMPUTABLE + transition = "RESIDUAL_REPLAY_READINESS_REVIEW_AUTHORIZATION" + elif direction == SUPPORTED and not tier3_concentrated and not extremes and not pos_split: + status = PROMISING + transition = "RESIDUAL_THRESHOLD_PREREGISTRATION_AUTHORIZATION" + else: + status = INSUFFICIENT + transition = "RESIDUAL_V2_DESIGN_AUTHORIZATION" + return { + "candidate_status": status, + "next_legal_transition": transition, + "next_transition_authorized": False, + "state": "RESIDUAL_DEVELOPMENT_ANALYZED_V2", + "threshold_eligible": False, + } + + +def _ranks(values: list[Decimal]) -> list[Decimal]: + order = sorted(range(len(values)), key=lambda index: values[index]) + ranks = [Decimal(0)] * len(values) + start = 0 + while start < len(values): + stop = start + 1 + while stop < len(values) and values[order[stop]] == values[order[start]]: + stop += 1 + rank = Decimal(start + 1 + stop) / Decimal(2) + for index in order[start:stop]: + ranks[index] = rank + start = stop + return ranks + + +def spearman(xs: list[str], ys: list[str]) -> str | None: + if len(xs) != len(ys) or len(xs) < 3: + return None + x_values = [Decimal(value) for value in xs] + y_values = [Decimal(value) for value in ys] + x_ranks = _ranks(x_values) + y_ranks = _ranks(y_values) + x_mean = sum(x_ranks, start=Decimal(0)) / Decimal(len(xs)) + y_mean = sum(y_ranks, start=Decimal(0)) / Decimal(len(ys)) + covariance = sum((x - x_mean) * (y - y_mean) for x, y in zip(x_ranks, y_ranks)) + x_scale = sum((x - x_mean) ** 2 for x in x_ranks) + y_scale = sum((y - y_mean) ** 2 for y in y_ranks) + if x_scale == 0 or y_scale == 0: + return None + correlation = covariance / (x_scale.sqrt() * y_scale.sqrt()) + return _quantize_effect(correlation) + + +def numeric_summary(values: list[str]) -> dict: + if not values: + return {"count": 0, "status": NOT_COMPUTABLE} + ordered = _ordered(values) + return { + "count": len(ordered), + "max": ordered[-1], + "mean": decimal_mean(ordered), + "median": percentile(ordered, 50), + "min": ordered[0], + "status": "DESCRIPTIVE", + } + + +def confound_report(rows: list[dict]) -> dict: + """Describe nuisance structure. It does not change a residual.""" + target = [row for row in rows if row["bucket"] in {"HIGH", "SECONDARY"}] + tier3 = tier3_concentration(target) + high_scores = _cell(target, "HIGH") + secondary_scores = _cell(target, "SECONDARY") + nuisance = {} + for field in ( + "token_count", + "character_length", + "content_count", + "max_candidate_senses", + "min_glossbert_confidence", + "min_glossbert_margin", + ): + nuisance[field] = _nuisance(target, field) + poses = {} + for pos in sorted({row["pos"] for row in target}): + chosen = [row for row in target if row["pos"] == pos] + high = _cell(chosen, "HIGH") + secondary = _cell(chosen, "SECONDARY") + if len(high) < TIER3_MIN_CELL or len(secondary) < TIER3_MIN_CELL: + poses[pos] = {"high_count": len(high), "secondary_count": len(secondary), "status": NOT_COMPUTABLE} + else: + poses[pos] = {"comparison": pair_comparison(high, secondary), "status": "DESCRIPTIVE"} + return { + "extreme_driven": extreme_driven(high_scores, secondary_scores), + "nuisance": nuisance, + "pos": poses, + "pos_partitioned": pos_partitioned(target), + "tier3": tier3, + "tier3_counts": { + "high_with_tier3": _tier_count(rows, "HIGH", True), + "high_without_tier3": _tier_count(rows, "HIGH", False), + "secondary_with_tier3": _tier_count(rows, "SECONDARY", True), + "secondary_without_tier3": _tier_count(rows, "SECONDARY", False), + }, + } + + +def _tier_count(rows: list[dict], bucket: str, uses_tier3: bool) -> int: + return sum(row["bucket"] == bucket and row["uses_tier3"] is uses_tier3 for row in rows) + + +def _nuisance(rows: list[dict], field: str) -> dict: + report = {} + for bucket in ("HIGH", "SECONDARY"): + paired = [ + (str(row[field]), row["residual_score"]) + for row in rows + if row["bucket"] == bucket and row[field] is not None + ] + if not paired: + report[bucket] = {"status": NOT_COMPUTABLE} + continue + values = [item[0] for item in paired] + report[bucket] = { + "spearman_with_residual": spearman(values, [item[1] for item in paired]), + "summary": numeric_summary(values), + } + return report + + +def outlier_pair(rows: list[dict], bucket: str) -> dict: + chosen = [row for row in rows if row["bucket"] == bucket] + if not chosen: + return {"status": NOT_COMPUTABLE} + low = min(chosen, key=lambda row: Decimal(row["residual_score"])) + high = max(chosen, key=lambda row: Decimal(row["residual_score"])) + return {"largest": _outlier(high), "smallest": _outlier(low), "status": "DESCRIPTIVE"} + + +def _outlier(row: dict) -> dict: + return { + "residual_score": row["residual_score"], + "resolution_tiers": list(row["tiers"]), + "row_id": row["row_id"], + "surface": row["surface"], + "synset": row["synset"], + } diff --git a/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1_replay.py b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1_replay.py new file mode 100644 index 00000000..83f4b467 --- /dev/null +++ b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1_replay.py @@ -0,0 +1,766 @@ +"""Replay the frozen residual on the frozen constituent-sense stack. + +Operator labels are read only after the score artifact and its receipt have +been hashed. The residual model, the three resolution tiers, and the residual +formula are not changed. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +from collections import Counter, defaultdict +from decimal import Decimal +from pathlib import Path + +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["NUMEXPR_NUM_THREADS"] = "1" +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["OPENBLAS_NUM_THREADS"] = "1" +os.environ["TOKENIZERS_PARALLELISM"] = "false" + +from hyperlexical.constituent_sense_resolution_v1_replay import ( + EVENTS, + EVIDENCE_MANIFEST, + HYPERLEX, + LEDGER_FILE, + OPERATOR_COUNTS, + OPERATORS, + REPLAY_PATH as RESOLVER_REPLAY_PATH, + SOURCE, + TRACKER, + WORDNET, + load_sealed_rows, + refuse, + sha256, + write_json, + write_jsonl, +) +from hyperlexical.model_based_wsd_candidate_v1_replay import ( + ANALYSIS_PATH as WSD_ANALYSIS_PATH, + DECISION_PATH as WSD_DECISION_PATH, + EXPECTED as WSD_EXPECTED, + LICENSE_PATH as WSD_LICENSE_PATH, + PROJECTION_PATH, + PROVENANCE_PATH as WSD_PROVENANCE_PATH, + RAW_PATH as WSD_RAW_PATH, + RESOLUTION_PATH as WSD_RESOLUTION_PATH, + SPEC_PATH as WSD_SPEC_PATH, + load_jsonl, + load_sense_index, + sense_keys_for, +) +from hyperlexical.residual_model_resolved_replay_v1 import ( + AMBIGUOUS, + BOOTSTRAP_RESAMPLES, + BOOTSTRAP_SEED, + EXACT, + RESOLVED, + TIER1_STRUCTURAL, + TIER3_GLOSSBERT, + UNRESOLVED, + abstention_reason, + analysis_plan, + bootstrap_intervals, + confound_report, + direction_result, + full_distribution, + integrate_constituent, + outlier_pair, + pair_comparison, + projection_token, + replay_decision, + row_projection, +) +from hyperlexical.semantic_compositionality_residual import ( + COMPOSITION_OPERATOR, + DISTANCE_METRIC, + MODEL_NAME, + MODEL_REVISION, + representation_text, + residual_score, + select_lemma, + vector_hash, +) +from hyperlexical.semantic_compositionality_residual_replay import ( + MAX_SEQUENCE_LENGTH, + OUTPUT_DIMENSION, + build_indexes, + encode_texts, + load_encoder, + pointer_records, + token_length, +) +from hyperlexical.unbind_sense_screen_v1 import load_exceptions + +RESIDUAL_SPEC = SOURCE / "RESIDUAL_CANDIDATE_SPEC.json" +RESIDUAL_SCORES = SOURCE / "RESIDUAL_DEVELOPMENT_SCORES.jsonl" +INTEGRATED_PATH = SOURCE / "INTEGRATED_CONSTITUENT_RESOLUTION_V1.jsonl" +SCORES_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_REPLAY_SCORES.jsonl" +RECEIPT_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_REPLAY_RECEIPT.json" +ANALYSIS_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_DEVELOPMENT_ANALYSIS.json" +CONFOUND_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_CONFOUND_ANALYSIS.json" +DECISION_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_CANDIDATE_DECISION.json" + +CURRENT_TRACKER = "6da1e9730c785d2784d23433455515989d312ed8c76f4763b3a2dd6a1f2c4b2d" +RESIDUAL_SPEC_SHA = "39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d" +RESIDUAL_SCORES_SHA = "cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7" +AUTHORIZATION = "RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION" +EXPECTED_READY = 73 +EXPECTED_UNKNOWN = 152 + +EXPECTED = dict(WSD_EXPECTED) +EXPECTED[TRACKER] = CURRENT_TRACKER +EXPECTED[WSD_SPEC_PATH] = "c861ff7fff11ae6a790531267229c18d6e6e0a171a9bf6c34cfb6f7e7b14498c" +EXPECTED[WSD_PROVENANCE_PATH] = "6d91283244a88f44477c836e6549118d5d96a36bb878bf448fcadb13ff765e11" +EXPECTED[WSD_LICENSE_PATH] = "67d45243959e3e76643a675d49b508a73d0f7f0bf8fbf060ce7b83c9200c621b" +EXPECTED[WSD_RAW_PATH] = "a0e508c225e6db4cdbcae701682202b1d854f3762546dbfe10b84a15e0e9e17c" +EXPECTED[WSD_RESOLUTION_PATH] = "ed945989cf4947ac84633ba2c4aa10c1ba381d2396da0b573a844f83ec367a18" +EXPECTED[PROJECTION_PATH] = "c75834faf4a84d36e83246244e0aa7c6c7788c3a57cfdb7f77c7628a52023328" +EXPECTED[WSD_ANALYSIS_PATH] = "8ca0d8c8dd7e510a04daef6b3fbd78ace6d20b173e88d2a3d0c2e10fdaafe30b" +EXPECTED[WSD_DECISION_PATH] = "c8ed0b8acaaa415277c5f9bdbf982d075136b795b5d95fd8813d5a3010072dbb" +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py"] = ( + "7f489772162dd0be249df3fe8c0a5476e69e5a4ca73fba7468f61c60cda84bad" +) +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py"] = ( + "5981978209da5884fe6cd6efbaf7593a37e1a40138a75aa70906694cee8b8a09" +) + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if not path.is_file() or sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def _digest(payload) -> str: + text = json.dumps(payload, sort_keys=True, ensure_ascii=True) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def attach_sense_key(item: dict, exceptions: dict, sense_index: dict) -> None: + """Fill a unique PWN 3.0 sense key. The selected synset stays as frozen.""" + synset = item["selected_pwn30_synset"] + if synset is None: + return + keys = sense_keys_for(synset, item["constituent_surface"], exceptions, sense_index) + current = item["selected_sense_key_if_available"] + if current is not None: + if current not in keys: + refuse(f"frozen sense key is not in the PWN 3.0 index for {synset}") + return + if len(keys) == 1: + item["selected_sense_key_if_available"] = keys[0] + + +def build_integrated(resolver_rows: list[dict], model_rows: list[dict], exceptions: dict, sense_index: dict): + model = {(row["parent_row_id"], row["constituent_index"]): row for row in model_rows} + if len(model) != len(model_rows): + refuse("duplicate model constituent keys") + consumed = set() + integrated = [] + by_parent = defaultdict(list) + for row in resolver_rows: + if row["constituent_index"] is None: + continue + key = (row["parent_row_id"], row["constituent_index"]) + model_row = model.get(key) + if model_row is not None: + consumed.add(key) + item = integrate_constituent(row, model_row) + attach_sense_key(item, exceptions, sense_index) + item["candidate_count"] = len(row["candidate_synsets"]) + integrated.append(item) + by_parent[row["parent_row_id"]].append(item) + if consumed != set(model): + refuse("a model row does not match a frozen lesk tie") + for items in by_parent.values(): + items.sort(key=lambda item: item["constituent_index"]) + return integrated, by_parent + + +def project_manifest(sealed: list[dict], by_parent: dict) -> list[dict]: + projected = [] + for row in sealed: + items = by_parent.get(row["row_id"], []) + tokens = [projection_token(item["resolution_tier"], item["resolution_status"]) for item in items] + projected.append( + { + "constituent_statuses": tokens, + "content_count": len(tokens), + "parent_row_id": row["row_id"], + "projected_status": row_projection(tokens), + } + ) + return projected + + +def projection_failure(projected: list[dict]) -> str | None: + frozen = json.loads(PROJECTION_PATH.read_text(encoding="utf-8")) + if frozen.get("operator_labels_included") is not False: + return "READINESS_REPRODUCTION_FAILURE projection includes operator labels" + frozen_rows = {row["parent_row_id"]: row for row in frozen["rows"]} + if len(frozen_rows) != 225 or len(projected) != 225: + return "READINESS_REPRODUCTION_FAILURE row count" + for row in projected: + prior = frozen_rows.get(row["parent_row_id"]) + if prior is None: + return f"READINESS_REPRODUCTION_FAILURE missing {row['parent_row_id']}" + if prior["constituent_statuses"] != row["constituent_statuses"] or prior["content_count"] != row["content_count"] or prior["projected_status"] != row["projected_status"]: + return ( + f"READINESS_REPRODUCTION_FAILURE {row['parent_row_id']} " + f"got {row['constituent_statuses']} {row['projected_status']} " + f"expected {prior['constituent_statuses']} {prior['projected_status']}" + ) + counts = Counter(row["projected_status"] for row in projected) + if counts["RESIDUAL_READY"] != EXPECTED_READY or counts["UNKNOWN"] != EXPECTED_UNKNOWN: + return "READINESS_REPRODUCTION_FAILURE totals" + return None + + +def public_integrated(item: dict) -> dict: + return { + "constituent_index": item["constituent_index"], + "constituent_pos": item["constituent_pos"], + "constituent_surface": item["constituent_surface"], + "parent_row_id": item["parent_row_id"], + "parent_surface": item["parent_surface"], + "parent_synset": item["parent_synset"], + "resolution_provenance": item["resolution_provenance"], + "resolution_status": item["resolution_status"], + "resolution_tier": item["resolution_tier"], + "selected_pwn30_synset": item["selected_pwn30_synset"], + "selected_sense_key_if_available": item["selected_sense_key_if_available"], + } + + +def constituent_representation(item: dict, by_id: dict, pointers: list, exceptions: dict) -> str: + synset = item["selected_pwn30_synset"] + record = by_id.get(synset) + if record is None: + refuse(f"selected synset is not in PWN 3.0: {synset}") + if item["resolution_tier"] == TIER1_STRUCTURAL: + pool = [ + lemma + for symbol, target_word, target_id, lemma in pointers + if target_id == synset and symbol in {"+", "\\"} and target_word > 0 and lemma + ] + else: + pool = list(record["lemmas"]) + lemma = select_lemma(pool, item["constituent_surface"], exceptions) + return representation_text(lemma, record["pos"], record["gloss"]) + + +def prepare_jobs(sealed: list[dict], by_parent: dict, by_id: dict, exceptions: dict, model_index: dict): + jobs = [] + unknown = [] + for row in sealed: + items = by_parent.get(row["row_id"], []) + tokens = [projection_token(item["resolution_tier"], item["resolution_status"]) for item in items] + readiness = row_projection(tokens) + if readiness != "RESIDUAL_READY": + statuses = [] + for item in items: + if item["resolution_status"] in {EXACT, RESOLVED}: + statuses.append(RESOLVED if item["resolution_status"] == RESOLVED else EXACT) + else: + statuses.append(item["resolution_status"]) + unknown.append( + { + "abstention_reason": abstention_reason(statuses), + "readiness": readiness, + "row_id": row["row_id"], + } + ) + continue + parent = f"{row['synset_pos']}:{row['synset_offset']}" + pointers = pointer_records(parent, by_id) + parts = [constituent_representation(item, by_id, pointers, exceptions) for item in items] + whole = representation_text(row["surface"], row["pos"], row["gloss"]) + confidences = [] + margins = [] + for item in items: + if item["resolution_tier"] != TIER3_GLOSSBERT: + continue + model_row = model_index[(row["row_id"], item["constituent_index"])] + confidences.append(Decimal(model_row["model_confidence"])) + margins.append(Decimal(model_row["model_margin"])) + jobs.append( + { + "constituent_representation_texts": parts, + "content_constituents": [item["constituent_surface"] for item in items], + "max_candidate_senses": max(item["candidate_count"] for item in items), + "min_glossbert_confidence": format(min(confidences), "f") if confidences else None, + "min_glossbert_margin": format(min(margins), "f") if margins else None, + "pos": row["pos"], + "readiness": readiness, + "resolved_constituent_synsets": [item["selected_pwn30_synset"] for item in items], + "row_id": row["row_id"], + "surface": row["surface"], + "synset": parent, + "tiers": [item["resolution_tier"] for item in items], + "whole_representation_text": whole, + } + ) + return jobs, unknown + + +def score_jobs(jobs: list[dict], vectors: dict[str, list[float]], integrated_sha: str) -> list[dict]: + scored = [] + for job in jobs: + whole = vectors[job["whole_representation_text"]] + parts = [vectors[text] for text in job["constituent_representation_texts"]] + result = residual_score(whole, parts) + if result is None: + refuse(f"residual formula returned no score for {job['row_id']}") + residual, composed = result + if len(whole) != OUTPUT_DIMENSION: + refuse("encoder width drifted") + scored.append( + { + "composed_vector_hash": vector_hash(composed), + "constituent_representation_texts": list(job["constituent_representation_texts"]), + "constituent_resolution_tiers": list(job["tiers"]), + "constituent_vector_hashes": [vector_hash(part) for part in parts], + "content_constituents": list(job["content_constituents"]), + "integrated_resolution_sha256": integrated_sha, + "pos": job["pos"], + "residual_candidate_spec_sha256": RESIDUAL_SPEC_SHA, + "residual_score": residual, + "resolved_constituent_synsets": list(job["resolved_constituent_synsets"]), + "row_id": job["row_id"], + "score_status": "SCORED", + "surface": job["surface"], + "synset": job["synset"], + "whole_representation_text": job["whole_representation_text"], + "whole_vector_hash": vector_hash(whole), + } + ) + return scored + + +def identity_rows(jobs: list[dict], scored: list[dict], unknown: list[dict]) -> list[dict]: + scored_by = {row["row_id"]: row for row in scored} + rows = [] + for job in jobs: + row = scored_by[job["row_id"]] + rows.append( + { + "composed_vector_hash": row["composed_vector_hash"], + "constituent_representation_texts": row["constituent_representation_texts"], + "constituent_vector_hashes": row["constituent_vector_hashes"], + "readiness": job["readiness"], + "residual_score": row["residual_score"], + "row_id": job["row_id"], + "whole_representation_text": row["whole_representation_text"], + "whole_vector_hash": row["whole_vector_hash"], + } + ) + for row in unknown: + rows.append( + { + "composed_vector_hash": None, + "constituent_representation_texts": None, + "constituent_vector_hashes": None, + "readiness": row["readiness"], + "residual_score": None, + "row_id": row["row_id"], + "whole_representation_text": None, + "whole_vector_hash": None, + } + ) + return sorted(rows, key=lambda row: row["row_id"]) + + +def encode_needed(model, jobs: list[dict]) -> dict[str, list[float]]: + needed = sorted({text for job in jobs for text in [job["whole_representation_text"], *job["constituent_representation_texts"]]}) + return encode_texts(model, needed) + + +def apply_overflow(model, jobs: list[dict]) -> tuple[list[dict], list[dict]]: + kept = [] + overflow = [] + for job in jobs: + texts = [job["whole_representation_text"], *job["constituent_representation_texts"]] + if any(token_length(model, text) > MAX_SEQUENCE_LENGTH for text in texts): + overflow.append({"abstention_reason": "representation_exceeds_max_sequence_length", "row_id": job["row_id"]}) + continue + kept.append(job) + return kept, overflow + + +def assemble_scores(sealed: list[dict], scored: list[dict], unknown: list[dict], overflow: list[dict]) -> list[dict]: + scored_by = {row["row_id"]: row for row in scored} + unknown_by = {row["row_id"]: row["abstention_reason"] for row in unknown} + overflow_by = {row["row_id"]: row["abstention_reason"] for row in overflow} + rows = [] + for row in sealed: + if row["row_id"] in scored_by: + rows.append(scored_by[row["row_id"]]) + continue + reason = unknown_by.get(row["row_id"], overflow_by.get(row["row_id"])) + if reason is None: + refuse(f"row has no score and no abstention: {row['row_id']}") + rows.append({"abstention_reason": reason, "row_id": row["row_id"], "score_status": "UNKNOWN"}) + return rows + + +def nuisance_for(job: dict) -> dict: + return { + "character_length": str(len(job["surface"])), + "content_count": str(len(job["content_constituents"])), + "max_candidate_senses": str(job["max_candidate_senses"]), + "min_glossbert_confidence": job["min_glossbert_confidence"], + "min_glossbert_margin": job["min_glossbert_margin"], + "pos": job["pos"], + "surface": job["surface"], + "synset": job["synset"], + "tiers": list(job["tiers"]), + "token_count": str(len(job["surface"].split())), + "uses_tier3": TIER3_GLOSSBERT in job["tiers"], + } + + +def load_operator_buckets() -> dict[str, str]: + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during scoring") + buckets = {} + for row in manifest["rows"]: + buckets[row["row_id"]] = row["operator_bucket"] + if row.get("sense_class") is not None: + refuse("manifest sense class changed") + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + return buckets + + +def tier_counts(rows: list[dict]) -> dict: + report = {} + for bucket in ("HIGH", "SECONDARY"): + chosen = [row for row in rows if row["bucket"] == bucket] + report[f"{bucket.lower()}_with_tier3"] = sum(row["uses_tier3"] for row in chosen) + report[f"{bucket.lower()}_without_tier3"] = sum(not row["uses_tier3"] for row in chosen) + return report + + +def composition_counts(rows: list[dict]) -> dict: + report = {} + for bucket in ("HIGH", "SECONDARY", "REJECT"): + counter = Counter(tuple(row["tiers"]) for row in rows if row["bucket"] == bucket) + report[bucket] = {",".join(key): value for key, value in sorted(counter.items())} + return report + + +def write_failure(status: str, transition: str, detail: str, hashes: dict | None = None) -> None: + payload = { + "candidate_status": status, + "detail": detail, + "emits_yes_no": False, + "json_schema_document": None, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": transition, + "next_transition_authorized": False, + "residual_formula_count": 1, + "runtime_integration": False, + "select_005_authorized": False, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "state": "RESIDUAL_DEVELOPMENT_ANALYZED_V2", + "threshold_eligible": False, + } + decision_sha = write_json(DECISION_PATH, payload) + recorded = dict(hashes or {}) + recorded["decision_sha256"] = decision_sha + update_tracker(payload, recorded) + refuse(detail) + + +def update_tracker(decision: dict, hashes: dict) -> str: + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if sha256(TRACKER) != CURRENT_TRACKER: + refuse("tracker hash drifted before update") + if tracker.get("model_based_wsd_candidate_status") != "CANDIDATE_PROMISING": + refuse("model WSD status drifted") + if tracker.get("constituent_sense_resolution_candidate_status") != "COVERAGE_INSUFFICIENT": + refuse("constituent resolver status drifted") + if tracker.get("residual_evaluation_status") != "CANDIDATE_DISTRIBUTION_FROZEN": + refuse("residual freeze status drifted") + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = CURRENT_TRACKER + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["residual_evaluation_status"] = "CANDIDATE_DISTRIBUTION_FROZEN" + tracker["residual_model_resolved_state"] = decision["state"] + tracker["residual_model_resolved_candidate_status"] = decision["candidate_status"] + tracker["residual_model_resolved_threshold_eligible"] = False + tracker["residual_model_resolved_integrated_sha256"] = hashes.get("integrated_sha256") + tracker["residual_model_resolved_scores_sha256"] = hashes.get("scores_sha256") + tracker["residual_model_resolved_receipt_sha256"] = hashes.get("receipt_sha256") + tracker["residual_model_resolved_analysis_sha256"] = hashes.get("analysis_sha256") + tracker["residual_model_resolved_confound_sha256"] = hashes.get("confound_sha256") + tracker["residual_model_resolved_decision_sha256"] = hashes.get("decision_sha256") + tracker["residual_threshold_eligible"] = False + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = decision["next_legal_transition"] + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + if sha256(RESIDUAL_SPEC) != RESIDUAL_SPEC_SHA or sha256(RESIDUAL_SCORES) != RESIDUAL_SCORES_SHA: + refuse("frozen residual artifact changed") + return tracker_sha + + +def main() -> None: + check_sealed() + if sha256(RESIDUAL_SPEC) != RESIDUAL_SPEC_SHA or sha256(RESIDUAL_SCORES) != RESIDUAL_SCORES_SHA: + refuse("frozen residual artifact changed") + if MODEL_NAME != "sentence-transformers/all-MiniLM-L6-v2": + refuse("residual model name drifted") + if MODEL_REVISION != "1110a243fdf4706b3f48f1d95db1a4f5529b4d41": + refuse("residual model revision drifted") + if COMPOSITION_OPERATOR != "normalized_mean_v1" or DISTANCE_METRIC != "one_minus_cosine_v1": + refuse("residual formula drifted") + sealed = load_sealed_rows() + resolver_rows = load_jsonl(RESOLVER_REPLAY_PATH) + model_rows = load_jsonl(WSD_RESOLUTION_PATH) + if any("operator_bucket" in row for row in resolver_rows) or any("operator_bucket" in row for row in model_rows): + refuse("upstream resolution artifact contains an operator bucket") + exceptions = load_exceptions(WORDNET) + sense_index = load_sense_index(WORDNET / "index.sense") + integrated, by_parent = build_integrated(resolver_rows, model_rows, exceptions, sense_index) + if len(integrated) != 504: + refuse(f"integrated constituent count is {len(integrated)}") + projected = project_manifest(sealed, by_parent) + failure = projection_failure(projected) + if failure: + refuse(failure) + public_rows = [public_integrated(item) for item in integrated] + integrated_sha = write_jsonl(INTEGRATED_PATH, public_rows) + if sha256(INTEGRATED_PATH) != integrated_sha: + refuse("integrated hash drifted at freeze") + print(f"INTEGRATED_FROZEN {integrated_sha}", file=sys.stderr, flush=True) + by_id, _index = build_indexes() + model_index = {(row["parent_row_id"], row["constituent_index"]): row for row in model_rows} + jobs, unknown = prepare_jobs(sealed, by_parent, by_id, exceptions, model_index) + if len(jobs) + len(unknown) != 225: + refuse("prepared row count drifted") + if len(jobs) != EXPECTED_READY or len(unknown) != EXPECTED_UNKNOWN: + refuse("READINESS_REPRODUCTION_FAILURE") + encoder = load_encoder() + kept, overflow = apply_overflow(encoder, jobs) + first_vectors = encode_needed(encoder, kept) + second_vectors = encode_needed(encoder, kept) + first_scored = score_jobs(kept, first_vectors, integrated_sha) + second_scored = score_jobs(kept, second_vectors, integrated_sha) + if identity_rows(kept, first_scored, unknown) != identity_rows(kept, second_scored, unknown): + write_failure( + "NOT_DETERMINISTIC", + "RESIDUAL_REPLAY_DETERMINISM_REVIEW_AUTHORIZATION", + "NOT_DETERMINISTIC", + {"integrated_sha256": integrated_sha}, + ) + scores = assemble_scores(sealed, first_scored, unknown, overflow) + if len(scores) != 225: + refuse("score row count drifted") + if any("operator_bucket" in row or row.get("score_status") not in {"SCORED", "UNKNOWN"} for row in scores): + refuse("score row is malformed") + if any(row.get("semantic_noncompositional") is not None for row in scores): + refuse("score row emits a semantic decision") + score_sha = write_jsonl(SCORES_PATH, scores) + if sha256(SCORES_PATH) != score_sha: + refuse("score hash drifted at freeze") + print(f"SCORES_FROZEN {score_sha}", file=sys.stderr, flush=True) + receipt = { + "analysis_plan": analysis_plan(), + "authorization": AUTHORIZATION, + "composition_operator": COMPOSITION_OPERATOR, + "determinism": "IDENTICAL", + "distance_metric": DISTANCE_METRIC, + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "model_settings": { + "batch_size": 1, + "device": "cpu", + "dtype": "float32", + "eval_mode": True, + "max_sequence_length": MAX_SEQUENCE_LENGTH, + "normalize_embeddings": True, + "seed": 0, + "threads": 1, + }, + "operator_labels_joined": False, + "original_residual_score_sha256": RESIDUAL_SCORES_SHA, + "overflow_count": len(overflow), + "readiness_unknown_count": EXPECTED_UNKNOWN, + "ready_count": EXPECTED_READY, + "residual_candidate_spec_sha256": RESIDUAL_SPEC_SHA, + "residual_formula": "1 - cosine_similarity(whole_sense_vector, normalized_mean(constituent_sense_vectors))", + "residual_formula_count": 1, + "scored_count": len(first_scored), + "scores_sha256": score_sha, + "sequence": [ + "integrated_resolution_hashed", + "residual_scores_hashed", + "operator_labels_not_yet_joined", + ], + "stored_precision": "10 decimal places", + "yes_no_emitted": False, + } + receipt_sha = write_json(RECEIPT_PATH, receipt) + if receipt["operator_labels_joined"] is not False: + refuse("receipt joined labels early") + buckets = load_operator_buckets() + joined = [] + job_by = {job["row_id"]: job for job in jobs} + score_by = {row["row_id"]: row for row in first_scored} + for row_id, score in score_by.items(): + job = job_by[row_id] + meta = nuisance_for(job) + joined.append( + { + "bucket": buckets[row_id], + "residual_score": score["residual_score"], + "row_id": row_id, + **meta, + } + ) + ready_joined = [] + for job in jobs: + meta = nuisance_for(job) + ready_joined.append({"bucket": buckets[job["row_id"]], "row_id": job["row_id"], **meta}) + ready_by = Counter(row["bucket"] for row in ready_joined) + if ready_by["HIGH"] != 28 or ready_by["SECONDARY"] != 11 or ready_by["REJECT"] != 34 or ready_by["QUARANTINE"] != 0: + write_failure( + "NOT_COMPUTABLE", + "RESIDUAL_REPLAY_READINESS_REVIEW_AUTHORIZATION", + "READINESS_REPRODUCTION_FAILURE", + ) + high = [row["residual_score"] for row in joined if row["bucket"] == "HIGH"] + secondary = [row["residual_score"] for row in joined if row["bucket"] == "SECONDARY"] + reject = [row["residual_score"] for row in joined if row["bucket"] == "REJECT"] + comparison = pair_comparison(high, secondary) + direction = direction_result(comparison) + intervals = bootstrap_intervals(high, secondary) + confounds = confound_report(joined) + decision = replay_decision( + readiness_reproduced=True, + determinism="IDENTICAL", + direction=direction, + tier3_concentrated=confounds["tier3"]["concentrated"], + extremes=confounds["extreme_driven"], + pos_split=confounds["pos_partitioned"], + ) + analysis = { + "bootstrap": intervals, + "comparison": comparison, + "direction": direction, + "high": full_distribution(high), + "hypothesis": "HIGH residual > SECONDARY residual", + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "outliers": { + "HIGH": outlier_pair(joined, "HIGH"), + "REJECT": outlier_pair(joined, "REJECT"), + "SECONDARY": outlier_pair(joined, "SECONDARY"), + }, + "ready_by_operator": {name: ready_by[name] for name in OPERATORS}, + "receipt_sha256": receipt_sha, + "reject": full_distribution(reject), + "scores_sha256": score_sha, + "secondary": full_distribution(secondary), + "threshold_eligible": False, + "yes_no_emitted": False, + } + analysis_sha = write_json(ANALYSIS_PATH, analysis) + confound_payload = { + "composition": composition_counts(joined), + "confounds": confounds, + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "ready_tier3_counts": tier_counts(ready_joined), + "receipt_sha256": receipt_sha, + "scores_sha256": score_sha, + "token_definition": "whitespace_separated_surface_tokens", + } + confound_sha = write_json(CONFOUND_PATH, confound_payload) + decision_payload = { + "analysis_sha256": analysis_sha, + "candidate_status": decision["candidate_status"], + "confound_sha256": confound_sha, + "determinism": "IDENTICAL", + "direction": direction, + "emits_yes_no": False, + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": decision["next_legal_transition"], + "next_transition_authorized": False, + "original_residual_score_sha256": RESIDUAL_SCORES_SHA, + "receipt_sha256": receipt_sha, + "residual_candidate_spec_sha256": RESIDUAL_SPEC_SHA, + "residual_formula_count": 1, + "runtime_integration": False, + "scores_sha256": score_sha, + "select_005_authorized": False, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "state": decision["state"], + "threshold_eligible": False, + } + decision_sha = write_json(DECISION_PATH, decision_payload) + hashes = { + "analysis_sha256": analysis_sha, + "confound_sha256": confound_sha, + "decision_sha256": decision_sha, + "integrated_sha256": integrated_sha, + "receipt_sha256": receipt_sha, + "scores_sha256": score_sha, + } + tracker_sha = update_tracker(decision_payload, hashes) + print( + json.dumps( + { + "analysis_sha256": analysis_sha, + "candidate_status": decision["candidate_status"], + "confound_sha256": confound_sha, + "decision_sha256": decision_sha, + "direction": direction, + "integrated_sha256": integrated_sha, + "receipt_sha256": receipt_sha, + "scores_sha256": score_sha, + "tracker_sha256": tracker_sha, + }, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/screen_eval.py b/scripts/shadow/hyperlexical/screen_eval.py new file mode 100644 index 00000000..a14224a4 --- /dev/null +++ b/scripts/shadow/hyperlexical/screen_eval.py @@ -0,0 +1,488 @@ +"""Evaluation lane for a frozen unbind screen. + +Prediction, operator judgment, gold, admission, and settlement stay separate. +This module does not settle a target, admit an identity, or authorize SELECT-005. +""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +from datetime import datetime +from pathlib import Path +from typing import Any, Iterable, Mapping, Sequence + +from hyperlexical.holdout_guard import normalized_text_sha256 + +SAMPLE_SCHEMA = "hyperlex.unbind_screen_sample_row.v1" +PREDICTION_SCHEMA = "hyperlex.unbind_screen_prediction.v1" +REVIEW_SCHEMA = "hyperlex.unbind_screen_review_row.v1" +LABEL_SCHEMA = "hyperlex.unbind_screen_operator_label.v1" +REPORT_SCHEMA = "hyperlex.unbind_screen_report.v1" +RECEIPT_SCHEMA = "hyperlex.unbind_screen_evaluation_receipt.v1" + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v3" +BUCKETS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE") +OPERATOR_BUCKETS = BUCKETS + ("UNRESOLVED",) +RAW_BUCKETS = {"HIGH_VALUE": "HIGH", "HIGH": "HIGH", "SECONDARY": "SECONDARY", "REJECT": "REJECT", "QUARANTINE": "QUARANTINE"} +REASONS = { + "HIGH": frozenset({"STRONG_IDIOM", "PHRASAL_BINDING", "FIXED_NONLITERAL", "VARIABLE_SLOT", "CONVENTIONALIZED_SHIFT"}), + "SECONDARY": frozenset({"LEXICALIZED_TRANSPARENT", "FIXED_COMPOSITIONAL", "DOMAIN_LEXICALIZED"}), + "REJECT": frozenset({ + "PERSON_NAME", "ORGANIZATION", "TITLE_OR_DESIGNATION", "TAXONOMY", + "SPECIES_COMMON_NAME", "TECHNICAL_PROCEDURE", "TECHNICAL_MEASUREMENT", + "PRODUCTIVE_NUMBER", "FREE_COMPOSITION", "REFERENTIAL_DOMINANCE", + }), + "QUARANTINE": frozenset({"DIALECTAL", "ARCHAIC", "AMBIGUOUS_SENSE", "PROVENANCE_UNCLEAR"}), + "UNRESOLVED": frozenset({"INSUFFICIENT_SIGNAL"}), +} +STATES = ( + "DRAFT", + "SAMPLE_FROZEN", + "PREDICTIONS_FROZEN", + "OPERATOR_LABELING", + "LABELS_FROZEN", + "SCORED", + "ERROR_ANALYZED", + "REVISION_ELIGIBLE", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", +}) + + +class ScreenEvalError(ValueError): + pass + + +def refuse(message: str) -> None: + raise ScreenEvalError(message) + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def canonical_bucket(raw: str) -> str: + bucket = RAW_BUCKETS.get(str(raw or "")) + if bucket is None: + refuse(f"prediction bucket {raw!r} is not a screen bucket") + return bucket + + +def _load_jsonl(path: Path) -> list[dict[str, Any]]: + rows = [] + for line in path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + rows.append(json.loads(line)) + return rows + + +def _dump_jsonl(path: Path, rows: Sequence[Mapping[str, Any]]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _parse_time(value: str) -> datetime: + text = str(value or "").strip() + if text.endswith("Z"): + text = text[:-1] + "+00:00" + try: + parsed = datetime.fromisoformat(text) + except ValueError as exc: + refuse(f"timestamp {value!r} is not ISO-8601") + raise exc + if parsed.tzinfo is None: + refuse(f"timestamp {value!r} needs a timezone") + return parsed + + +def _surfaces(path: Path) -> set[str]: + return {line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip()} + + +def _train_ids(path: Path) -> set[str]: + found = set() + for row in _load_jsonl(path): + text = str(row.get("text") or row.get("surface") or "") + if text: + found.add(normalized_text_sha256(text)) + return found + + +def _identity_rows(frozen: Sequence[Mapping[str, Any]], *, evaluation_id: str, sample_id: str, sample_sha: str) -> tuple[list[dict[str, Any]], list[dict[str, Any]], list[dict[str, Any]]]: + samples, predictions, reviews = [], [], [] + seen: set[str] = set() + for raw in frozen: + surface = str(raw.get("text") or "") + tokens = list(raw.get("tokens") or []) + if " ".join(str(tok) for tok in tokens) != surface: + refuse(f"tokens do not reconstruct {surface!r}") + row_id = normalized_text_sha256(surface) + if row_id in seen: + refuse(f"duplicate row identity {row_id}") + seen.add(row_id) + provenance = {"source": "wordnet-3.0", "sample_sha256": sample_sha} + samples.append({ + "schema": SAMPLE_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "row_id": row_id, + "surface": surface, + "pos": raw.get("source_pos"), + "token_count": len(tokens), + "provenance": provenance, + }) + predictions.append({ + "schema": PREDICTION_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "row_id": row_id, + "bucket": canonical_bucket(str(raw.get("predicted") or "")), + "bucket_raw": raw.get("predicted"), + "relation": raw.get("rule"), + "rule_version": RULE_VERSION, + "provenance": provenance, + }) + reviews.append({ + "schema": REVIEW_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "row_id": row_id, + "surface": surface, + "pos": raw.get("source_pos"), + "token_count": len(tokens), + "gloss": raw.get("gloss") or "", + "provenance": provenance, + }) + samples.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + reviews.sort(key=lambda row: row["row_id"]) + return samples, predictions, reviews + + +def _assert_blind(rows: Sequence[Mapping[str, Any]]) -> None: + for row in rows: + leaked = LEAK_KEYS.intersection(row) + if leaked: + refuse(f"blind review carries {sorted(leaked)}") + + +def _assert_disjoint(heldout: Iterable[str], blocked: set[str], name: str) -> None: + overlap = sorted(set(heldout) & blocked) + if overlap: + refuse(f"held-out row is also in {name}") + + +def materialize( + frozen_sample: Path, + out_dir: Path, + *, + expected_sha256: str, + development: Path, + validation_development: Path, + train_jsonl: Path | None = None, + evaluation_id: str = "HLX-EVAL-UNBIND-SCREEN-V3-001", + sample_id: str = "heldout-001", + frozen_at: str = "2026-09-27T22:57:17Z", +) -> dict[str, Any]: + """Copy a frozen prediction file into separate sample, prediction, and blind review artifacts.""" + digest = file_sha256(frozen_sample) + if digest != expected_sha256: + refuse("frozen sample hash does not match the expected seal") + frozen = _load_jsonl(frozen_sample) + samples, predictions, reviews = _identity_rows( + frozen, evaluation_id=evaluation_id, sample_id=sample_id, sample_sha=digest, + ) + _assert_blind(reviews) + held_ids = {row["row_id"] for row in samples} + dev_ids = {normalized_text_sha256(text) for text in _surfaces(development)} + val_ids = {normalized_text_sha256(text) for text in _surfaces(validation_development)} + _assert_disjoint(held_ids, dev_ids, "development") + _assert_disjoint(held_ids, val_ids, "validation_development") + train_checked = train_jsonl is not None + if train_jsonl is not None: + _assert_disjoint(held_ids, _train_ids(train_jsonl), "training_gold") + out = Path(out_dir) + sample_sha = _dump_jsonl(out / "samples" / f"{sample_id}.jsonl", samples) + prediction_sha = _dump_jsonl(out / "predictions" / f"{sample_id}.predictions.jsonl", predictions) + review_sha = _dump_jsonl(out / "operator" / f"{sample_id}.review.jsonl", reviews) + if file_sha256(frozen_sample) != digest: + refuse("frozen sample changed during materialize") + receipt = { + "schema": RECEIPT_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "state": "PREDICTIONS_FROZEN", + "rule_version": RULE_VERSION, + "sample_sha256": digest, + "sample_artifact_sha256": sample_sha, + "prediction_artifact_sha256": prediction_sha, + "blind_review_sha256": review_sha, + "operator_label_sha256": None, + "sample_frozen_at": frozen_at, + "development_rows": len(dev_ids), + "validation_development_rows": len(val_ids), + "held_out_rows": len(samples), + "v3_application_count": 1, + "hand_corrections": 0, + "operator_labels": "pending", + "held_out_precision": "NOT_COMPUTABLE", + "confusion_matrix": "NOT_COMPUTABLE", + "train_gold_checked": train_checked, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + } + (out / "reports").mkdir(parents=True, exist_ok=True) + _write_json(out / "reports" / f"{sample_id}.receipt.json", receipt) + return receipt + + +def _write_json(path: Path, body: Mapping[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(body, indent=2, sort_keys=True) + "\n", encoding="utf-8") + + +def _receipt(out_dir: Path, sample_id: str) -> dict[str, Any]: + path = out_dir / "reports" / f"{sample_id}.receipt.json" + return json.loads(path.read_text(encoding="utf-8")) + + +def _check_hashes(out_dir: Path, receipt: Mapping[str, Any]) -> None: + sample_id = str(receipt["sample_id"]) + pairs = { + "sample_artifact_sha256": out_dir / "samples" / f"{sample_id}.jsonl", + "prediction_artifact_sha256": out_dir / "predictions" / f"{sample_id}.predictions.jsonl", + "blind_review_sha256": out_dir / "operator" / f"{sample_id}.review.jsonl", + } + for field, path in pairs.items(): + if file_sha256(path) != receipt[field]: + refuse(f"{field} changed after freeze") + _assert_blind(_load_jsonl(pairs["blind_review_sha256"])) + + +def validate_label(row: Mapping[str, Any], *, frozen_at: str) -> dict[str, Any]: + bucket = str(row.get("operator_bucket") or "") + if bucket not in OPERATOR_BUCKETS: + refuse(f"operator bucket {bucket!r} is not allowed") + reason = str(row.get("reason_code") or "") + if reason not in REASONS[bucket]: + refuse(f"reason {reason!r} is not valid for {bucket}") + row_id = str(row.get("row_id") or "") + if len(row_id) != 64: + refuse("operator label row_id must be the immutable identity") + labeled_at = str(row.get("labeled_at") or "") + if _parse_time(labeled_at) <= _parse_time(frozen_at): + refuse("operator label timestamp is not after the sample freeze") + return { + "schema": LABEL_SCHEMA, + "evaluation_id": row.get("evaluation_id"), + "row_id": row_id, + "operator_bucket": bucket, + "reason_code": reason, + "note": row.get("note"), + "labeled_at": labeled_at, + } + + +def freeze_labels(out_dir: Path, labels_path: Path, *, sample_id: str = "heldout-001") -> dict[str, Any]: + """Store operator labels beside the frozen predictions. Does not score or settle.""" + out_dir = Path(out_dir) + receipt = _receipt(out_dir, sample_id) + if receipt.get("state") not in {"PREDICTIONS_FROZEN", "OPERATOR_LABELING", "LABELS_FROZEN"}: + refuse(f"cannot freeze labels from state {receipt.get('state')}") + _check_hashes(out_dir, receipt) + predictions = _load_jsonl(out_dir / "predictions" / f"{sample_id}.predictions.jsonl") + expected = {row["row_id"] for row in predictions} + cleaned = [validate_label(row, frozen_at=str(receipt["sample_frozen_at"])) for row in _load_jsonl(labels_path)] + got = [row["row_id"] for row in cleaned] + if len(got) != len(set(got)): + refuse("operator labels repeat a row identity") + if set(got) != expected: + refuse("operator labels do not cover the frozen rows exactly once") + cleaned.sort(key=lambda row: row["row_id"]) + label_sha = _dump_jsonl(out_dir / "operator" / f"{sample_id}.labels.jsonl", cleaned) + receipt["operator_label_sha256"] = label_sha + receipt["operator_labels"] = "frozen" + receipt["state"] = "LABELS_FROZEN" + receipt["held_out_precision"] = "NOT_COMPUTABLE" + receipt["confusion_matrix"] = "NOT_COMPUTABLE" + receipt["select_authorized"] = False + _write_json(out_dir / "reports" / f"{sample_id}.receipt.json", receipt) + return receipt + + +def _metrics(pairs: Sequence[tuple[str, str, str, str]]) -> dict[str, Any]: + """pairs are (row_id, predicted, operator, reason).""" + resolved = [item for item in pairs if item[2] != "UNRESOLVED"] + unresolved = [item for item in pairs if item[2] == "UNRESOLVED"] + matrix = {op: {pred: 0 for pred in BUCKETS} for op in OPERATOR_BUCKETS} + for _row_id, pred, op, _reason in pairs: + matrix[op][pred] += 1 + per_bucket = {} + for bucket in BUCKETS: + tp = sum(1 for _i, pred, op, _r in resolved if pred == bucket and op == bucket) + fp = sum(1 for _i, pred, op, _r in resolved if pred == bucket and op != bucket) + fn = sum(1 for _i, pred, op, _r in resolved if op == bucket and pred != bucket) + precision = None if tp + fp == 0 else tp / (tp + fp) + recall = None if tp + fn == 0 else tp / (tp + fn) + f1 = None + if precision is not None and recall is not None and precision + recall: + f1 = 2 * precision * recall / (precision + recall) + per_bucket[bucket] = { + "precision": precision, + "recall": recall, + "f1": f1, + "support": tp + fn, + } + accuracy = None if not resolved else sum(1 for _i, pred, op, _r in resolved if pred == op) / len(resolved) + reasons: dict[str, int] = {} + for _i, _p, _o, reason in pairs: + reasons[reason] = reasons.get(reason, 0) + 1 + return { + "overall_accuracy": accuracy, + "per_bucket": per_bucket, + "confusion_matrix": matrix, + "reason_code_distribution": reasons, + "unresolved_count": len(unresolved), + "resolved_count": len(resolved), + } + + +def _error_class(predicted: str, operator: str) -> str | None: + if operator == "UNRESOLVED" or predicted == operator: + return None + return { + "HIGH": "false_high", + "SECONDARY": "false_secondary", + "REJECT": "false_reject", + "QUARANTINE": "false_quarantine", + }[predicted] + + +def score(out_dir: Path, *, sample_id: str = "heldout-001") -> dict[str, Any]: + """Join labels to predictions on row_id. Pending labels stay NOT_COMPUTABLE.""" + out_dir = Path(out_dir) + receipt = _receipt(out_dir, sample_id) + _check_hashes(out_dir, receipt) + label_path = out_dir / "operator" / f"{sample_id}.labels.jsonl" + report: dict[str, Any] = { + "schema": REPORT_SCHEMA, + "evaluation_id": receipt["evaluation_id"], + "sample_id": sample_id, + "rule_version": RULE_VERSION, + "select_authorized": False, + "revision_eligible": False, + "admitted": 0, + "settled": 0, + "gold": 0, + } + if receipt.get("operator_labels") != "frozen" or not label_path.is_file(): + report["status"] = "NOT_COMPUTABLE" + report["held_out_precision"] = "NOT_COMPUTABLE" + report["confusion_matrix"] = "NOT_COMPUTABLE" + report["overall_accuracy"] = "NOT_COMPUTABLE" + _write_json(out_dir / "reports" / f"{sample_id}.metrics.json", report) + return report + if file_sha256(label_path) != receipt.get("operator_label_sha256"): + refuse("operator artifact hash changed after label freeze") + predictions = {row["row_id"]: row for row in _load_jsonl(out_dir / "predictions" / f"{sample_id}.predictions.jsonl")} + samples = {row["row_id"]: row for row in _load_jsonl(out_dir / "samples" / f"{sample_id}.jsonl")} + labels = {row["row_id"]: row for row in _load_jsonl(label_path)} + if set(labels) != set(predictions): + refuse("join row identities do not match") + pairs = [] + errors = [] + classes = {name: 0 for name in ("false_high", "false_secondary", "false_reject", "false_quarantine")} + for row_id in sorted(predictions): + pred = predictions[row_id] + label = labels[row_id] + kind = _error_class(pred["bucket"], label["operator_bucket"]) + pairs.append((row_id, pred["bucket"], label["operator_bucket"], label["reason_code"])) + if kind: + classes[kind] += 1 + errors.append({ + "row_id": row_id, + "surface": samples[row_id]["surface"], + "predicted_bucket": pred["bucket"], + "operator_bucket": label["operator_bucket"], + "prediction_relation": pred["relation"], + "operator_reason": label["reason_code"], + "error_class": kind, + }) + metrics = _metrics(pairs) + report.update(metrics) + report["status"] = "SCORED" + report["error_classes"] = classes + report["select_authorized"] = False + _write_json(out_dir / "reports" / f"{sample_id}.metrics.json", report) + _dump_jsonl(out_dir / "reports" / f"{sample_id}.errors.jsonl", errors) + with (out_dir / "reports" / f"{sample_id}.confusion.csv").open("w", encoding="utf-8", newline="") as handle: + writer = csv.writer(handle) + writer.writerow(["operator_bucket", "predicted_bucket", "count"]) + for operator, cols in metrics["confusion_matrix"].items(): + for predicted, count in cols.items(): + writer.writerow([operator, predicted, count]) + receipt["state"] = "SCORED" + receipt["held_out_precision"] = "SCORED" + receipt["confusion_matrix"] = "SCORED" + receipt["select_authorized"] = False + receipt["revision_eligible"] = False + _write_json(out_dir / "reports" / f"{sample_id}.receipt.json", receipt) + return report + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Score a frozen unbind screen without settling it.") + sub = parser.add_subparsers(dest="command", required=True) + mat = sub.add_parser("materialize") + mat.add_argument("--frozen-sample", type=Path, required=True) + mat.add_argument("--expected-sha", required=True) + mat.add_argument("--development", type=Path, required=True) + mat.add_argument("--validation-development", type=Path, required=True) + mat.add_argument("--train-jsonl", type=Path) + mat.add_argument("--out", type=Path, required=True) + mat.add_argument("--evaluation-id", default="HLX-EVAL-UNBIND-SCREEN-V3-001") + mat.add_argument("--frozen-at", default="2026-09-27T22:57:17Z") + sc = sub.add_parser("score") + sc.add_argument("--evaluation", type=Path, required=True) + fr = sub.add_parser("freeze-labels") + fr.add_argument("--evaluation", type=Path, required=True) + fr.add_argument("--labels", type=Path, required=True) + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + args = build_parser().parse_args(argv) + try: + if args.command == "materialize": + receipt = materialize( + args.frozen_sample, + args.out, + expected_sha256=args.expected_sha, + development=args.development, + validation_development=args.validation_development, + train_jsonl=args.train_jsonl, + evaluation_id=args.evaluation_id, + frozen_at=args.frozen_at, + ) + elif args.command == "freeze-labels": + receipt = freeze_labels(args.evaluation, args.labels) + else: + receipt = score(args.evaluation) + except ScreenEvalError as exc: + raise SystemExit(f"REFUSE: {exc}") from exc + print(json.dumps(receipt, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/shadow/hyperlexical/semantic_compositionality_residual.py b/scripts/shadow/hyperlexical/semantic_compositionality_residual.py new file mode 100644 index 00000000..6287e027 --- /dev/null +++ b/scripts/shadow/hyperlexical/semantic_compositionality_residual.py @@ -0,0 +1,564 @@ +"""Continuous semantic residual for one bound PWN 3.0 sense. + +The score compares the supplied synset with a normalized mean of resolved +constituent synsets. It is not a yes/no label, and it does not select a source. +""" + +from __future__ import annotations + +import hashlib +import math +import struct +from decimal import Decimal, ROUND_HALF_EVEN + +from hyperlexical.km_candidate_evaluation import lookup_key + +CANDIDATE = "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1" +COMPOSITION_OPERATOR = "normalized_mean_v1" +DISTANCE_METRIC = "one_minus_cosine_v1" +MODEL_NAME = "sentence-transformers/all-MiniLM-L6-v2" +MODEL_REVISION = "1110a243fdf4706b3f48f1d95db1a4f5529b4d41" +RESIDUAL_QUANTUM = Decimal("0.0000000001") +RELATION_SYMBOLS = frozenset({"+", "\\"}) +MIN_CONTENT_CONSTITUENTS = 2 +EXTRACTED = "EXTRACTED" +UNKNOWN = "UNKNOWN" +EXACT = "EXACT" +UNIQUE = "UNIQUE" +AMBIGUOUS = "AMBIGUOUS" +UNRESOLVED = "UNRESOLVED" +SCORED = "SCORED" +RESOLVED = frozenset({EXACT, UNIQUE}) + +STRUCTURAL_TOKENS = frozenset( + { + "a", + "about", + "across", + "after", + "against", + "all", + "amid", + "among", + "amongst", + "an", + "and", + "any", + "are", + "as", + "at", + "be", + "been", + "before", + "behind", + "being", + "below", + "beside", + "between", + "beyond", + "both", + "but", + "by", + "can", + "could", + "did", + "do", + "does", + "down", + "during", + "each", + "every", + "except", + "for", + "from", + "had", + "has", + "have", + "her", + "his", + "if", + "in", + "into", + "is", + "it", + "its", + "least", + "less", + "may", + "might", + "more", + "most", + "must", + "my", + "no", + "none", + "nor", + "not", + "of", + "off", + "on", + "one's", + "onto", + "or", + "our", + "out", + "over", + "per", + "shall", + "should", + "some", + "than", + "that", + "the", + "their", + "them", + "then", + "these", + "this", + "those", + "through", + "to", + "toward", + "towards", + "under", + "up", + "upon", + "via", + "versus", + "was", + "were", + "will", + "with", + "within", + "without", + "would", + "your", + } +) + + +_RULE_AMBIGUOUS = ( + "A content constituent with two or more exact targets, or with no exact " + "target and two or more lexical synsets, abstains the row." +) +_RULE_SHORT = ( + "Whitespace tokenization must leave at least two tokens outside the frozen structural class." +) +_RULE_OVERFLOW = ( + "A whole-sense or constituent text longer than the pinned max sequence length " + "abstains the row. The encoder must not truncate it." +) +_RULE_UNRESOLVED = "A content constituent with no exact target and no lexical synset abstains the row." +_RULE_ZERO = "A non-finite or zero encoder vector, or a zero composed vector, abstains the row." +_HIGH_PROXY = "a proxy for expected noncompositionality" +_SECONDARY_PROXY = "a proxy for expected compositionality" +_REJECT_AXIS = "a referential axis, not semantic no" +_QUARANTINE_AXIS = "not semantic evidence" +_PERCENTILE_RULE = ( + "linear interpolation at rank (n-1)*(p/100), then round-half-even to 10 decimal places" +) +_COMPOSITION_PROCEDURE = ( + "L2-normalize each constituent vector in binary64, take the arithmetic mean, " + "then L2-normalize that mean" +) +_DUPLICATE_SYNSETS = "kept once per content token" +_HYPHEN_RULE = "kept as one token" +_MEMBERSHIP_RULE = "lookup_key of the whitespace token is in the structural class" +_EXACT_RULE = ( + "one synset reached by a derivation or pertainym pointer whose target word " + "number is nonzero and whose target lemma matches the constituent or a one-hop " + "exception neighbor" +) +_AMBIGUOUS_RULE = ( + "more than one exact target synset, or no exact target and more than one lexical synset" +) +_LEXICAL_SCOPE = ( + "one lookup key plus one-hop bidirectional exception neighbors, across noun, verb, adj, and adv" +) +_SOURCE_WORD = "not a filter" +_TARGET_ZERO = "does not name a lemma" +_UNIQUE_RULE = "no exact target, and the lexical keys name one synset" +_UNRESOLVED_RULE = "no exact target and no lexical synset" +_DISTANCE_FORMULA = "1 - binary64_dot(l2_normalize(whole), l2_normalize(composed))" +_DISTANCE_RECORD = "format(value, '.10f'), round-half-even, no clamp" +_TEMPLATE = "{surface} ({pos}): {gloss}" +_TEXT_NORMALIZATION = "underscores become spaces; whitespace collapses; gloss is the supplied first clause" +_VECTOR_HASH_RULE = "sha256 of little-endian binary32 bytes in dimension order" +_WHOLE_SENSE_RULE = "the supplied Hyperlex surface, POS, and frozen gloss; no substitute synset" + + +def candidate_policy() -> dict: + """Return the preregistered design. The dict has no row scores.""" + return { + "abstention_priority": [ + "fewer_than_two_content_constituents", + "leftmost_unresolved_or_ambiguous_content_constituent", + "representation_exceeds_max_sequence_length", + "zero_vector", + ], + "abstention_rules": { + "ambiguous_content_constituent": _RULE_AMBIGUOUS, + "fewer_than_two_content_constituents": _RULE_SHORT, + "representation_exceeds_max_sequence_length": _RULE_OVERFLOW, + "unresolved_content_constituent": _RULE_UNRESOLVED, + "zero_vector": _RULE_ZERO, + }, + "analysis_plan": { + "high_comparison_if_no_high_row_is_scored": "NOT_COMPUTABLE", + "high_is": _HIGH_PROXY, + "join_operator_labels_only_after_score_artifact_is_hashed": True, + "operator_labels_are_scoring_inputs": False, + "percentile": _PERCENTILE_RULE, + "primary_bucket": "HIGH", + "quarantine_is": _QUARANTINE_AXIS, + "reject_is": _REJECT_AXIS, + "secondary_is": _SECONDARY_PROXY, + "semantic_noncompositionality_threshold": None, + }, + "candidate": CANDIDATE, + "composition_operator": { + "duplicate_constituent_synsets": _DUPLICATE_SYNSETS, + "name": COMPOSITION_OPERATOR, + "procedure": _COMPOSITION_PROCEDURE, + "weights": None, + }, + "constituent_extraction": { + "content_minimum": MIN_CONTENT_CONSTITUENTS, + "hyphenated_token": _HYPHEN_RULE, + "membership": _MEMBERSHIP_RULE, + "structural_tokens": sorted(STRUCTURAL_TOKENS), + "tokenizer": "str.split", + }, + "constituent_sense_resolution": { + "ambiguous": _AMBIGUOUS_RULE, + "exact": _EXACT_RULE, + "forbidden": [ + "arbitrary_first_sense", + "embedding_similarity", + "gloss_similarity", + "language_model_judge", + "manual_selection", + "operator_labels", + ], + "lexical_scope": _LEXICAL_SCOPE, + "source_word_number": _SOURCE_WORD, + "target_word_number_zero": _TARGET_ZERO, + "unique": _UNIQUE_RULE, + "unresolved": _UNRESOLVED_RULE, + }, + "distance_metric": { + "formula": _DISTANCE_FORMULA, + "name": DISTANCE_METRIC, + "record": _DISTANCE_RECORD, + }, + "emits_yes_no": False, + "model_identity": { + "name": MODEL_NAME, + "revision": MODEL_REVISION, + }, + "representation_template": _TEMPLATE, + "representation_text_normalization": _TEXT_NORMALIZATION, + "row_unknown_if_any_required_content_constituent_is_not_exact_or_unique": True, + "semantic_noncompositionality_threshold": None, + "status_rule": { + "scored_count_eq_0": { + "next_legal_transition": "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + "status": "CANDIDATE_INSUFFICIENT", + }, + "scored_count_gt_0": { + "next_legal_transition": "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION", + "status": "CANDIDATE_DISTRIBUTION_FROZEN", + }, + }, + "vector_hash": _VECTOR_HASH_RULE, + "whole_sense": _WHOLE_SENSE_RULE, + } + + +def extract_constituents(surface: str) -> dict: + """Split a surface on whitespace and apply the frozen structural class.""" + tokens = surface.split() + content = [] + structural = [] + for token in tokens: + if lookup_key(token) in STRUCTURAL_TOKENS: + structural.append(token) + else: + content.append(token) + status = EXTRACTED if len(content) >= MIN_CONTENT_CONSTITUENTS else UNKNOWN + return { + "constituent_extraction_status": status, + "content_constituents": content, + "ignored_structural_tokens": structural, + "surface_tokens": tokens, + } + + +def neighbor_keys(text: str, exceptions: dict[str, set[str]]) -> set[str]: + """Return the lookup key plus one-hop exception neighbors.""" + own = lookup_key(text) + keys = {own} + raw = text.casefold().replace("\u2019", "'").replace("\u2018", "'").replace("`", "'") + related: set[str] = set() + for form in (raw, own, own.replace("_", " ")): + related.update(exceptions.get(form, ())) + for item in related: + keys.add(lookup_key(item)) + return keys + + +def exact_synset_ids( + pointers: list[tuple[str, int, str, str]], + constituent: str, + exceptions: dict[str, set[str]], +) -> list[str]: + """Collect derivation and pertainym targets that name this constituent.""" + keys = neighbor_keys(constituent, exceptions) + found = [] + for symbol, target_word, synset_id, lemma in pointers: + if symbol not in RELATION_SYMBOLS or target_word <= 0 or not lemma: + continue + if lookup_key(lemma) not in keys: + continue + found.append(synset_id) + return found + + +def lexical_synset_ids( + index: dict[str, list[str]], + constituent: str, + exceptions: dict[str, set[str]], +) -> list[str]: + """Collect synsets named by the constituent key or one exception hop.""" + found = [] + for key in sorted(neighbor_keys(constituent, exceptions)): + found.extend(index.get(key, ())) + return found + + +def resolve_constituent(exact_synset_ids_found: list[str], lexical_synset_ids_found: list[str]) -> str: + """Resolve one constituent. Exact evidence outranks lemma polysemy.""" + exact = list(dict.fromkeys(exact_synset_ids_found)) + if len(exact) == 1: + return EXACT + if len(exact) > 1: + return AMBIGUOUS + lexical = list(dict.fromkeys(lexical_synset_ids_found)) + if len(lexical) == 1: + return UNIQUE + if len(lexical) > 1: + return AMBIGUOUS + return UNRESOLVED + + +def resolved_synset(exact_synset_ids_found: list[str], lexical_synset_ids_found: list[str]) -> str | None: + status = resolve_constituent(exact_synset_ids_found, lexical_synset_ids_found) + if status == EXACT: + return list(dict.fromkeys(exact_synset_ids_found))[0] + if status == UNIQUE: + return list(dict.fromkeys(lexical_synset_ids_found))[0] + return None + + +def select_lemma(lemmas: list[str], constituent: str, exceptions: dict[str, set[str]]) -> str: + """Pick one lemma. Matching keys win, and lookup order breaks remaining ties.""" + keys = neighbor_keys(constituent, exceptions) + matches = [lemma for lemma in lemmas if lookup_key(lemma) in keys] + pool = matches or list(lemmas) + if not pool: + raise RuntimeError("resolved synset has no lemma") + return min(pool, key=lookup_key) + + +def representation_text(surface: str, pos: str, gloss: str) -> str: + shown = " ".join(surface.replace("_", " ").split()) + gloss_text = " ".join(gloss.split()) + return f"{shown} ({pos}): {gloss_text}" + + +def l2_normalize(values: list[float]) -> list[float] | None: + if not values or any(not math.isfinite(value) for value in values): + return None + norm = math.sqrt(sum(value * value for value in values)) + if not math.isfinite(norm) or norm == 0.0: + return None + return [value / norm for value in values] + + +def composed_vector(parts: list[list[float]]) -> list[float] | None: + """Normalized mean. Each input vector is normalized again in binary64.""" + if len(parts) < MIN_CONTENT_CONSTITUENTS: + return None + width = len(parts[0]) + if any(len(part) != width for part in parts): + raise RuntimeError("constituent vectors differ in width") + normalized = [] + for part in parts: + unit = l2_normalize(part) + if unit is None: + return None + normalized.append(unit) + count = float(len(normalized)) + mean = [sum(part[index] for part in normalized) / count for index in range(width)] + return l2_normalize(mean) + + +def format_residual(dot: float) -> str: + value = 1.0 - dot + if not math.isfinite(value): + raise RuntimeError("non-finite residual") + if value == 0.0: + value = 0.0 + return format(value, ".10f") + + +def residual_score(whole: list[float], constituents: list[list[float]]) -> tuple[str, list[float]] | None: + """Return the 10-decimal residual and the composed vector.""" + whole_unit = l2_normalize(whole) + composed = composed_vector(constituents) + if whole_unit is None or composed is None: + return None + dot = sum(left * right for left, right in zip(whole_unit, composed)) + if not math.isfinite(dot): + return None + return format_residual(dot), composed + + +def vector_hash(values: list[float]) -> str: + blob = b"".join(struct.pack(" str | None: + if extraction_status != EXTRACTED or len(resolutions) < MIN_CONTENT_CONSTITUENTS: + return "fewer_than_two_content_constituents" + for status in resolutions: + if status == AMBIGUOUS: + return "ambiguous_content_constituent" + if status == UNRESOLVED: + return "unresolved_content_constituent" + if status not in RESOLVED: + raise RuntimeError(f"unknown resolution status {status}") + return None + + +def score_record( + *, + row_id: str, + surface: str, + pos: str, + synset: str, + extraction: dict, + resolutions: list[str], + resolved_synsets: list[str | None], + resolved_lemmas: list[str | None], + whole_representation: str | None, + constituent_representations: list[str] | None, + whole_vector: list[float] | None, + constituent_vectors: list[list[float]] | None, + candidate_spec_sha256: str, + sequence_overflow: bool = False, +) -> dict: + """Score one row. Operator labels are not parameters.""" + reason = primary_abstention(extraction["constituent_extraction_status"], resolutions) + if reason is None and sequence_overflow: + reason = "representation_exceeds_max_sequence_length" + residual = None + whole_hash = None + constituent_hashes = None + composed_hash = None + if reason is None: + if whole_vector is None or constituent_vectors is None: + raise RuntimeError("a resolvable row has no vectors") + if len(constituent_vectors) != len(resolutions): + raise RuntimeError("constituent vector count does not match resolutions") + scored = residual_score(whole_vector, constituent_vectors) + whole_hash = vector_hash(whole_vector) + constituent_hashes = [vector_hash(vector) for vector in constituent_vectors] + if scored is None: + reason = "zero_vector" + else: + residual, composed = scored + composed_hash = vector_hash(composed) + elif whole_vector is not None or constituent_vectors is not None: + raise RuntimeError("an abstaining row was encoded") + extracted = extraction["constituent_extraction_status"] == EXTRACTED + return { + "candidate_spec_sha256": candidate_spec_sha256, + "composition_operator": COMPOSITION_OPERATOR, + "composed_vector_hash": composed_hash, + "constituent_extraction_status": extraction["constituent_extraction_status"], + "constituent_representations": constituent_representations if residual is not None else None, + "constituent_resolution_status": list(resolutions), + "constituent_vector_hashes": constituent_hashes, + "content_constituents": list(extraction["content_constituents"]), + "distance_metric": DISTANCE_METRIC, + "ignored_structural_tokens": list(extraction["ignored_structural_tokens"]), + "pos": pos, + "primary_abstention_reason": reason, + "residual_score": residual, + "resolved_constituent_lemmas": list(resolved_lemmas) if extracted else [], + "resolved_constituent_synsets": list(resolved_synsets) if extracted else [], + "row_id": row_id, + "score_status": SCORED if residual is not None else UNKNOWN, + "surface": surface, + "surface_tokens": list(extraction["surface_tokens"]), + "synset": synset, + "whole_representation": whole_representation if residual is not None else None, + "whole_vector_hash": whole_hash, + } + + +def evaluation_status(scored_count: int) -> tuple[str, str]: + """Map the scored count to a status. The count is not a label agreement.""" + if scored_count < 0: + raise RuntimeError("negative scored count") + if scored_count == 0: + return "CANDIDATE_INSUFFICIENT", "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION" + return "CANDIDATE_DISTRIBUTION_FROZEN", "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION" + + +def percentile(sorted_values: list[str], percent: int) -> str: + """Linear interpolation on already-sorted 10-decimal residual strings.""" + count = len(sorted_values) + if count == 0: + raise RuntimeError("percentile of an empty sample") + if count == 1: + return sorted_values[0] + rank = Decimal(count - 1) * (Decimal(percent) / Decimal(100)) + low = int(rank) + high = min(low + 1, count - 1) + weight = rank - Decimal(low) + blended = Decimal(sorted_values[low]) + (Decimal(sorted_values[high]) - Decimal(sorted_values[low])) * weight + return str(blended.quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def decimal_mean(values: list[str]) -> str: + total = sum((Decimal(value) for value in values), start=Decimal(0)) + mean = total / Decimal(len(values)) + return str(mean.quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def distribution(scores: list[str]) -> dict: + if not scores: + return { + "count": 0, + "max": None, + "mean": None, + "median": None, + "min": None, + "p25": None, + "p75": None, + "status": "NOT_COMPUTABLE", + } + ordered = sorted(scores, key=Decimal) + return { + "count": len(ordered), + "max": ordered[-1], + "mean": decimal_mean(ordered), + "median": percentile(ordered, 50), + "min": ordered[0], + "p25": percentile(ordered, 25), + "p75": percentile(ordered, 75), + "status": "DESCRIPTIVE", + } diff --git a/scripts/shadow/hyperlexical/unbind_screen_v4.py b/scripts/shadow/hyperlexical/unbind_screen_v4.py new file mode 100644 index 00000000..c8de7e1a --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v4.py @@ -0,0 +1,560 @@ +"""Secondary-only coverage patch over a frozen v3 bucket. + +Input is the v3 bucket, the surface, and the first-sense gloss. +A row is inspected only when that bucket is SECONDARY. +Patch A may move SECONDARY to REJECT. Patch B may move SECONDARY to HIGH. +Existing HIGH and REJECT decisions are returned unchanged. +""" + +from __future__ import annotations + +import ast +import re +import unicodedata +from pathlib import Path + +from hyperlexical.unbind_screen_v3 import NUMBERS, PARTICLES, STOP, stems + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v4" +PATCH_A = ( + "multi_token_person_name", + "organization_from_gloss", + "species_or_common_name_referent", + "medical_technical_expression", + "productive_number", +) +PATCH_B = ( + "nonliteral_semantic_shift", + "conventionalized_idiom", + "noncompositional_phrasal_binding", + "fixed_lexicalized_expression", +) +_CANONICAL = { + "HIGH_VALUE": "HIGH", + "HIGH": "HIGH", + "SECONDARY": "SECONDARY", + "REJECT": "REJECT", + "QUARANTINE": "QUARANTINE", +} +_HYPHEN = re.compile(r"(?<=\w)[\u2010\u2011\u2012\u2013\u2014-](?=\w)") +_PUNCT = re.compile(r"[^\w\s']+", re.UNICODE) +_WORD = re.compile(r"[A-Za-z]+") +_DELIMITERS = frozenset({ + "with", "that", "which", "used", "yielding", "having", "resulting", "who", "whose", +}) +_MULTIPLIERS = frozenset({"times", "fold"}) +_CLINICAL = frozenset({ + "impairment", "disease", "disorder", "syndrome", "inflammation", "symptom", + "lesion", "pathology", "hemorrhage", "haemorrhage", "infection", "paralysis", + "fracture", "tumor", "tumour", "carcinoma", "edema", "oedema", "surgery", + "surgical", "clinical", +}) +_COMPARATIVES = frozenset({ + "better", "worse", "greater", "lesser", "more", "less", "higher", "lower", + "bigger", "smaller", "older", "younger", "sooner", "later", "richer", "poorer", + "well", +}) +_INTENSIFIERS = frozenset({ + "bone", "brand", "stone", "rock", "pitch", "crystal", "soaking", "dripping", + "stark", "dirt", "stock", "wide", +}) +_LIGHT_VERBS = frozenset({"give", "take", "have", "make", "get"}) +_DETERMINERS = frozenset({"a", "an", "the"}) +_LIFE = frozenset({"noun.animal", "noun.plant"}) + +SUCCESS_CRITERIA = { + "schema": "hyperlex.unbind_screen_v4_success_criteria.v1", + "high_precision_floor": 1.0, + "reject_precision_floor": 1.0, + "false_high_allowed": 0, + "false_reject_allowed": 0, + "false_secondary_rate_must_be_strictly_below": "13/29", + "gate_b_violations_allowed": 0, + "gate_e_violations_allowed": 0, + "hand_corrections_allowed": 0, + "sample_reuse_allowed": False, + "perfect_accuracy_required": False, + "question": "reduce_secondary_fallthrough_without_false_high_or_false_reject", +} + + +class ScreenV4Error(ValueError): + pass + + +class EmptyLexicon: + def noun_lex(self, lemma: str) -> str | None: + return None + + def has_adjective(self, lemma: str) -> bool: + return False + + +class WordNetLexicon: + """First-sense lex-file names. Used as gloss evidence, not as a phrase list.""" + + def __init__(self, root: str | Path): + base = Path(root) + self._lexnames = _lexnames(base / "lexnames") + noun_offset = _offset_lexnum(base / "data.noun") + self._noun = { + lemma: self._lexnames.get(noun_offset[offset]) + for lemma, offset in _first_offsets(base / "index.noun").items() + if offset in noun_offset + } + self._adj = set(_first_offsets(base / "index.adj")) + + def noun_lex(self, lemma: str) -> str | None: + return self._noun.get(lemma.casefold()) + + def has_adjective(self, lemma: str) -> bool: + return lemma.casefold() in self._adj + + +def normalize_lexical(surface: str) -> str: + """Fold hyphenation and punctuation. Apostrophes stay lexical.""" + text = unicodedata.normalize("NFKC", surface).casefold() + text = text.replace("\u2019", "'").replace("`", "'") + text = _HYPHEN.sub(" ", text) + text = _PUNCT.sub(" ", text) + return re.sub(r"\s+", " ", text).strip() + + +def canonical_bucket(raw: str) -> str: + bucket = _CANONICAL.get(str(raw or "")) + if bucket is None: + raise ScreenV4Error(f"bucket {raw!r} is not a screen bucket") + return bucket + + +def apply_v4(v3_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v4 bucket. HIGH and REJECT are not inspected.""" + bucket = canonical_bucket(v3_bucket) + normalized = normalize_lexical(surface) + if bucket != "SECONDARY": + return _decision(bucket, bucket, None, [], normalized, inspected=False) + tokens = normalized.split() if normalized else [] + gloss_text = gloss or "" + gstem = set(stems(gloss_text)) + patch_a = _patch_a(tokens, source_pos, gloss_text, lexicon) + if patch_a: + return _decision("SECONDARY", "REJECT", patch_a[0], patch_a[1:], normalized, inspected=True) + patch_b = _patch_b(tokens, source_pos, gloss_text, gstem) + if patch_b: + return _decision("SECONDARY", "HIGH", patch_b[0], patch_b[1:], normalized, inspected=True) + return _decision("SECONDARY", "SECONDARY", None, [], normalized, inspected=True) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 113) -> dict: + """Mechanical GATE_A through GATE_E report. This is not a precision score.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = {"secondary_to_high": 0, "secondary_to_reject": 0, "secondary_unchanged": 0} + held = {"high": 0, "reject": 0} + conflicts = 0 + for row in rows: + v3 = canonical_bucket(str(row["v3_bucket"])) + v4 = canonical_bucket(str(row["v4_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if v3 == "HIGH": + held["high"] += 1 + if v4 != "HIGH": + failures.append("HIGH row changed bucket") + if operator == "HIGH" and v4 != "HIGH": + failures.append("v3-correct HIGH is no longer operator-correct") + elif v3 == "REJECT": + held["reject"] += 1 + if v4 != "REJECT": + failures.append("REJECT row changed bucket") + if operator == "REJECT" and v4 != "REJECT": + failures.append("v3-correct REJECT is no longer operator-correct") + elif v3 == "SECONDARY": + if v4 == "SECONDARY": + moves["secondary_unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged SECONDARY carries transition evidence") + elif v4 == "HIGH": + moves["secondary_to_high"] += 1 + _require_transition(primary, supporting, PATCH_B, failures) + elif v4 == "REJECT": + moves["secondary_to_reject"] += 1 + _require_transition(primary, supporting, PATCH_A, failures) + else: + failures.append("SECONDARY moved outside HIGH and REJECT") + else: + failures.append("v3 bucket is outside HIGH, REJECT, and SECONDARY") + if v3 != v4 and operator != v4: + conflicts += 1 + if v3 != v4 and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + verified = not failures + return { + "schema": "hyperlex.unbind_screen_v4_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "high_unchanged": held["high"], + "reject_unchanged": held["reject"], + "moves": moves, + "operator_conflict_on_move": conflicts, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "gate_b_violations": sum(1 for item in failures if "operator-correct" in item or "changed bucket" in item), + "gate_e_violations": sum( + 1 for item in failures + if "changed bucket" in item or "outside HIGH and REJECT" in item or "no primary" in item + or "unchanged SECONDARY" in item or "moved outside" in item + ), + "assertions": { + "A_historical_replay": len(rows) == expected_rows and len(set(surfaces)) == len(surfaces), + "B_outer_bucket_preservation": not any("operator-correct" in item or "changed bucket" in item for item in failures), + "C_patch_a_targeting": not any(item.startswith("SECONDARY to REJECT") for item in failures), + "D_patch_b_targeting": not any(item.startswith("SECONDARY to HIGH") for item in failures), + "E_no_other_movement": not any( + "changed bucket" in item or "outside HIGH" in item or "moved outside" in item + for item in failures + ), + "no_phrase_specific_rule": not phrase_specific_rule_fired, + }, + "measurement_eligible": verified, + "select_authorized": False, + "revision_eligible": False, + } + + +def measurement_allowed(report: dict) -> bool: + return bool( + report.get("regression") == "REGRESSION_VERIFIED" + and report.get("phrase_specific_rule_fired") is False + and not report.get("failures") + and report.get("measurement_eligible") is True + ) + + +def rule_surface_violations(source: str, forbidden: list[str] | tuple[str, ...]) -> list[str]: + """Return phrase-rule violations in scorer source. The probe list is not a rule.""" + found = [phrase for phrase in forbidden if phrase and phrase in source] + tree = ast.parse(source) + for node in ast.walk(tree): + if isinstance(node, ast.Compare): + if not any(isinstance(op, (ast.Eq, ast.NotEq)) for op in node.ops): + continue + candidates = [node.left, *node.comparators] + elif isinstance(node, (ast.List, ast.Tuple, ast.Set)): + candidates = list(node.elts) + elif isinstance(node, ast.Dict): + candidates = [item for item in (*node.keys, *node.values) if item is not None] + else: + continue + for item in candidates: + if ( + isinstance(item, ast.Constant) + and isinstance(item.value, str) + and len(item.value.split()) >= 2 + ): + found.append(item.value) + found.extend(re.findall(r"\b[0-9a-f]{64}\b", source)) + return found + + +def _decision(v3, v4, primary, supporting, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v3_bucket": v3, + "v4_bucket": v4, + "primary_evidence": primary, + "supporting_evidence": list(supporting), + "normalized": normalized, + "inspected": inspected, + } + + +def _require_transition(primary, supporting, allowed: tuple[str, ...], failures: list[str]) -> None: + label = "SECONDARY to HIGH" if allowed is PATCH_B else "SECONDARY to REJECT" + if primary not in allowed: + failures.append(f"{label} lacks patch evidence") + return + if supporting.count(primary) or primary in supporting: + failures.append(f"{label} repeats its primary evidence") + if len([primary]) != 1: + failures.append(f"{label} lacks one primary evidence") + extra = [item for item in supporting if item not in allowed] + if extra: + failures.append(f"{label} supporting evidence is outside the patch") + + +def _patch_a(tokens: list[str], pos: str, gloss: str, lexicon) -> list[str]: + nouns = _span_nouns(gloss, lexicon) + matched = [] + if _person(tokens, pos, nouns): + matched.append("multi_token_person_name") + if _organization(tokens, pos, nouns): + matched.append("organization_from_gloss") + if _species(tokens, pos, nouns): + matched.append("species_or_common_name_referent") + if _medical(tokens, pos, gloss, lexicon): + matched.append("medical_technical_expression") + if _productive_number(tokens): + matched.append("productive_number") + return matched + + +def _patch_b(tokens: list[str], pos: str, gloss: str, gstem: set[str]) -> list[str]: + matched = [] + if _nonliteral(tokens, pos, gstem): + matched.append("nonliteral_semantic_shift") + if _conventionalized(tokens, pos, gstem): + matched.append("conventionalized_idiom") + if _phrasal(tokens, pos, gloss, gstem): + matched.append("noncompositional_phrasal_binding") + if _fixed(tokens, pos, gstem): + matched.append("fixed_lexicalized_expression") + return matched + + +def _person(tokens: list[str], pos: str, nouns: list[tuple[str, str]]) -> bool: + if pos != "noun" or not _name_shape(tokens) or not nouns: + return False + head, lex = nouns[0] + return lex == "noun.person" and head not in tokens + + +def _organization(tokens: list[str], pos: str, nouns: list[tuple[str, str]]) -> bool: + if pos != "noun" or not nouns: + return False + head, lex = nouns[0] + return lex == "noun.group" and head not in tokens + + +def _species(tokens: list[str], pos: str, nouns: list[tuple[str, str]]) -> bool: + if pos != "noun": + return False + return any(lex in _LIFE and lemma not in tokens for lemma, lex in nouns) + + +def _medical(tokens: list[str], pos: str, gloss: str, lexicon) -> bool: + if pos != "noun": + return False + words = {tok.lower() for tok in _WORD.findall(gloss or "")} + if words.isdisjoint(_CLINICAL): + return False + for lemma, lex in _all_nouns(gloss, lexicon): + if lex == "noun.body" and lemma not in tokens: + return True + return False + + +def _productive_number(tokens: list[str]) -> bool: + if len(tokens) < 2: + return False + body = list(tokens) + if body[0] in {"a", "an"}: + body = body[1:] + if body and body[-1] in _MULTIPLIERS: + body = body[:-1] + if not body: + return False + return all(tok in NUMBERS or tok == "and" for tok in body) and any(tok in NUMBERS for tok in body) + + +def _name_shape(tokens: list[str]) -> bool: + if len(tokens) < 2: + return False + return all( + re.fullmatch(r"[a-z]+", tok) and tok not in STOP and tok not in PARTICLES + for tok in tokens + ) + + +def _span_nouns(gloss: str, lexicon) -> list[tuple[str, str]]: + found = [] + for low in _gloss_words(gloss): + if low in _DELIMITERS and found: + break + item = _noun(low, lexicon) + if item is not None: + found.append(item) + return found + + +def _all_nouns(gloss: str, lexicon) -> list[tuple[str, str]]: + return [item for low in _gloss_words(gloss) if (item := _noun(low, lexicon)) is not None] + + +def _gloss_words(gloss: str) -> list[str]: + return [tok.lower() for tok in _WORD.findall(gloss or "") if tok.lower() not in STOP] + + +def _noun(low: str, lexicon) -> tuple[str, str] | None: + if lexicon.has_adjective(low): + return None + lex = lexicon.noun_lex(low) + if not lex: + return None + return low, lex + + +def _content(tokens: list[str]) -> list[str]: + return [tok for tok in tokens if len(tok) >= 3 and tok not in STOP] + + +def _absent(token: str, gstem: set[str]) -> bool: + return len(token) >= 3 and token[:4] not in gstem + + +def _nonliteral(tokens: list[str], pos: str, gstem: set[str]) -> bool: + return any(( + _verb_the(tokens, pos, gstem), + _verb_determiner(tokens, pos, gstem), + _verb_pivot_noun(tokens, pos, gstem), + _in_the_head(tokens, pos, gstem), + )) + + +def _conventionalized(tokens: list[str], pos: str, gstem: set[str]) -> bool: + return any(( + _like_vehicle(tokens, pos, gstem), + _as_frame(tokens, pos, gstem), + _for_all(tokens, pos, gstem), + _light_verb(tokens, pos, gstem), + )) + + +def _verb_the(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or len(tokens) < 3 or tokens[1] != "the": + return False + content = _content(tokens) + return bool(content) and all(_absent(tok, gstem) for tok in content) + + +def _verb_determiner(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or len(tokens) != 3 or tokens[1] not in {"a", "an"}: + return False + content = _content(tokens) + return bool(content) and all(_absent(tok, gstem) for tok in content) + + +def _verb_pivot_noun(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or len(tokens) != 4: + return False + if tokens[1] not in {"on", "in"} or tokens[2] not in {"a", "an"}: + return False + return _absent(tokens[3], gstem) + + +def _in_the_head(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "adv", "noun"} or len(tokens) != 4: + return False + if tokens[0] != "in" or tokens[1] != "the": + return False + content = _content(tokens) + present = [tok for tok in content if tok[:4] in gstem] + return bool(present) and _absent(tokens[-1], gstem) + + +def _like_vehicle(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "adv"} or len(tokens) != 3: + return False + if tokens[0] != "like" or tokens[1] not in {"a", "an"}: + return False + return _absent(tokens[2], gstem) + + +def _as_frame(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "adv"} or len(tokens) != 3 or tokens[1] != "as": + return False + return _absent(tokens[2], gstem) + + +def _for_all(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "adv" or len(tokens) < 4 or tokens[:2] != ["for", "all"]: + return False + content = _content(tokens) + present = [tok for tok in content if tok[:4] in gstem] + return bool(present) and _absent(tokens[-1], gstem) + + +def _light_verb(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or not tokens or tokens[0] not in _LIGHT_VERBS: + return False + if not any(tok in _DETERMINERS for tok in tokens) or not gstem: + return False + surface = {tok[:4] for tok in _content(tokens)} + if not gstem <= surface: + return False + return _absent(tokens[0], gstem) or len(tokens[0]) < 3 + + +def _phrasal(tokens: list[str], pos: str, gloss: str, gstem: set[str]) -> bool: + if pos not in {"verb", "adj"} or not tokens or tokens[-1] not in PARTICLES: + return False + if tokens[0] in _COMPARATIVES: + return False + gloss_words = {tok.lower() for tok in _WORD.findall(gloss or "")} + if tokens[-1] in gloss_words: + return False + return _absent(tokens[0], gstem) or len(tokens[0]) < 3 + + +def _fixed(tokens: list[str], pos: str, gstem: set[str]) -> bool: + return _intensifier(tokens, pos, gstem) or _for_result(tokens, pos, gstem) + + +def _intensifier(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "adj" or len(tokens) != 2 or tokens[0] not in _INTENSIFIERS: + return False + return _absent(tokens[1], gstem) + + +def _for_result(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "verb"} or not (2 <= len(tokens) <= 4) or "for" not in tokens: + return False + content = _content(tokens) + return bool(content) and all(_absent(tok, gstem) for tok in content) + + +def _lexnames(path: Path) -> dict[int, str]: + names = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line.strip(): + continue + number, name, *_rest = line.split() + names[int(number)] = name + return names + + +def _offset_lexnum(path: Path) -> dict[str, int]: + found = {} + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + offset, lexnum, *_rest = line.split(" ", 2) + found[offset] = int(lexnum) + return found + + +def _first_offsets(path: Path) -> dict[str, str]: + found = {} + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + parts = line.split() + synset_cnt = int(parts[2]) + pointer_cnt = int(parts[3]) + rest = parts[4 + pointer_cnt:] + offsets = rest[2:2 + synset_cnt] + if not offsets: + continue + found[parts[0].casefold()] = offsets[0] + return found diff --git a/scripts/shadow/hyperlexical/unbind_screen_v4_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v4_measure.py new file mode 100644 index 00000000..e080b47a --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v4_measure.py @@ -0,0 +1,415 @@ +"""Replay and measurement driver for the v4 unbind screen. + +This module does not admit, settle, or append the ledger. +The forbidden-surface tuple is a regression probe. The scorer does not import it. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import ( + RULE_VERSION, + SUCCESS_CRITERIA, + WordNetLexicon, + apply_v4, + assess, + canonical_bucket, + measurement_allowed, + normalize_lexical, + rule_surface_violations, +) + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +TRAIN = Path("/home/morpheus/hlx-private/d1-spark-tree-20260924T213846Z/morph78-train-export.jsonl") +FIT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-2026-09-27-004/development_fit.json" +EVAL = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V3-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V4-001/unbind_screen_v4" +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_SAMPLE = "8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e" +EXPECTED_LABELS = "4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5" +EXPECTED_ACCEPTANCE = "ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b" +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", "v3", + "primary_evidence", "supporting_evidence", "evidence", "v3_bucket", "v4_bucket", +}) +SCORER_FILES = ( + Path(__file__).with_name("unbind_screen_v3.py"), + Path(__file__).with_name("unbind_screen_v4.py"), +) + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def scorer_violations() -> list[str]: + found = [] + for path in SCORER_FILES: + found.extend(rule_surface_violations(path.read_text(encoding="utf-8"), PROBE_SURFACES)) + return found + + +def load_replay_rows() -> list[dict]: + fit = json.loads(FIT.read_text(encoding="utf-8")) + rows = [] + for raw in fit["rows"]: + rows.append({ + "surface": raw["text"], + "pos": raw["source_pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["predicted"], + "operator_bucket": raw["operator"], + "split": raw.get("split") or "development", + }) + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (EVAL / "operator/heldout-001.labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (EVAL / "predictions/heldout-001.predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + reviews = { + json.loads(line)["row_id"]: json.loads(line) + for line in (EVAL / "operator/heldout-001.review.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(labels) != set(predictions) or set(labels) != set(reviews): + raise SystemExit("held-out label, prediction, and review identities differ") + for row_id, review in reviews.items(): + rows.append({ + "surface": review["surface"], + "pos": review["pos"], + "gloss": review.get("gloss") or "", + "v3_bucket": predictions[row_id]["bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "held_out_29", + }) + if len(rows) != 113: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 113") + if len({row["surface"] for row in rows}) != 113: + raise SystemExit("reviewed surfaces are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed before replay") + sample_sha = file_sha256( + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-2026-09-27-004/held_out_sample.jsonl" + ) + label_sha = file_sha256(EVAL / "operator/heldout-001.labels.jsonl") + if sample_sha != EXPECTED_SAMPLE or label_sha != EXPECTED_LABELS: + raise SystemExit("v3 sample or label hash changed") + lexicon = WordNetLexicon(WORDNET) + predictions = [] + assessed_rows = [] + for raw in load_replay_rows(): + decision = apply_v4(raw["v3_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + row_id = normalized_text_sha256(raw["surface"]) + record = { + "schema": "hyperlex.unbind_screen_v4_replay_row.v1", + "row_id": row_id, + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + **decision, + } + predictions.append(record) + assessed_rows.append(record) + violations = scorer_violations() + report = assess(assessed_rows, phrase_specific_rule_fired=bool(violations), expected_rows=113) + report["phrase_violations"] = violations + report["acceptance_sha256"] = EXPECTED_ACCEPTANCE + report["events_sha256"] = events_sha256() + report["v3_sample_sha256"] = sample_sha + report["v3_label_sha256"] = label_sha + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v3_bucket"] != row["v4_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + prediction_sha = _dump_jsonl(out_dir / "replay_113_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "replay_113_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "replay_113_gate_report.json", report) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "implementation_receipt.json", receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + receipt_path = out_dir / "implementation_receipt.json" + report = json.loads((out_dir / "replay_113_gate_report.json").read_text(encoding="utf-8")) + if not measurement_allowed(report): + raise SystemExit("measurement draw refused: regression is not verified") + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed before the draw") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", dict(SUCCESS_CRITERIA)) + reviewed = load_replay_rows() + blocked_hash = {normalized_text_sha256(row["surface"]) for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + lexicon = WordNetLexicon(WORDNET) + blind = [] + predictions = [] + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + decision = apply_v4(v3_bucket, surface, gloss, pos, lexicon) + row_id = normalized_text_sha256(surface) + provenance = { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 113, + } + blind.append({ + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V4-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": provenance, + }) + leaked = LEAK_KEYS.intersection(blind[-1]) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + predictions.append({ + "schema": "hyperlex.unbind_screen_v4_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V4-001", + "sample_id": "measurement-001", + "row_id": row_id, + "v3_rule": v3_rule, + "application_index": 1, + **decision, + }) + if {row["surface"] for row in blind} & {row["surface"] for row in reviewed}: + raise SystemExit("measurement sample reuses a reviewed surface") + if {normalize_lexical(row["surface"]) for row in blind} & blocked_norm: + raise SystemExit("measurement sample reuses a normalized reviewed identity") + blind.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blind) + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v4_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(blind), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 113, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "v4_application_count": 1, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads(receipt_path.read_text(encoding="utf-8")) + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_state"] = "SAMPLE_FROZEN" + receipt["measurement_rows"] = len(blind) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v4_application_count"] = 1 + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["events_sha256"] = events_sha256() + _dump_json(receipt_path, receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + return freeze + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + source_sha = { + path.name: file_sha256(path) + for path in SCORER_FILES + } + return { + "schema": "hyperlex.unbind_screen_v4_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "ENCODED", + "regression": report["regression"], + "measurement_eligible": report["measurement_eligible"], + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v3": "proposed_coverage_extension_only", + "acceptance_sha256": EXPECTED_ACCEPTANCE, + "source_sha256": source_sha, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "moves": report["moves"], + "operator_conflict_on_move": report["operator_conflict_on_move"], + "assertions": report["assertions"], + "v3_sample_sha256": report["v3_sample_sha256"], + "v3_label_sha256": report["v3_label_sha256"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + } + + +def _stratified_sample( + blocked_hash: set[str], blocked_norm: set[str], *, per_cell: int +) -> tuple[list[dict], list[dict]]: + from hyperlexical.clean_unbind import ( + WORDNET_LICENSE, + gate_rows, + load_jsonl, + read_wordnet_index, + rows_from_wordnet_atoms, + ) + from hyperlexical.identity_ledger import IdentityLedger + + atoms = read_wordnet_index(WORDNET) + rows = rows_from_wordnet_atoms(atoms, license=WORDNET_LICENSE) + train_rows = load_jsonl(TRAIN) + ledger = IdentityLedger.load(LEDGER) + admissible, _rejections, _account = gate_rows( + rows, + train_rows=train_rows, + ledger=ledger, + require_settlement=False, + ) + best: dict[str, dict] = {} + for row in admissible: + if row.get("role_scheme") != "positional": + continue + surface = str(row.get("text") or "") + identity = normalize_lexical(surface) + digest = normalized_text_sha256(surface) + if digest in blocked_hash or identity in blocked_norm: + continue + current = best.get(identity) + if current is None or digest < normalized_text_sha256(str(current.get("text") or "")): + best[identity] = row + cells: dict[tuple[str, int], list[dict]] = {} + for row in best.values(): + fillers = [str(tok) for tok in row.get("fillers") or []] + cells.setdefault((str(row.get("source_pos") or ""), len(fillers)), []).append(row) + picked = [] + strata = [] + for key in sorted(cells): + group = cells[key] + group.sort(key=lambda row: normalized_text_sha256(str(row.get("text") or ""))) + taken = group[:per_cell] + picked.extend(taken) + strata.append({ + "source_pos": key[0], + "token_count": key[1], + "available": len(group), + "taken": len(taken), + }) + if not picked: + raise SystemExit("measurement sample is empty") + return picked, strata + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Replay v4 and, if verified, draw the measurement sample") + parser.add_argument("command", choices=("replay", "draw")) + args = parser.parse_args() + if args.command == "replay": + receipt = replay() + else: + receipt = draw_measurement() + summary = { + key: receipt[key] + for key in receipt + if key in { + "state", "regression", "measurement_eligible", "measurement_state", + "replay_rows", "diff_rows", "moves", "operator_conflict_on_move", + "failures", "phrase_specific_rule_fired", "rows", "sample_sha256", + "precision", "v4_application_count", "events_sha256", "measurement_rows", + "measurement_sample_sha256", + } + } + print(json.dumps(summary, sort_keys=True, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v5_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v5_measure.py new file mode 100644 index 00000000..625b6a52 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v5_measure.py @@ -0,0 +1,425 @@ +"""Replay and measurement driver for the v5 unbind screen. + +This module does not admit, settle, or append the ledger. +The forbidden-surface tuple is a regression probe. The scorer does not import it. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import WordNetLexicon as V4Lexicon +from hyperlexical.unbind_screen_v4 import apply_v4, canonical_bucket, normalize_lexical, rule_surface_violations +from hyperlexical.unbind_screen_v4_measure import _stratified_sample +from hyperlexical.unbind_screen_v5 import ( + RULE_VERSION, + WordNetLexicon, + apply_v5, + assess, + measurement_allowed, +) + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +EVAL3 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V3-001" +V4 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V4-001/unbind_screen_v4" +HYP = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-HYPOTHESIS-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-001/unbind_screen_v5" +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_V4_SAMPLE = "dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b" +EXPECTED_V4_PREDICTIONS = "a854847e516fbcd37fbb221456e8caf8c420552795ac7fc6225960bb5434084f" +EXPECTED_V4_LABELS = "023691f8349f0dda12c234691f235ae109289fcf9eab86ec20be1e23bfed9463" +EXPECTED_V4_ACCEPTANCE = "ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b" +EXPECTED_V3_SAMPLE = "8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e" +EXPECTED_V3_LABELS = "4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5" +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", + "on the go", + "flat out", + "in the way", + "to a t", + "slip of the tongue", + "run low", + ".22 caliber", + "phi correlation", + "blue-eyed african daisy", + "monoamine oxidase inhibitor", + "air force research laboratory", + "martin luther king jr's birthday", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", "v3", "v5", + "primary_evidence", "supporting_evidence", "evidence", "v3_bucket", "v4_bucket", "v5_bucket", +}) +SCORER = Path(__file__).with_name("unbind_screen_v5.py") + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def scorer_violations() -> list[str]: + return rule_surface_violations(SCORER.read_text(encoding="utf-8"), PROBE_SURFACES) + + +def load_replay_rows() -> list[dict]: + rows = [] + for line in (V4 / "replay_113_predictions.jsonl").read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + raw = json.loads(line) + rows.append({ + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["v3_bucket"], + "v4_bucket": raw["v4_bucket"], + "operator_bucket": raw["operator_bucket"], + "split": raw.get("split") or "reviewed_113", + "row_id": raw["row_id"], + }) + sample = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V4 / "measurement_sample.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V4 / "measurement_labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V4 / "measurement_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(sample) != set(labels) or set(sample) != set(predictions): + raise SystemExit("v4 measurement identities differ") + for row_id, blind in sample.items(): + rows.append({ + "surface": blind["surface"], + "pos": blind["pos"], + "gloss": blind.get("gloss") or "", + "v3_bucket": predictions[row_id]["v3_bucket"], + "v4_bucket": predictions[row_id]["v4_bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "v4_measurement_28", + "row_id": row_id, + }) + if len(rows) != 141 or len({row["surface"] for row in rows}) != 141: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 141 unique") + if len({row["row_id"] for row in rows}) != 141: + raise SystemExit("reviewed row ids are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("replay refused: the v5 measurement sample is already frozen") + _require_frozen_parents() + lexicon = WordNetLexicon(WORDNET) + predictions = [] + for raw in load_replay_rows(): + decision = apply_v5(raw["v4_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + record = { + "schema": "hyperlex.unbind_screen_v5_replay_row.v1", + "row_id": raw["row_id"], + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + "v3_bucket": canonical_bucket(raw["v3_bucket"]), + **decision, + } + predictions.append(record) + violations = scorer_violations() + report = assess(predictions, phrase_specific_rule_fired=bool(violations), expected_rows=141) + report["phrase_violations"] = violations + report["acceptance_sha256"] = file_sha256(HYP / "ACCEPTANCE.json") + report["events_sha256"] = events_sha256() + report["v4_sample_sha256"] = EXPECTED_V4_SAMPLE + report["v4_prediction_sha256"] = EXPECTED_V4_PREDICTIONS + report["v4_label_sha256"] = EXPECTED_V4_LABELS + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v4_bucket"] != row["v5_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + out_dir.chmod(0o700) + prediction_sha = _dump_jsonl(out_dir / "v5_replay_141_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "v5_replay_141_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "v5_replay_141_gate_report.json", report) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "v5_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is already frozen") + report = json.loads((out_dir / "v5_replay_141_gate_report.json").read_text(encoding="utf-8")) + if not measurement_allowed(report): + raise SystemExit("measurement draw refused: regression is not verified") + _require_frozen_parents() + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + criteria = dict(acceptance["success_criteria"]) + if criteria.get("false_secondary_rate_must_be_strictly_below") != "12/28": + raise SystemExit("acceptance bar is not the frozen false-secondary rate") + if criteria.get("accuracy_gain_required") is not False or criteria.get("high_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + if criteria.get("reject_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", criteria) + reviewed = load_replay_rows() + blocked_hash = {row["row_id"] for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + lexicon = WordNetLexicon(WORDNET) + v4_lexicon = V4Lexicon(WORDNET) + blind = [] + predictions = [] + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + v4_decision = apply_v4(v3_bucket, surface, gloss, pos, v4_lexicon) + decision = apply_v5(v4_decision["v4_bucket"], surface, gloss, pos, lexicon) + row_id = normalized_text_sha256(surface) + blind.append({ + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V5-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 141, + }, + }) + leaked = LEAK_KEYS.intersection(blind[-1]) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + predictions.append({ + "schema": "hyperlex.unbind_screen_v5_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V5-001", + "sample_id": "measurement-001", + "row_id": row_id, + "v3_rule": v3_rule, + "application_index": 1, + "v3_bucket": canonical_bucket(v3_bucket), + **decision, + }) + if {row["surface"] for row in blind} & {row["surface"] for row in reviewed}: + raise SystemExit("measurement sample reuses a reviewed surface") + if {normalize_lexical(row["surface"]) for row in blind} & blocked_norm: + raise SystemExit("measurement sample reuses a normalized reviewed identity") + if {row["row_id"] for row in blind} & blocked_hash: + raise SystemExit("measurement sample reuses a reviewed row id") + blind.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blind) + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v5_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(blind), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 141, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "v5_application_count": 1, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads((out_dir / "v5_implementation_receipt.json").read_text(encoding="utf-8")) + receipt["state"] = "MEASUREMENT_ELIGIBLE" + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_state"] = "SAMPLE_FROZEN" + receipt["measurement_rows"] = len(blind) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v5_application_count"] = 1 + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["events_sha256"] = events_sha256() + _dump_json(out_dir / "v5_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash drifted") + return freeze + + +def _require_frozen_parents() -> None: + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed") + checks = { + V4 / "measurement_sample.jsonl": EXPECTED_V4_SAMPLE, + V4 / "measurement_predictions.jsonl": EXPECTED_V4_PREDICTIONS, + V4 / "measurement_labels.jsonl": EXPECTED_V4_LABELS, + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V4-HYPOTHESIS-001/ACCEPTANCE.json": EXPECTED_V4_ACCEPTANCE, + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-2026-09-27-004/held_out_sample.jsonl": EXPECTED_V3_SAMPLE, + EVAL3 / "operator/heldout-001.labels.jsonl": EXPECTED_V3_LABELS, + } + for path, digest in checks.items(): + if file_sha256(path) != digest: + raise SystemExit(f"frozen parent changed: {path.name}") + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + return { + "schema": "hyperlex.unbind_screen_v5_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "REGRESSION_VERIFIED" if report["measurement_eligible"] else "ENCODED", + "regression": report["regression"], + "measurement_eligible": report["measurement_eligible"], + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v4": "pure_wrapper_over_a_frozen_v4_bucket", + "acceptance_sha256": report["acceptance_sha256"], + "source_sha256": {SCORER.name: file_sha256(SCORER)}, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "moves": report["moves"], + "operator_conflict_on_move": report["operator_conflict_on_move"], + "high_unchanged": report["high_unchanged"], + "reject_unchanged": report["reject_unchanged"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "precision": "NOT_COMPUTABLE", + } + + +def _touch_hypothesis(receipt: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + draft = file_sha256(HYP / "HYPOTHESIS.draft.json") + acceptance = file_sha256(HYP / "ACCEPTANCE.json") + if hypothesis.get("draft_hypothesis_sha256") != draft or hypothesis.get("acceptance_sha256") != acceptance: + raise SystemExit("v5 hypothesis no longer points at the frozen draft and acceptance") + hypothesis["encoded"] = True + hypothesis["state"] = receipt["state"] + hypothesis["regression"] = receipt["regression"] + hypothesis["measurement_eligible"] = receipt["measurement_eligible"] + hypothesis["applied"] = bool(receipt.get("applied_to_measurement")) + hypothesis["applied_to_measurement"] = bool(receipt.get("applied_to_measurement")) + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["operator_conflict_on_move"] = receipt["operator_conflict_on_move"] + hypothesis["moves"] = receipt["moves"] + hypothesis["replay_rows"] = receipt["replay_rows"] + if receipt.get("measurement_sample_sha256"): + hypothesis["measurement_sample_sha256"] = receipt["measurement_sample_sha256"] + hypothesis["measurement_rows"] = receipt["measurement_rows"] + hypothesis["measurement_state"] = receipt["measurement_state"] + hypothesis["precision"] = "NOT_COMPUTABLE" + text = json.dumps(hypothesis, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Replay v5 and, if verified, draw the measurement sample") + parser.add_argument("command", choices=("replay", "draw")) + args = parser.parse_args() + receipt = replay() if args.command == "replay" else draw_measurement() + summary = { + key: receipt[key] + for key in ( + "state", "regression", "measurement_eligible", "measurement_state", + "replay_rows", "diff_rows", "moves", "operator_conflict_on_move", + "failures", "phrase_specific_rule_fired", "rows", "sample_sha256", + "precision", "v5_application_count", "events_sha256", "measurement_rows", + "measurement_sample_sha256", "high_unchanged", "reject_unchanged", + ) + if key in receipt + } + print(json.dumps(summary, sort_keys=True, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v6_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v6_measure.py new file mode 100644 index 00000000..dee8d2f6 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v6_measure.py @@ -0,0 +1,503 @@ +"""Replay and measurement driver for the v6 unbind screen. + +This module does not admit, settle, or append the ledger. +The forbidden-surface tuple is a regression probe. The scorer does not import it. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import WordNetLexicon as V4Lexicon +from hyperlexical.unbind_screen_v4 import apply_v4, canonical_bucket, normalize_lexical, rule_surface_violations +from hyperlexical.unbind_screen_v4_measure import _stratified_sample +from hyperlexical.unbind_screen_v5 import WordNetLexicon as V5Lexicon +from hyperlexical.unbind_screen_v5 import apply_v5 +from hyperlexical.unbind_screen_v6 import RULE_VERSION, apply_v6, assess, measurement_allowed + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +V5 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-001/unbind_screen_v5" +V5H = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-HYPOTHESIS-001" +HYP = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-HYPOTHESIS-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-001/unbind_screen_v6" +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_V5_SAMPLE = "a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359" +EXPECTED_V5_PREDICTIONS = "51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f" +EXPECTED_V5_LABELS = "b05dc1c35d3d10eed4b615ca1ee8251a875a450f57560fef4b5a84b3e1fc075f" +EXPECTED_V5_ACCEPTANCE = "752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a" +EXPECTED_V6_ACCEPTANCE = "68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f" +EXPECTED_V6_DRAFT = "942eeab825d1c89c21e91af84da0d753281a91012e9ff219a633d27fe6648aa5" +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", + "on the go", + "flat out", + "in the way", + "to a t", + "slip of the tongue", + "run low", + ".22 caliber", + "phi correlation", + "blue-eyed african daisy", + "monoamine oxidase inhibitor", + "air force research laboratory", + "martin luther king jr's birthday", + "keep out", + "on the job", + "take orders", + "from nowhere", + "all told", + "rolled into one", + "handle with kid gloves", + "fleet ballistic missile submarine", + "rapid eye movement sleep", + "law of conservation of energy", + "coronoid process of the mandible", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", "v3", "v5", "v6", + "primary_evidence", "supporting_evidence", "evidence", "v3_bucket", "v4_bucket", + "v5_bucket", "v6_bucket", +}) +SCORER = Path(__file__).with_name("unbind_screen_v6.py") + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def scorer_violations() -> list[str]: + return rule_surface_violations(SCORER.read_text(encoding="utf-8"), PROBE_SURFACES) + + +def load_replay_rows() -> list[dict]: + rows = [] + for line in (V5 / "v5_replay_141_predictions.jsonl").read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + raw = json.loads(line) + rows.append({ + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["v3_bucket"], + "v4_bucket": raw["v4_bucket"], + "v5_bucket": raw["v5_bucket"], + "operator_bucket": raw["operator_bucket"], + "split": raw.get("split") or "reviewed_141", + "row_id": raw["row_id"], + }) + sample = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V5 / "measurement_sample.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V5 / "measurement_labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V5 / "measurement_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(sample) != set(labels) or set(sample) != set(predictions): + raise SystemExit("v5 measurement identities differ") + seen = {row["row_id"] for row in rows} + if seen & set(sample): + raise SystemExit("v5 measurement overlaps the prior reviewed surfaces") + for row_id, blind in sample.items(): + pred = predictions[row_id] + rows.append({ + "surface": blind["surface"], + "pos": blind["pos"], + "gloss": blind.get("gloss") or "", + "v3_bucket": pred["v3_bucket"], + "v4_bucket": pred["v4_bucket"], + "v5_bucket": pred["v5_bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "v5_measurement_28", + "row_id": row_id, + }) + if len(rows) != 169 or len({row["row_id"] for row in rows}) != 169: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 169 unique") + if len({row["surface"] for row in rows}) != 169: + raise SystemExit("reviewed surfaces are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("replay refused: the v6 measurement sample is already frozen") + _require_frozen_parents() + lexicon = V5Lexicon(WORDNET) + predictions = [] + for raw in load_replay_rows(): + decision = apply_v6(raw["v5_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + predictions.append({ + "schema": "hyperlex.unbind_screen_v6_replay_row.v1", + "row_id": raw["row_id"], + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + "v3_bucket": canonical_bucket(raw["v3_bucket"]), + "v4_bucket": canonical_bucket(raw["v4_bucket"]), + **decision, + }) + violations = scorer_violations() + report = assess(predictions, phrase_specific_rule_fired=bool(violations), expected_rows=169) + report["phrase_violations"] = violations + report["acceptance_sha256"] = file_sha256(HYP / "ACCEPTANCE.json") + report["events_sha256"] = events_sha256() + if report["replay_rows"] != 169: + report["failures"].append("replay count is not 169") + report["assertions"]["replay_count"] = False + report["regression"] = "REGRESSION_FAILED" + report["state"] = "ENCODED" + report["measurement_eligible"] = False + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v5_bucket"] != row["v6_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + out_dir.chmod(0o700) + prediction_sha = _dump_jsonl(out_dir / "v6_replay_169_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "v6_replay_169_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "v6_replay_169_gate_report.json", report) + summary = _error_summary(predictions, report) + _dump_json(out_dir / "v6_replay_169_error_summary.json", summary) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "v6_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is already frozen") + report = json.loads((out_dir / "v6_replay_169_gate_report.json").read_text(encoding="utf-8")) + if not measurement_allowed(report) or report.get("replay_rows") != 169: + raise SystemExit("measurement draw refused: regression is not verified") + _require_frozen_parents() + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + criteria = dict(acceptance["success_criteria"]) + if criteria.get("high_precision") != 1.0 or criteria.get("reject_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_high_allowed") != 0 or criteria.get("false_reject_allowed") != 0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_secondary_rate_required") is not False or criteria.get("accuracy_gain_required") is not False: + raise SystemExit("acceptance bar gained a coverage or accuracy target") + if criteria.get("next_measurement_excludes_reviewed_surfaces") != 169: + raise SystemExit("acceptance exclusion count changed") + if "false_secondary_rate_must_be_strictly_below" in criteria: + raise SystemExit("a false-secondary floor was added after the contract") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", criteria) + reviewed = load_replay_rows() + blocked_hash = {row["row_id"] for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + v5_lexicon = V5Lexicon(WORDNET) + v4_lexicon = V4Lexicon(WORDNET) + v6_lexicon = V5Lexicon(WORDNET) + blind = [] + predictions = [] + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + v4_decision = apply_v4(v3_bucket, surface, gloss, pos, v4_lexicon) + v5_decision = apply_v5(v4_decision["v4_bucket"], surface, gloss, pos, v5_lexicon) + decision = apply_v6(v5_decision["v5_bucket"], surface, gloss, pos, v6_lexicon) + row_id = normalized_text_sha256(surface) + blind.append({ + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V6-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 169, + }, + }) + leaked = LEAK_KEYS.intersection(blind[-1]) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + predictions.append({ + "schema": "hyperlex.unbind_screen_v6_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V6-001", + "sample_id": "measurement-001", + "row_id": row_id, + "v3_rule": v3_rule, + "application_index": 1, + "v3_bucket": canonical_bucket(v3_bucket), + "v4_bucket": v4_decision["v4_bucket"], + "v5_bucket": v5_decision["v5_bucket"], + **decision, + }) + if {row["surface"] for row in blind} & {row["surface"] for row in reviewed}: + raise SystemExit("measurement sample reuses a reviewed surface") + if {normalize_lexical(row["surface"]) for row in blind} & blocked_norm: + raise SystemExit("measurement sample reuses a normalized reviewed identity") + if {row["row_id"] for row in blind} & blocked_hash: + raise SystemExit("measurement sample reuses a reviewed row id") + blind.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blind) + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v6_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(blind), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 169, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "v6_application_count": 1, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "false_secondary_rate_required": False, + "accuracy_gain_required": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads((out_dir / "v6_implementation_receipt.json").read_text(encoding="utf-8")) + receipt["state"] = "MEASUREMENT_ELIGIBLE" + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_state"] = "SAMPLE_FROZEN" + receipt["measurement_rows"] = len(blind) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v6_application_count"] = 1 + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["events_sha256"] = events_sha256() + _dump_json(out_dir / "v6_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash drifted") + return freeze + + +def _error_summary(predictions: list[dict], report: dict) -> dict: + reversals = [] + for row in predictions: + if row["v5_bucket"] == row["v6_bucket"]: + continue + reversals.append({ + "row_id": row["row_id"], + "surface": row["surface"], + "v5_bucket": row["v5_bucket"], + "v6_bucket": row["v6_bucket"], + "operator_bucket": row["operator_bucket"], + "primary_evidence": row["primary_evidence"], + "previously_correct": row["v5_bucket"] == row["operator_bucket"], + }) + stayed_wrong = sum( + 1 for row in predictions + if row["v5_bucket"] == row["v6_bucket"] and row["v6_bucket"] != row["operator_bucket"] + ) + return { + "schema": "hyperlex.unbind_screen_v6_replay_error_summary.v1", + "role": "regression_summary", + "executable_rule_source": False, + "motivating_cases_are_not_a_required_fit": True, + "regression": report["regression"], + "replay_rows": report["replay_rows"], + "moves": report["moves"], + "previously_correct": report["previously_correct"], + "previously_correct_lost": report["previously_correct_lost"], + "direct_swaps": report["direct_swaps"], + "reversals": reversals, + "unchanged_disagreement_count": stayed_wrong, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + } + + +def _require_frozen_parents() -> None: + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed") + checks = { + V5 / "measurement_sample.jsonl": EXPECTED_V5_SAMPLE, + V5 / "measurement_predictions.jsonl": EXPECTED_V5_PREDICTIONS, + V5 / "measurement_labels.jsonl": EXPECTED_V5_LABELS, + V5H / "ACCEPTANCE.json": EXPECTED_V5_ACCEPTANCE, + HYP / "ACCEPTANCE.json": EXPECTED_V6_ACCEPTANCE, + HYP / "HYPOTHESIS.draft.json": EXPECTED_V6_DRAFT, + } + for path, digest in checks.items(): + if file_sha256(path) != digest: + raise SystemExit(f"frozen parent changed: {path.name}") + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + verified = report["regression"] == "REGRESSION_VERIFIED" + return { + "schema": "hyperlex.unbind_screen_v6_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "regression": report["regression"], + "measurement_eligible": verified, + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v5": "challengeable_outer_buckets_over_a_frozen_v5_bucket", + "demotion_stops_for_this_application": True, + "acceptance_sha256": report["acceptance_sha256"], + "source_sha256": {SCORER.name: file_sha256(SCORER)}, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "assertions": report["assertions"], + "moves": report["moves"], + "previously_correct": report["previously_correct"], + "previously_correct_lost": report["previously_correct_lost"], + "direct_swaps": report["direct_swaps"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "precision": "NOT_COMPUTABLE", + } + + +def _touch_hypothesis(receipt: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + draft = file_sha256(HYP / "HYPOTHESIS.draft.json") + acceptance = file_sha256(HYP / "ACCEPTANCE.json") + if draft != EXPECTED_V6_DRAFT or acceptance != EXPECTED_V6_ACCEPTANCE: + raise SystemExit("v6 draft or acceptance bytes changed") + if hypothesis.get("draft_hypothesis_sha256") not in (None, EXPECTED_V6_DRAFT): + raise SystemExit("v6 hypothesis no longer points at the frozen draft") + if hypothesis.get("acceptance_sha256") != EXPECTED_V6_ACCEPTANCE: + raise SystemExit("v6 hypothesis no longer points at the frozen acceptance") + hypothesis["encoded"] = True + hypothesis["state"] = receipt["state"] + hypothesis["regression"] = receipt["regression"] + hypothesis["measurement_eligible"] = receipt["measurement_eligible"] + hypothesis["applied"] = bool(receipt.get("applied_to_measurement")) + hypothesis["applied_to_measurement"] = bool(receipt.get("applied_to_measurement")) + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["previously_correct_lost"] = receipt["previously_correct_lost"] + hypothesis["direct_swaps"] = receipt["direct_swaps"] + hypothesis["moves"] = receipt["moves"] + hypothesis["replay_rows"] = receipt["replay_rows"] + if receipt.get("measurement_sample_sha256"): + hypothesis["measurement_sample_sha256"] = receipt["measurement_sample_sha256"] + hypothesis["measurement_rows"] = receipt["measurement_rows"] + hypothesis["measurement_state"] = receipt["measurement_state"] + hypothesis["precision"] = "NOT_COMPUTABLE" + path.write_text(json.dumps(hypothesis, indent=2) + "\n", encoding="utf-8") + path.chmod(0o600) + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Replay v6 and, if verified, draw the measurement sample") + parser.add_argument("command", choices=("replay", "draw", "run")) + args = parser.parse_args() + if args.command == "replay": + receipt = replay() + elif args.command == "draw": + receipt = draw_measurement() + else: + receipt = replay() + if receipt.get("regression") == "REGRESSION_VERIFIED": + receipt = draw_measurement() + print(json.dumps({ + key: receipt.get(key) + for key in ( + "state", "regression", "measurement_eligible", "measurement_state", + "replay_rows", "diff_rows", "moves", "previously_correct_lost", + "direct_swaps", "failures", "rows", "sample_sha256", "precision", + "v6_application_count", "events_sha256", + ) + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v2_replay.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v2_replay.py new file mode 100644 index 00000000..19501fc8 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v2_replay.py @@ -0,0 +1,362 @@ +"""Replay the frozen 225-row development manifest through procedure v2. + +The replay is a development experiment. It does not draw a measurement sample, +restore the co-lemma shortcut, or write a class back onto the manifest. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.screen_eval import _error_class, _metrics +from hyperlexical.unbind_sense_screen_v2 import classify, load_exceptions, load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +PREDICTIONS = OUT / "development_replay_v2_predictions.jsonl" +REPORT = OUT / "development_replay_v2_report.json" +TRACKER = OUT / "HYPOTHESIS.json" +EVIDENCE = OUT / "DEVELOPMENT_EVIDENCE.json" +PROCEDURE_V1 = OUT / "CLASSIFICATION_PROCEDURE.json" +PROCEDURE_V2 = OUT / "CLASSIFICATION_PROCEDURE.v2.json" +V1_PREDICTIONS = OUT / "development_replay_predictions.jsonl" +V1_REPORT = OUT / "development_replay_report.json" + +EXPECTED = { + PROCEDURE_V1: "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + PROCEDURE_V2: "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + OUT / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + OUT / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + OUT / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + OUT / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + OUT / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + V1_PREDICTIONS: "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + V1_REPORT: "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + TRACKER: "af11d20ebefec5629718617ebacc07d8b4cc36c61e59e829d153b05b9397ca40", + LEDGER / "events.jsonl": "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", +} +SENSE_CLASSES = ( + "REFERENTIAL", + "LEXICALIZED_NONCOMPOSITIONAL", + "LEXICALIZED_COMPOSITIONAL", + "ORDINARY_COMPOSITIONAL", + "AMBIGUOUS", +) +BUCKETS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE") +OPERATOR_BUCKETS = BUCKETS + ("UNRESOLVED",) +STATES = ("YES", "NO", "UNKNOWN", "CONFLICT") + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str | None: + if denominator == 0: + return None + return f"{numerator}/{denominator}" + + +def main() -> None: + for path, expected in EXPECTED.items(): + found = sha256(path) + if found != expected: + refuse(f"hash mismatch {path.name}: {found}") + if PREDICTIONS.exists() or REPORT.exists(): + refuse("procedure v2 replay artifacts already exist") + evidence = json.loads(EVIDENCE.read_text(encoding="utf-8")) + rows = evidence["rows"] + if len(rows) != 225 or any(row.get("sense_class") is not None for row in rows): + refuse("development manifest is not the unclassified 225-row inventory") + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if tracker.get("state") != "DEVELOPMENT_ANALYZED" or tracker.get("procedure_v2_state") != "PROCEDURE_V2_FROZEN": + refuse("tracker is not waiting at PROCEDURE_V2_FROZEN") + if tracker.get("procedure_v2_encoded") is not False or tracker.get("measurement_sample_drawn") is not False: + refuse("procedure v2 is already encoded or a sample exists") + procedure = json.loads(PROCEDURE_V2.read_text(encoding="utf-8")) + if procedure.get("encoded") is not False or procedure.get("state") != "PROCEDURE_V2_FROZEN": + refuse("procedure v2 artifact is not the sealed unencoded freeze") + + synsets, glosses = load_wordnet(WORDNET) + exceptions = load_exceptions(WORDNET) + targets = {key: synset.lemmas for key, synset in synsets.items()} + classified = [] + for row in rows: + key = (row["synset_pos"], row["synset_offset"]) + synset = synsets.get(key) + gloss = glosses.get(key) + if synset is None or gloss is None: + refuse(f"missing synset {row['synset_pos']} {row['synset_offset']}") + if gloss != row["gloss"]: + refuse(f"frozen gloss does not match the data-line first clause for {row['row_id']}") + decision = classify(row["surface"], row["gloss"], synset, exceptions, targets) + if decision["sense_class"] not in SENSE_CLASSES or decision["compositional_state"] == "NO": + refuse(f"classifier left the frozen catalog for {row['row_id']}") + if decision["bucket"] == "HIGH" or decision["sense_class"] == "LEXICALIZED_NONCOMPOSITIONAL": + refuse("encoder emitted HIGH from the empty compositional NO catalog") + classified.append((row, decision)) + + by_surface = {row["surface"]: decision for row, decision in classified} + alternation = by_surface.get("as far as possible") + if alternation is None or ( + alternation["referential_state"], + alternation["lexicalized_state"], + alternation["compositional_state"], + alternation["sense_class"], + ) != ("NO", "UNKNOWN", "YES", "AMBIGUOUS"): + refuse("as far as possible did not follow the frozen v2 illustration") + damascus = by_surface.get("road to damascus") + if damascus is None or ( + damascus["referential_state"], + damascus["lexicalized_state"], + damascus["compositional_state"], + damascus["sense_class"], + ) != ("UNKNOWN", "UNKNOWN", "UNKNOWN", "AMBIGUOUS"): + refuse("road to damascus did not follow the frozen v2 illustration") + + predictions = [] + for row, decision in sorted(classified, key=lambda item: item[0]["row_id"]): + predictions.append( + { + "schema": "hyperlex.unbind_sense_screen_v2_development_row.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "procedure": "hyperlex.unbind_sense_screen_v1_classification_procedure.v2", + "application_index": 1, + "row_id": row["row_id"], + "surface": row["surface"], + "pos": row["pos"], + "synset": f"{row['synset_pos']}:{row['synset_offset']}", + "synset_offset": row["synset_offset"], + "synset_pos": row["synset_pos"], + "referential_state": decision["referential_state"], + "lexicalized_state": decision["lexicalized_state"], + "compositional_state": decision["compositional_state"], + "sense_class": decision["sense_class"], + "bucket": decision["bucket"], + "primary_evidence_code": decision["primary_evidence_code"], + "supporting_evidence_codes": decision["supporting_evidence_codes"], + "evidence_sources": decision["evidence_sources"], + "confidence_status": decision["confidence_status"], + } + ) + prediction_sha = write_jsonl(PREDICTIONS, predictions) + + by_id = {row["row_id"]: decision for row, decision in classified} + ordered_rows = sorted(rows, key=lambda row: row["row_id"]) + pairs = [] + sense_by_operator = {op: {sense: 0 for sense in SENSE_CLASSES} for op in OPERATOR_BUCKETS} + error_rows = {name: [] for name in ("false_high", "false_reject", "false_secondary", "false_quarantine")} + for row in ordered_rows: + decision = by_id[row["row_id"]] + operator = row["operator_bucket"] + pairs.append((row["row_id"], decision["bucket"], operator, decision["primary_evidence_code"])) + sense_by_operator[operator][decision["sense_class"]] += 1 + kind = _error_class(decision["bucket"], operator) + if kind: + error_rows[kind].append( + { + "row_id": row["row_id"], + "surface": row["surface"], + "operator_bucket": operator, + "bucket": decision["bucket"], + "sense_class": decision["sense_class"], + "referential_state": decision["referential_state"], + "lexicalized_state": decision["lexicalized_state"], + "compositional_state": decision["compositional_state"], + "primary_evidence_code": decision["primary_evidence_code"], + "evidence_sources": decision["evidence_sources"], + } + ) + metrics = _metrics(pairs) + sense_counts = Counter(row["sense_class"] for row in predictions) + bucket_counts = Counter(row["bucket"] for row in predictions) + evidence_counts = Counter(row["primary_evidence_code"] for row in predictions) + state_counts = { + "referential": Counter(row["referential_state"] for row in predictions), + "lexicalized": Counter(row["lexicalized_state"] for row in predictions), + "compositional": Counter(row["compositional_state"] for row in predictions), + } + triples = Counter( + f"{row['referential_state']}|{row['lexicalized_state']}|{row['compositional_state']}" + for row in predictions + ) + + def bucket_block(name: str) -> dict: + stats = metrics["per_bucket"][name] + support = stats["support"] + predicted = bucket_counts[name] + true_positive = sum(1 for _i, pred, op, _r in pairs if pred == name and op == name) + return { + "precision": stats["precision"], + "precision_fraction": fraction(true_positive, predicted), + "recall": stats["recall"], + "recall_fraction": fraction(true_positive, support), + "operator_support": support, + "predicted": predicted, + } + + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + report = { + "schema": "hyperlex.unbind_sense_screen_v2_development_replay.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "procedure": "hyperlex.unbind_sense_screen_v1_classification_procedure.v2", + "role": "development experiment, not validation", + "question": "Can the frozen WordNet-only evidence model instantiate the intended classes, especially LEXICALIZED_NONCOMPOSITIONAL, without the co-lemma shortcut?", + "not_a_validation_set": True, + "state": "DEVELOPMENT_ANALYZED_V2", + "state_path": [ + "PROCEDURE_V2_FROZEN", + "ENCODE_PROCEDURE_V2_AUTHORIZATION", + "ENCODED", + "225_ROW_DEVELOPMENT_REPLAY", + "DEVELOPMENT_ANALYZED_V2", + ], + "analyzed_at": when, + "rows": 225, + "applications_per_row": 1, + "prediction_sha256": prediction_sha, + "procedure_v2_sha256": EXPECTED[PROCEDURE_V2], + "procedure_v2_mutated": False, + "procedure_v1_sha256": EXPECTED[PROCEDURE_V1], + "procedure_v1_mutated": False, + "acceptance_sha256": EXPECTED[OUT / "ACCEPTANCE.json"], + "development_evidence_sha256": EXPECTED[EVIDENCE], + "development_manifest_sense_classes_written": False, + "v1_prediction_sha256": EXPECTED[V1_PREDICTIONS], + "v1_report_sha256": EXPECTED[V1_REPORT], + "lexeme_architecture_sha256": EXPECTED[OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json"], + "lexeme_architecture_mutated": False, + "events_sha256": EXPECTED[LEDGER / "events.jsonl"], + "sense_class_counts": {name: sense_counts[name] for name in SENSE_CLASSES}, + "bucket_counts": {name: bucket_counts[name] for name in BUCKETS}, + "evidence_state_counts": { + dimension: {state: counts[state] for state in STATES} + for dimension, counts in state_counts.items() + }, + "evidence_state_triples": dict(sorted(triples.items())), + "operator_vs_bucket_confusion": metrics["confusion_matrix"], + "operator_vs_sense_class": sense_by_operator, + "ambiguous_count": sense_counts["AMBIGUOUS"], + "ambiguous_rate_fraction": fraction(sense_counts["AMBIGUOUS"], 225), + "ambiguous_rate_is_not_a_success_criterion": True, + "high_support": bucket_counts["HIGH"], + "high_support_is_zero": bucket_counts["HIGH"] == 0, + "lexicalized_noncompositional_support": sense_counts["LEXICALIZED_NONCOMPOSITIONAL"], + "lexicalized_noncompositional_instantiated": sense_counts["LEXICALIZED_NONCOMPOSITIONAL"] > 0, + "colemma_shortcut_restored": False, + "compositional_no_emitted": state_counts["compositional"]["NO"], + "evidence_code_counts": dict(sorted(evidence_counts.items())), + "HIGH": bucket_block("HIGH"), + "SECONDARY": bucket_block("SECONDARY"), + "REJECT": bucket_block("REJECT"), + "QUARANTINE": bucket_block("QUARANTINE"), + "referential_precision_fraction": bucket_block("REJECT")["precision_fraction"], + "referential_recall_fraction": bucket_block("REJECT")["recall_fraction"], + "lexicalized_compositional_support": sense_counts["LEXICALIZED_COMPOSITIONAL"], + "lexicalized_compositional_support_is_a_readiness_signal_only": True, + "false_high": len(error_rows["false_high"]), + "false_reject": len(error_rows["false_reject"]), + "false_secondary": len(error_rows["false_secondary"]), + "false_quarantine": len(error_rows["false_quarantine"]), + "false_high_rows": error_rows["false_high"], + "false_reject_rows": error_rows["false_reject"], + "diagnostics_are_not_pass_fail": True, + "measurement_bar_applied": False, + "measurement_sample_drawn": False, + "measurement_eligible": False, + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "ledger_appended": False, + "next_legal_transition": "V2_DEVELOPMENT_RESULT_REVIEW", + "next_transition_authorized": False, + } + report_sha = write_json(REPORT, report) + if any(row.get("sense_class") is not None for row in json.loads(EVIDENCE.read_text(encoding="utf-8"))["rows"]): + refuse("replay wrote a sense class onto the development manifest") + for path, expected in EXPECTED.items(): + if path == TRACKER: + continue + if sha256(path) != expected: + refuse(f"replay mutated {path.name}") + + tracker["previous_state"] = "DEVELOPMENT_ANALYZED" + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["v1_screen_state"] = "DEVELOPMENT_ANALYZED" + tracker["state"] = "DEVELOPMENT_ANALYZED_V2" + tracker["procedure_v2_state"] = "DEVELOPMENT_ANALYZED_V2" + tracker["procedure_v2_encoded"] = True + tracker["procedure_v2_encoding_authorized"] = True + tracker["procedure_v2_applied"] = True + tracker["procedure_v2_development_replay_run"] = True + tracker["procedure_v2_rows_classified"] = 225 + tracker["procedure_v2_prediction_sha256"] = prediction_sha + tracker["procedure_v2_report_sha256"] = report_sha + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["next_legal_transition"] = "V2_DEVELOPMENT_RESULT_REVIEW" + tracker["next_transition_authorized"] = False + tracker["high_support_v2"] = bucket_counts["HIGH"] + tracker["lexicalized_noncompositional_support_v2"] = sense_counts["LEXICALIZED_NONCOMPOSITIONAL"] + tracker_sha = write_json(TRACKER, tracker) + if sha256(PROCEDURE_V2) != EXPECTED[PROCEDURE_V2] or sha256(EVIDENCE) != EXPECTED[EVIDENCE]: + refuse("tracker update mutated a sealed artifact") + print(json.dumps({ + "prediction_sha256": prediction_sha, + "report_sha256": report_sha, + "tracker_sha256": tracker_sha, + "sense_class_counts": report["sense_class_counts"], + "bucket_counts": report["bucket_counts"], + "evidence_state_counts": report["evidence_state_counts"], + "ambiguous_rate_fraction": report["ambiguous_rate_fraction"], + "HIGH": report["HIGH"], + "SECONDARY": report["SECONDARY"], + "REJECT": report["REJECT"], + "false_high": report["false_high"], + "false_reject": report["false_reject"], + "false_secondary": report["false_secondary"], + "lexicalized_compositional_support": report["lexicalized_compositional_support"], + "lexicalized_noncompositional_instantiated": report["lexicalized_noncompositional_instantiated"], + "evidence_code_counts": report["evidence_code_counts"], + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main()