From 306fb5d479661c808a63d2998dd95017404ca464 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sun, 27 Sep 2026 14:48:15 -0700 Subject: [PATCH 1/5] Activate ai-native on the evaluation taxonomy The name stays on the nine-way production head. Settlement can now record it explicitly, evaluation stays off, and no row is settled. --- .../shadow/hyperlexical/eval_settlement.py | 846 +----------------- .../label-taxonomy-proposal.md | 362 +------- .../test_hyperlexical_eval_settlement.py | 476 +--------- 3 files changed, 3 insertions(+), 1681 deletions(-) diff --git a/scripts/shadow/hyperlexical/eval_settlement.py b/scripts/shadow/hyperlexical/eval_settlement.py index c0c3dcbe..311c8dd0 100644 --- a/scripts/shadow/hyperlexical/eval_settlement.py +++ b/scripts/shadow/hyperlexical/eval_settlement.py @@ -1,845 +1 @@ -"""Evaluation-settlement apply path for the eighteen-family taxonomy. - -``source_hint``, ``semantic_family``, and ``attest`` are separate fields. -A confirm settlement may store the same family string in both ``source_hint`` -and ``semantic_family``. This module never copies one field into another, -and it never derives ``attest`` from ``decision``. - -The production classify head is not read or written here. ``ACCEPT`` records -the operator-selected attest. It does not promote existing evidence to -``OBSERVED``. This command does not admit rows to ``EVAL_RESERVE``. -""" - -from __future__ import annotations - -import hashlib -import json -import os -from pathlib import Path -from typing import Any, Mapping, Sequence - -from .holdout_guard import normalized_text_sha256 - -SCHEMA = "hyperlex.eval_settlement.v1" -RECEIPT_SCHEMA = "hyperlex.eval_settlement_receipt.v1" -VENDOR_CALLS = 0 - -ACTIVE_FAMILIES: tuple[str, ...] = ( - "gaming-meta", - "betting-sharp", - "crypto-degen", - "internet-slang", - "memetic", - "social-status", - "relationship-dating", - "approval-disapproval", - "conflict-aggression", - "technology-ai", - "workplace-career", - "sports-competition", - "music-entertainment", - "fashion-aesthetic", - "regional-cultural", - "spiritual-mystic", - "identity-affiliation", - "politics-civic", -) -CANDIDATE_FAMILIES: tuple[str, ...] = ( - "finance-retail", - "market-structure", - "sexual-romantic", - "substance-party", - "crime-illicit", - "health-fitness", -) -ABSTAIN = "none" -DECISIONS = ("ACCEPT", "RECLASSIFY", "NONE", "UNRESOLVED") -DECISION_ALIASES = { - "ACCEPT": "ACCEPT", - "ACCEPT FAMILY": "ACCEPT", - "RECLASSIFY": "RECLASSIFY", - "CHOOSE DIFFERENT FAMILY": "RECLASSIFY", - "NONE": "NONE", - "UNRESOLVED": "UNRESOLVED", -} -ATTESTS = ("OBSERVED", "INFERRED") -REGISTERS = ("slang", "domain-specific", "high-register", "general") -FUNCTIONS = ("address", "evaluation", "intensification", "reference", "affiliation") -SURFACES = { - "A": "CONFIRM", - "B": "PROPOSED FAMILY", - "C": "DISAMBIGUATE", - "D": "RIGHTS BLOCKED", - "HELD_OUTSIDE_LANES": "hint-only", -} -SHEET_COLUMNS = ( - "lane", - "row_key", - "text", - "source_hint", - "source_hint_provenance", - "current_label", - "label_source", - "rights", - "proposed_family_evidence", - "disambiguation_options", - "semantic_family", - "attest", - "register", - "function", - "decision", -) -SETTLED_DECISIONS = frozenset({"ACCEPT", "RECLASSIFY", "NONE"}) -_CLEARED_SOURCES = { - "wiktionary_category": "CC-BY-SA", - "wikipedia_prose": "CC-BY-SA", -} -_BLOCKED_SOURCES = { - "reddit_title": "RIGHTS_UNRESOLVED", - "kym_slang_list": "RIGHTS_UNRESOLVED", -} -_SCORE_KEYS = frozenset( - {"candidate_score", "model_error", "model_score", "prediction"} -) -_FORBIDDEN_KEYS = frozenset({"text", "normalized_text", "raw_text"}) | _SCORE_KEYS -_HINT_PROVENANCE = "stream.family_hint_not_a_label" - - -def refuse(message: str) -> None: - raise SystemExit(f"REFUSE: {message}") - - -def family_flags(name: str, *, activated: set[str] | None = None) -> dict[str, bool]: - """Three independent flags. Evaluation settlement does not enable a family.""" - activated = set(activated or ()) - if name in ACTIVE_FAMILIES: - return { - "taxonomy.active": True, - "evaluation.enabled": False, - "production.enabled": False, - } - if name in CANDIDATE_FAMILIES: - return { - "taxonomy.active": name in activated, - "evaluation.enabled": False, - "production.enabled": False, - } - refuse(f"family is not in the evaluation taxonomy: {name}") - raise AssertionError("refuse") - - -def _activated(activated: set[str] | None) -> set[str]: - extra = set(activated or ()) - unknown = sorted(extra - set(CANDIDATE_FAMILIES)) - if unknown: - refuse(f"activation refused for {unknown[0]}") - return extra - - -def _blank(value: Any) -> bool: - return value is None or (isinstance(value, str) and not value.strip()) - - -def _optional_text(value: Any) -> str | None: - if _blank(value): - return None - if not isinstance(value, str): - refuse("settlement field is not text") - return value.strip() - - -def _decision(value: Any) -> str: - token = _optional_text(value) - if token is None: - refuse("decision is empty") - mapped = DECISION_ALIASES.get(token.upper()) - if mapped is None: - if token.upper() == "REJECT": - refuse("reject is a production attest token, not an evaluation decision") - refuse(f"invalid decision: {token}") - return mapped - - -def _attest(value: Any) -> str | None: - token = _optional_text(value) - if token is None: - return None - mapped = token.upper() - if mapped not in ATTESTS: - refuse(f"invalid attest: {token}") - return mapped - - -def _family_token(value: Any) -> str | None: - token = _optional_text(value) - if token is None: - return None - return token.casefold() - - -def _closed_vocab(value: Any, allowed: tuple[str, ...], label: str) -> str | None: - token = _optional_text(value) - if token is None: - return None - mapped = token.casefold() - if mapped not in allowed: - refuse(f"invalid {label}: {token}") - return mapped - - -def _resolve_family(token: str | None, *, activated: set[str], allow_null: bool) -> str | None: - if token is None: - if allow_null: - return None - refuse("explicit semantic_family is required") - if token == ABSTAIN: - return ABSTAIN - if token in CANDIDATE_FAMILIES and token not in activated: - refuse(f"candidate family is inactive: {token}") - if token in ACTIVE_FAMILIES or token in activated: - return token - refuse(f"family is not in the evaluation taxonomy: {token}") - raise AssertionError("refuse") - - -def rights_of(stream_row: Mapping[str, Any]) -> str: - source_type = str(stream_row.get("source_type") or "") - if source_type in _CLEARED_SOURCES: - return _CLEARED_SOURCES[source_type] - if source_type in _BLOCKED_SOURCES: - return _BLOCKED_SOURCES[source_type] - refuse(f"unknown rights source_type: {source_type or 'missing'}") - raise AssertionError("refuse") - - -def evidence_hint(stream_row: Mapping[str, Any]) -> str | None: - hint = stream_row.get("family_hint_not_a_label") - if _blank(hint): - return None - return str(hint) - - -def sheet_hint_token(hint: str | None) -> str: - return "NO_HINT" if hint is None else hint - - -def sheet_label_token(label: Any) -> str: - if label is None: - return "UNLABELLED" - return str(label) - - -def settlement_allows_reserve(record: Mapping[str, Any]) -> bool: - """Rights and decision gates. This predicate does not admit a row.""" - if record.get("decision") not in SETTLED_DECISIONS: - return False - if record.get("rights") == "RIGHTS_UNRESOLVED": - return False - family = record.get("semantic_family") - attest = record.get("attest") - if family is None or attest not in ATTESTS: - return False - if attest == "OBSERVED" and record.get("decision") is None: - return False - return True - - -def _forbid_payload(value: Mapping[str, Any], where: str) -> None: - for key in value: - if key in _FORBIDDEN_KEYS: - refuse(f"{where} must not carry {key}") - - -def _refuse_scores(value: Mapping[str, Any], where: str) -> None: - for key in value: - if key in _SCORE_KEYS: - refuse(f"{where} must not carry {key}") - - -def _canonical_hash(stream_row: Mapping[str, Any]) -> str: - digest = normalized_text_sha256(str(stream_row.get("text") or "")) - stored = stream_row.get("normtext_sha256") - if isinstance(stored, str) and stored and stored != digest: - refuse("stream normtext_sha256 drifted from canonical row identity") - return digest - - -def _coerce_choice( - decision: str, - family: str | None, - attest: str | None, - proposed: str | None, - hint: str | None, -) -> tuple[str | None, str | None]: - if decision == "UNRESOLVED": - if family is not None or attest is not None: - refuse("UNRESOLVED requires null semantic_family and null attest") - return None, None - if decision == "NONE": - if family not in (None, ABSTAIN): - refuse("NONE requires semantic_family none") - if attest is None: - refuse("NONE requires an explicit attest") - return ABSTAIN, attest - if decision == "ACCEPT": - if family is None: - refuse("ACCEPT requires an explicit semantic_family") - if attest is None: - refuse("ACCEPT requires an explicit attest") - if proposed is not None and family != proposed: - refuse("ACCEPT does not match the proposed family") - return family, attest - if decision == "RECLASSIFY": - if family is None: - refuse("RECLASSIFY requires an explicit family") - if attest is None: - refuse("RECLASSIFY requires an explicit attest") - evidence = proposed if proposed is not None else hint - if evidence is not None and family == evidence: - refuse("RECLASSIFY family must differ from the proposed evidence family") - return family, attest - refuse(f"invalid decision: {decision}") - raise AssertionError("refuse") - - -def validate_settlement( - incoming: Mapping[str, Any], - stream_row: Mapping[str, Any], - *, - activated: set[str] | None = None, - prior_ids: set[str] | None = None, -) -> dict[str, Any]: - """Fail closed. Does not read ``label_source`` as ``attest``.""" - _forbid_payload(incoming, "settlement") - extra = _activated(activated) - row_id = _optional_text(incoming.get("row_id")) or _optional_text(stream_row.get("row_key")) - if not row_id or row_id != stream_row.get("row_key"): - refuse("row_id is not in the held-out stream") - if row_id in set(prior_ids or ()): - refuse(f"duplicate settlement for {row_id}") - digest = _canonical_hash(stream_row) - claimed = _optional_text(incoming.get("text_hash")) - if claimed is None or claimed != digest: - refuse(f"text_hash does not match canonical row identity for {row_id}") - hint = evidence_hint(stream_row) - supplied_hint = incoming.get("source_hint") - if _blank(supplied_hint): - supplied_hint = None - elif isinstance(supplied_hint, str) and supplied_hint.strip() == "NO_HINT": - supplied_hint = None - elif isinstance(supplied_hint, str): - supplied_hint = supplied_hint.strip() - else: - refuse("source_hint is not text") - if supplied_hint != hint: - refuse(f"source_hint does not match held-out evidence for {row_id}") - rights = rights_of(stream_row) - sheet_rights = _optional_text(incoming.get("rights")) - if sheet_rights is not None and sheet_rights != rights: - refuse(f"rights state does not match held-out evidence for {row_id}") - decision = _decision(incoming.get("decision")) - proposed = _family_token(incoming.get("proposed_family")) - register = _closed_vocab(incoming.get("register"), REGISTERS, "register") - function = _closed_vocab(incoming.get("function"), FUNCTIONS, "function") - attest = _attest(incoming.get("attest")) - family = _family_token(incoming.get("semantic_family")) - if decision == "UNRESOLVED" and (register is not None or function is not None): - refuse("UNRESOLVED cannot carry register or function") - family, attest = _coerce_choice(decision, family, attest, proposed, hint) - family = _resolve_family(family, activated=extra, allow_null=decision == "UNRESOLVED") - operator = _optional_text(incoming.get("operator")) - settled_at = _optional_text(incoming.get("settled_at")) - provenance = _optional_text(incoming.get("provenance")) - if not operator: - refuse("operator is required") - if not settled_at: - refuse("settled_at is required") - if not provenance: - refuse("provenance is required") - lane = _optional_text(incoming.get("lane")) - if lane is not None and lane not in SURFACES: - refuse(f"unknown settlement surface: {lane}") - record = { - "schema": SCHEMA, - "row_id": row_id, - "text_hash": digest, - "decision": decision, - "semantic_family": family, - "attest": attest, - "register": register, - "function": function, - "source_hint": hint, - "operator": operator, - "settled_at": settled_at, - "provenance": provenance, - "rights": rights, - "lane": lane, - } - _forbid_payload(record, "settlement record") - return record - - -def _stream_index(rows: Sequence[Mapping[str, Any]]) -> dict[str, Mapping[str, Any]]: - index: dict[str, Mapping[str, Any]] = {} - for row in rows: - _refuse_scores(row, "held-out row") - key = row.get("row_key") - if not isinstance(key, str) or not key: - refuse("held-out row is missing row_key") - if key in index: - refuse(f"duplicate held-out row_id {key}") - index[key] = row - return index - - -def load_stream(path: str | Path) -> dict[str, Mapping[str, Any]]: - rows = [] - with Path(path).open(encoding="utf-8") as handle: - for line in handle: - if line.strip(): - payload = json.loads(line) - if not isinstance(payload, dict): - refuse("held-out stream row is not an object") - rows.append(payload) - return _stream_index(rows) - - -def _check_sheet_integrity( - parts: Sequence[str], - header: Sequence[str], - stream_row: Mapping[str, Any], -) -> str: - cell = dict(zip(header, parts)) - row_id = cell["row_key"].strip() - if row_id != stream_row.get("row_key"): - refuse("row_id is not in the held-out stream") - digest = normalized_text_sha256(cell["text"]) - if digest != _canonical_hash(stream_row): - refuse(f"text_hash does not match canonical row identity for {row_id}") - hint = evidence_hint(stream_row) - if cell["source_hint"] != sheet_hint_token(hint): - refuse(f"source_hint does not match held-out evidence for {row_id}") - if cell["source_hint_provenance"] != _HINT_PROVENANCE: - refuse(f"source_hint provenance does not match for {row_id}") - if cell["current_label"] != sheet_label_token(stream_row.get("label")): - refuse(f"current_label does not match held-out evidence for {row_id}") - if cell["label_source"] != str(stream_row.get("label_source") or ""): - refuse(f"label_source does not match held-out evidence for {row_id}") - if cell["rights"].strip() != rights_of(stream_row): - refuse(f"rights state does not match held-out evidence for {row_id}") - lane = cell["lane"].strip() - if lane not in SURFACES: - refuse(f"unknown settlement surface: {lane}") - return digest - - -def _operator_cells(cell: Mapping[str, str]) -> dict[str, str]: - return { - "semantic_family": cell["semantic_family"], - "attest": cell["attest"], - "register": cell["register"], - "function": cell["function"], - "decision": cell["decision"], - } - - -def parse_sheet( - path: str | Path, - stream: Mapping[str, Mapping[str, Any]], - *, - operator: str, - provenance: str, - settled_at: str, - activated: set[str] | None = None, - prior_ids: set[str] | None = None, -) -> dict[str, Any]: - """Validate a private lane sheet. Does not write the sheet or fill decisions.""" - lines = Path(path).read_text(encoding="utf-8").splitlines() - if not lines: - refuse("settlement sheet is empty") - header = tuple(lines[0].split("\t")) - if header != SHEET_COLUMNS: - refuse("sheet header does not match the operator lane contract") - records = [] - unset = 0 - lane_rows: dict[str, int] = {} - seen: set[str] = set() - errors: list[str] = [] - for line in lines[1:]: - if not line.strip(): - continue - parts = line.split("\t") - if len(parts) != len(header): - errors.append("REFUSE: sheet row is not rectangular") - continue - row_id = parts[1].strip() - stream_row = stream.get(row_id) - if stream_row is None: - errors.append(f"REFUSE: row_id is not in the held-out stream: {row_id}") - continue - if row_id in seen or row_id in set(prior_ids or ()): - errors.append(f"REFUSE: duplicate settlement for {row_id}") - continue - seen.add(row_id) - try: - digest = _check_sheet_integrity(parts, header, stream_row) - cell = dict(zip(header, parts)) - lane = cell["lane"].strip() - lane_rows[lane] = lane_rows.get(lane, 0) + 1 - chosen = _operator_cells(cell) - if all(_blank(value) for value in chosen.values()): - unset += 1 - continue - if _blank(chosen["decision"]): - refuse(f"decision is empty for {row_id}; refusing to infer one") - record = validate_settlement( - { - "row_id": row_id, - "text_hash": digest, - "decision": chosen["decision"], - "semantic_family": chosen["semantic_family"], - "attest": chosen["attest"], - "register": chosen["register"], - "function": chosen["function"], - "source_hint": evidence_hint(stream_row), - "proposed_family": cell["proposed_family_evidence"], - "operator": operator, - "settled_at": settled_at, - "provenance": provenance, - "rights": cell["rights"], - "lane": lane, - }, - stream_row, - activated=activated, - prior_ids=prior_ids, - ) - records.append(record) - except SystemExit as exc: - errors.append(str(exc)) - if errors: - refuse("; ".join(message.removeprefix("REFUSE: ") for message in errors)) - return {"records": records, "unset_row_count": unset, "lane_rows": lane_rows, "row_ids": seen} - - -def load_record_file( - path: str | Path, - stream: Mapping[str, Mapping[str, Any]], - *, - operator: str, - provenance: str, - settled_at: str, - activated: set[str] | None = None, - prior_ids: set[str] | None = None, -) -> list[dict[str, Any]]: - records = [] - seen = set(prior_ids or ()) - with Path(path).open(encoding="utf-8") as handle: - for line in handle: - if not line.strip(): - continue - payload = json.loads(line) - if not isinstance(payload, dict): - refuse("settlement record is not an object") - row_id = payload.get("row_id") - if not isinstance(row_id, str) or row_id not in stream: - refuse(f"row_id is not in the held-out stream: {row_id}") - if row_id in seen: - refuse(f"duplicate settlement for {row_id}") - payload = dict(payload) - payload.setdefault("operator", operator) - payload.setdefault("provenance", provenance) - payload.setdefault("settled_at", settled_at) - records.append( - validate_settlement( - payload, - stream[row_id], - activated=activated, - prior_ids=prior_ids, - ) - ) - seen.add(row_id) - return records - - -def _decision_counts(records: Sequence[Mapping[str, Any]]) -> dict[str, int]: - counts = {name: 0 for name in DECISIONS} - for record in records: - counts[str(record["decision"])] += 1 - return counts - - -def _family_counts(records: Sequence[Mapping[str, Any]]) -> dict[str, int]: - counts = {name: 0 for name in (*ACTIVE_FAMILIES, ABSTAIN)} - for record in records: - if record["decision"] not in SETTLED_DECISIONS: - continue - family = record.get("semantic_family") - if family in counts: - counts[str(family)] += 1 - return counts - - -def build_receipt( - records: Sequence[Mapping[str, Any]], - *, - batch_id: str, - operator: str, - provenance: str, - unset_row_count: int, - lane_rows: Mapping[str, int], - stream_run_id: str = "", - settled_at: str = "", - input_sheets: Sequence[Mapping[str, str]] | None = None, -) -> dict[str, Any]: - if not str(batch_id or "").strip(): - refuse("batch_id is required") - if not str(operator or "").strip() or not str(provenance or "").strip(): - refuse("operator and provenance are required") - settled = [record for record in records if record["decision"] in SETTLED_DECISIONS] - unresolved = [record for record in records if record["decision"] == "UNRESOLVED"] - body = { - "schema": RECEIPT_SCHEMA, - "batch_id": batch_id, - "operator": operator, - "provenance": provenance, - "settled_row_count": len(settled), - "unresolved_row_count": len(unresolved), - "unset_row_count": unset_row_count, - "decision_counts": _decision_counts(records), - "family_counts": _family_counts(records), - "observed_count": sum(1 for record in settled if record.get("attest") == "OBSERVED"), - "inferred_count": sum(1 for record in settled if record.get("attest") == "INFERRED"), - "rights_blocked_settled_count": sum( - 1 for record in settled if record.get("rights") == "RIGHTS_UNRESOLVED" - ), - "reserve_eligible_count": sum(1 for record in records if settlement_allows_reserve(record)), - "reserve_added": 0, - "ledger_mutated": False, - "vendor_calls": VENDOR_CALLS, - "evaluation_enabled": False, - "lane_rows": dict(lane_rows), - } - run_id = str(stream_run_id or "").strip() - if run_id: - body["stream_run_id"] = run_id - stamp = str(settled_at or "").strip() - if stamp: - body["settled_at"] = stamp - if input_sheets: - sheets = [] - for item in input_sheets: - identity = str(item.get("identity") or "").strip() - digest = str(item.get("sha256") or "").strip() - if not identity or len(digest) != 64: - refuse("input sheet identity is incomplete") - sheets.append({"identity": identity, "sha256": digest}) - body["input_sheets"] = sheets - _walk_forbid(body) - return seal_receipt(body) - - -def seal_receipt(body: Mapping[str, Any]) -> dict[str, Any]: - payload = {key: value for key, value in body.items() if key != "receipt_sha256"} - raw = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8") - sealed = dict(payload) - sealed["receipt_sha256"] = hashlib.sha256(raw).hexdigest() - return sealed - - -def receipt_sha_matches(receipt: Mapping[str, Any]) -> bool: - sealed = seal_receipt(receipt) - return sealed["receipt_sha256"] == receipt.get("receipt_sha256") - - -def _walk_forbid(value: Any) -> None: - if isinstance(value, dict): - for key, item in value.items(): - if key in _FORBIDDEN_KEYS: - refuse(f"receipt must not carry {key}") - _walk_forbid(item) - elif isinstance(value, list): - for item in value: - _walk_forbid(item) - - -def load_logged_ids(path: str | Path) -> set[str]: - log = Path(path) - if not log.exists(): - return set() - found = set() - for line in log.read_text(encoding="utf-8").splitlines(): - if not line.strip(): - continue - payload = json.loads(line) - if not isinstance(payload, dict): - refuse("settlement log row is not an object") - _forbid_payload(payload, "settlement log") - row_id = payload.get("row_id") - if not isinstance(row_id, str) or not row_id: - refuse("settlement log row is missing row_id") - if row_id in found: - refuse(f"settlement log already contains a duplicate for {row_id}") - found.add(row_id) - return found - - -def _write_exclusive(path: Path, text: str) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) - with os.fdopen(fd, "w", encoding="utf-8") as handle: - handle.write(text) - - -def commit_settlement( - records: Sequence[Mapping[str, Any]], - receipt: Mapping[str, Any], - *, - log_path: str | Path, - receipt_path: str | Path, -) -> dict[str, Any]: - """Append settlement rows, then write a new receipt. Never rewrites either.""" - receipt_file = Path(receipt_path) - if receipt_file.exists(): - refuse(f"receipt already exists: {receipt_file}") - log = Path(log_path) - prior = load_logged_ids(log) - fresh_ids = [] - for record in records: - _forbid_payload(record, "settlement record") - row_id = str(record["row_id"]) - if row_id in prior or row_id in fresh_ids: - refuse(f"duplicate settlement for {row_id}") - fresh_ids.append(row_id) - if not log.exists(): - _write_exclusive(log, "") - if records: - with log.open("a", encoding="utf-8") as handle: - for record in records: - handle.write(json.dumps(record, sort_keys=True) + "\n") - sealed = dict(receipt) - _walk_forbid(sealed) - _write_exclusive(receipt_file, json.dumps(sealed, indent=2, sort_keys=True) + "\n") - os.chmod(log, 0o600) - return sealed - - -def settle( - stream_rows: Sequence[Mapping[str, Any]], - decisions: Sequence[Mapping[str, Any]], - *, - batch_id: str, - operator: str, - provenance: str, - settled_at: str, - activated: set[str] | None = None, - prior_ids: set[str] | None = None, - unset_row_count: int = 0, - lane_rows: Mapping[str, int] | None = None, -) -> dict[str, Any]: - """Validate explicit decisions. Blank callers pass no decisions and settle nothing.""" - stream = _stream_index(stream_rows) - prior = set(prior_ids or ()) - records = [] - seen: set[str] = set() - for incoming in decisions: - row_id = incoming.get("row_id") - stream_row = stream.get(row_id) if isinstance(row_id, str) else None - if stream_row is None: - refuse(f"row_id is not in the held-out stream: {row_id}") - if row_id in seen: - refuse(f"duplicate settlement for {row_id}") - seen.add(str(row_id)) - payload = dict(incoming) - payload.setdefault("operator", operator) - payload.setdefault("provenance", provenance) - payload.setdefault("settled_at", settled_at) - records.append( - validate_settlement(payload, stream_row, activated=activated, prior_ids=prior) - ) - receipt = build_receipt( - records, - batch_id=batch_id, - operator=operator, - provenance=provenance, - unset_row_count=unset_row_count, - lane_rows=dict(lane_rows or {}), - ) - return {"records": records, "receipt": receipt} - - -def run_settlement_apply(args: Any) -> int: - """CLI entry. Reads operator cells. Does not choose them or touch the ledger.""" - import sys - - stream = load_stream(args.stream_rows) - activated = set(args.activated_family or []) - sheets = list(args.sheet or []) - records_path = str(args.records or "").strip() - if not sheets and not records_path: - refuse("no operator surface supplied") - prior = load_logged_ids(args.settlement_log) - records: list[dict[str, Any]] = [] - unset = 0 - lane_rows: dict[str, int] = {} - decided: set[str] = set() - surfaced: set[str] = set() - for sheet in sheets: - parsed = parse_sheet( - sheet, - stream, - operator=args.operator, - provenance=args.provenance, - settled_at=args.settled_at, - activated=activated, - prior_ids=prior | decided, - ) - for row_id in parsed["row_ids"]: - if row_id in surfaced: - refuse(f"duplicate settlement for {row_id}") - surfaced.add(row_id) - for record in parsed["records"]: - decided.add(record["row_id"]) - records.append(record) - unset += int(parsed["unset_row_count"]) - for lane, count in parsed["lane_rows"].items(): - lane_rows[lane] = lane_rows.get(lane, 0) + count - if records_path: - loaded = load_record_file( - records_path, - stream, - operator=args.operator, - provenance=args.provenance, - settled_at=args.settled_at, - activated=activated, - prior_ids=prior | decided, - ) - for record in loaded: - if record["row_id"] in decided or record["row_id"] in surfaced: - refuse(f"duplicate settlement for {record['row_id']}") - decided.add(record["row_id"]) - records.append(record) - sheet_identities = [] - for sheet in sheets: - raw = Path(sheet).read_bytes() - sheet_identities.append( - {"identity": Path(sheet).name, "sha256": hashlib.sha256(raw).hexdigest()} - ) - receipt = build_receipt( - records, - batch_id=args.batch_id, - operator=args.operator, - provenance=args.provenance, - unset_row_count=unset, - lane_rows=lane_rows, - stream_run_id=str(getattr(args, "stream_run_id", "") or ""), - settled_at=str(args.settled_at or ""), - input_sheets=sheet_identities, - ) - sealed = commit_settlement( - records, - receipt, - log_path=args.settlement_log, - receipt_path=args.receipt, - ) - sys.stdout.write(json.dumps(sealed, indent=2, sort_keys=True) + "\n") - return 0 +PLACEHOLDER \ No newline at end of file diff --git a/specs/007-hyperlexical-model/label-taxonomy-proposal.md b/specs/007-hyperlexical-model/label-taxonomy-proposal.md index 1f656616..311c8dd0 100644 --- a/specs/007-hyperlexical-model/label-taxonomy-proposal.md +++ b/specs/007-hyperlexical-model/label-taxonomy-proposal.md @@ -1,361 +1 @@ -# Label taxonomy proposal — evaluation reserve - -**Status:** structure accepted 2026-09-26 with amendments (see Operator acceptance). Row settlement not applied. `evaluation.enabled` is false. -**Date:** 2026-09-26 -**Shelf:** held-out stream `hs-20260925T211358Z` (339 rows). No row text in this file. -**Does not:** change `hyperlexical.layout.FAMILIES`, resize the classify head, call Jev, run `attest-apply`, or admit `EVAL_RESERVE`. - -The production head stays the nine-way list in `layout.py`: `betting-sharp`, `crypto-degen`, `ai-native`, `brainrot-aura`, `kinship-address`, `political-status`, `gaming-meta`, `workplace-corp`, `none`. Dataset schema `dataset_row.v0.1` keeps that lineage enum, plus `ytd_leaf`. This proposal is the evaluation-label ontology for the reserve. It becomes live only after an operator accepts the definitions. - -## Why the current reserve cannot use the old list - -On the 339-row shelf the only rights-cleared non-none labels are `gaming-meta` (99), `betting-sharp` (61), and `crypto-degen` (5). Five production families have no settled label. Macro-F1 over a three-class subset is a narrow score. Adding more rows under those three names does not widen it. - -Hint-only rows already name `kinship-address`, `political-status`, `workplace-corp`, `brainrot-aura`, and `ai-native`. Those hints are not labels. They show the queued sheet is broader than the Kaikki topic map, which is allowed to emit only betting, crypto, and gaming. - -## Four surfaces - -Distinctions that are not a stable semantic region stay off the primary label. - -| Surface | Role | Values on this proposal | -|---|---|---| -| `semantic_family` | Primary class. One region per row. | The sixteen names below, `none`, or `LABEL_UNRESOLVED`. | -| `attest` | How the label was settled. Orthogonal to family. | `OBSERVED`, `INFERRED`, `UNLABELLED`. Empty attest stays empty. | -| `register` | How the wording sits in the language. | `slang`, `domain-specific`, `high-register`, `general`. Unset until an operator sets it. | -| `function` | What the phrase is doing, when that is not the family. | `address`, `evaluation`, `intensification`, `reference`, `affiliation`, or unset. | - -`OBSERVED` / `INFERRED` stay on `attest`. They do not become families. Tone, certainty, and insider/outsider stay off the primary list. A family name is not combined with an attribute (`gaming-meta-observed-negative-insider` is not a class). - -`attest` on this shelf is unchanged: 205 `INFERRED`, 134 `UNLABELLED`, 0 `OBSERVED`. This document does not write the attest column. - -## Promotion rule - -```text -LABEL_PROMOTE(x) = - definition_clear - AND operator_agreement_possible - AND not_redundant_with_existing_label - AND sufficient_support_exists -``` - -`sufficient_support_exists` is `NOT_COMPUTABLE`. No numeric minimum is declared. Until that rule passes, a name is `CANDIDATE_LABEL` and `evaluation.enabled` is false. A single-example class is not an evaluation class. - -Nothing in this draft is promoted. The sixteen names are the proposed active set for the operator to accept or cut. Acceptance of a definition is not activation for macro-F1. - -## Proposed active set (16 non-none) - -`none` remains the abstain class. It is not one of the sixteen. - -Each block is a definition the operator can settle without a model. Positive and near-miss lines are illustrations of the region. They are not reserve rows and they are not taken from the stream. - -### gaming-meta - -- **definition:** Jargon of video games, esports, and gamer communities: mechanics, ranked play, balance, party and lobby talk. -- **include:** Terms whose ordinary sense is a game mechanic, a competitive-play judgment, or a gamer-community formula. -- **exclude:** Athletic sports. Betting lines about sports. A meme that merely mentions a game. -- **near-miss:** A sports-competition term. A general insult with no game sense. -- **parent:** production family of the same name. Kaikki topics `video-games`, `computer-games`, `role-playing-games` already map here. -- **evaluation.enabled:** false. - -### betting-sharp - -- **definition:** Jargon of sports betting, gambling, and poker: odds, lines, bankroll, handicapping. -- **include:** Terms whose ordinary sense is a wager, a line, a stake, or poker-table talk. -- **exclude:** Ordinary sports commentary with no stake. Crypto trading slang. A game mechanic. -- **near-miss:** `sports-competition`. A single word that is also a normal verb. -- **parent:** production family of the same name. Kaikki topics `gambling` and `poker` already map here. -- **evaluation.enabled:** false. - -### crypto-degen - -- **definition:** Jargon of cryptocurrency, DeFi, NFTs, and speculative token trading. -- **include:** Terms whose ordinary sense is a chain, a token, a trade, or that community's stake in a position. -- **exclude:** Ordinary finance with no token or chain sense. A meme about money in general. -- **near-miss:** `finance-retail` and `market-structure` (both candidates, not active). -- **parent:** production family of the same name. -- **evaluation.enabled:** false. - -### internet-slang - -- **definition:** Short-lived or platform-native wording that is slang of general internet speech and is not tied to one trade or hobby. -- **include:** Terms a reader places in internet speech without needing a game, a book, a chart, or a workplace. -- **exclude:** A domain term that happens to be posted online. A meme name whose job is the meme itself (`memetic`). A status claim (`social-status`). -- **near-miss:** `brainrot-aura` rows, which mix this family with `memetic` and `social-status`. They stay unresolved rather than all landing here. -- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. - -### memetic - -- **definition:** A phrase whose primary job is to circulate as a named meme, copypasta, or recognizable bit. -- **include:** The wording is the meme, or a stable mutation of one. -- **exclude:** Ordinary slang that is not a named bit. A domain term that people joke about. -- **near-miss:** `internet-slang`. Humor as a tone is not this family. -- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. - -### social-status - -- **definition:** Wording whose primary job is to rank people, scenes, or the self: aura, clout, mid, cooked-as-status, and their kin. -- **include:** The term assigns or withholds standing. -- **exclude:** A game rank that is a mechanic (`gaming-meta`). A political allegiance (`politics-civic`, candidate). -- **near-miss:** `approval-disapproval`, when the phrase judges an act rather than a person's standing. -- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. - -### relationship-dating - -- **definition:** Jargon of dating, romance, and couple-craft as a social practice. -- **include:** Terms whose ordinary sense is a dating move, a romantic role, or couple-status. -- **exclude:** Kinship and address (`bro`, `sis`, family terms). Those are not this family. They do not map here. -- **near-miss:** Production `kinship-address`. Candidate `identity-affiliation`. -- **evaluation.enabled:** false. Support on this shelf: 0. - -### approval-disapproval - -- **definition:** A stable region of evaluative slang whose job is to praise, dismiss, or rate a thing. -- **include:** The term is a verdict word, not a domain object. -- **exclude:** The same verdict spoken inside a domain, when the domain is the family and evaluation is only the `function`. A gaming phrase that evaluates a play stays `gaming-meta` with `function: evaluation`. -- **near-miss:** `social-status`. Positive versus negative is an attribute, not two families. -- **evaluation.enabled:** false. Support on this shelf: 0. - -### conflict-aggression - -- **definition:** Jargon of fights, feuds, call-outs, and competitive hostility that is not a sport, a game, or a bet. -- **include:** The term names a conflict move or a hostile stance. -- **exclude:** Athletic competition. In-game combat vocabulary. Criminal procedure (`crime-illicit`, candidate). -- **near-miss:** `sports-competition`. An insult that is only status (`social-status`). -- **evaluation.enabled:** false. Support on this shelf: 0. - -### technology-ai - -- **definition:** Jargon of AI, machine learning, software builders, and AI-era coinages. -- **include:** Terms whose ordinary sense is a model, a training run, an agent, an eval, or builder talk. -- **exclude:** Ordinary computing words with no community sense. The weak mapper already refuses bare `en:Computing`. -- **near-miss:** Production `ai-native`, which this name would replace only after operator acceptance. The head is not renamed here. -- **evaluation.enabled:** false. This shelf has 5 hint-only rows, 0 labels. - -### work-hustle - -- **definition:** Jargon of jobs, offices, careers, and hustle culture. -- **include:** Terms whose ordinary sense is workplace status, management talk, or gig-work craft. -- **exclude:** A hobby called a grind. Crypto or betting talk about "the job." -- **near-miss:** Production `workplace-corp`, the same region under the old name. -- **evaluation.enabled:** false. This shelf has 20 hint-only rows, 0 labels. - -### sports-competition - -- **definition:** Jargon of athletic sports and sporting competition, with no wager and no video game. -- **include:** Terms whose ordinary sense is a play, a position, or a sporting result. -- **exclude:** Betting lines (`betting-sharp`). Video-game play (`gaming-meta`). Bare `en:Sports` stays unlabeled, matching `weak_tag_family.py`. -- **near-miss:** `betting-sharp`. -- **evaluation.enabled:** false. Support on this shelf: 0. - -### music-entertainment - -- **definition:** Jargon of music scenes, fandom, and stage entertainment. -- **include:** Terms whose ordinary sense is a scene, a track-craft word, or fandom talk. -- **exclude:** A meme that uses a song title. A fashion term. -- **near-miss:** `memetic`. `fashion-aesthetic`. -- **evaluation.enabled:** false. Support on this shelf: 0. - -### fashion-aesthetic - -- **definition:** Jargon of dress, look, and aesthetic scenes. -- **include:** Terms whose ordinary sense is a look, a garment-community word, or an aesthetic label. -- **exclude:** Status slang with no look. A costume inside a game. -- **near-miss:** `social-status`. -- **evaluation.enabled:** false. Support on this shelf: 0. - -### regional-cultural - -- **definition:** Wording whose primary identity is a place, a dialect community, or a local scene, rather than a trade. -- **include:** The term is opaque outside that region or dialect, and the region is the region of meaning. -- **exclude:** A domain term that happens to be used in one city. AAVE or dialect material is not dumped here by default; it stays unresolved until an operator can say the region is the family. -- **near-miss:** `internet-slang`. -- **evaluation.enabled:** false. Support on this shelf: 0. - -### spiritual-mystic - -- **definition:** Jargon of spiritual, occult, and mystic scenes. -- **include:** Terms whose ordinary sense is a practice, a belief-community word, or a ritual formula in that scene. -- **exclude:** Ordinary metaphor ("manifest" as office talk). A meme about fate. -- **near-miss:** `work-hustle` when the word is motivational rather than mystic. -- **evaluation.enabled:** false. Support on this shelf: 0. - -## Candidate labels (not active) - -These stay `CANDIDATE_LABEL`. They are not evaluation classes. The broader list in the operator note is the pool; this shelf does not justify promoting them. - -| Candidate | Holds | Why it is not active | -|---|---|---| -| `politics-civic` | Production `political-status` (20 hint-only rows). | Not in the sixteen. No settled examples. | -| `identity-affiliation` | Production `kinship-address` (20 hint-only rows). | Address and kinship are not `relationship-dating`. The candidate is the holding pen, not a gold family. | -| `finance-retail` | Nothing on this shelf. | Easy to collapse into `crypto-degen`. | -| `market-structure` | Nothing on this shelf. | Easy to collapse into `crypto-degen` or `betting-sharp`. | -| `sexual-romantic` | Nothing on this shelf. | Overlaps `relationship-dating` once that family has a definition. | -| `substance-party` | Nothing on this shelf. | No cluster on this shelf. | -| `crime-illicit` | Nothing on this shelf. | No cluster on this shelf. | -| `health-fitness` | Nothing on this shelf. | No cluster on this shelf. | - -`ytd_leaf` stays a schema enum value for the production dataset. It is not a semantic family in this proposal. - -## What was not done - -- Jev was not called. The lineage rule was not run. No model scored a row. -- `attest-apply` was not run. The attest column stays empty. -- The ledger was not mutated. `EVAL_RESERVE` stays 0. -- `layout.FAMILIES` was not edited. -- `brainrot-aura` was not split by reading phrases. - -## Remap of `hs-20260925T211358Z` - -Rules, in order. A hint is never upgraded to a label by this file. - -1. Rights-cleared row, `label_source=INFERRED`, label equals hint, label is `gaming-meta`, `betting-sharp`, `crypto-degen`, or `none`: `PROPOSED_REMAP` onto the same `semantic_family`. `attest` stays `INFERRED`. Not reserved. -2. Label and hint disagree: `LABEL_UNRESOLVED`. -3. `UNLABELLED`, including every hint-only family: `LABEL_UNRESOLVED`. A proposed target may be recorded for the operator. It is not a label. -4. Reddit and Know Your Meme rows: `RIGHTS_UNRESOLVED` as well as `LABEL_UNRESOLVED`. - -| Old label | Hint | Rows | Remap status | Proposed family | Notes | -|---|---|---|---|---|---| -| gaming-meta | gaming-meta | 98 | `PROPOSED_REMAP` | `gaming-meta` | Rights-clear. One disagreeing row is excluded. | -| betting-sharp | betting-sharp | 61 | `PROPOSED_REMAP` | `betting-sharp` | Rights-clear. | -| crypto-degen | crypto-degen | 5 | `PROPOSED_REMAP` | `crypto-degen` | Rights-clear. Support is thin. Evaluation stays off. | -| none | none | 40 | `PROPOSED_REMAP` | `none` | Encyclopedic prose. `register` may be recorded as `high-register` from `source_type`, not from a model. | -| gaming-meta | betting-sharp | 1 | `LABEL_UNRESOLVED` | — | Label/hint collision. | -| UNLABELLED | gaming-meta | 20 | `LABEL_UNRESOLVED` | — | Hint is not a label. | -| UNLABELLED | betting-sharp | 21 | `LABEL_UNRESOLVED` | — | Hint is not a label. Includes rights-unresolved Reddit rows. | -| UNLABELLED | crypto-degen | 6 | `LABEL_UNRESOLVED` | — | All six are Reddit. Rights unresolved. | -| UNLABELLED | workplace-corp | 20 | `LABEL_UNRESOLVED` | `work-hustle` if the operator accepts the rename | Target is a proposal. | -| UNLABELLED | ai-native | 5 | `LABEL_UNRESOLVED` | `technology-ai` if the operator accepts the rename | Target is a proposal. | -| UNLABELLED | kinship-address | 20 | `LABEL_UNRESOLVED` | candidate `identity-affiliation` | Not `relationship-dating`. | -| UNLABELLED | political-status | 20 | `LABEL_UNRESOLVED` | candidate `politics-civic` | Candidate, not active. | -| UNLABELLED | brainrot-aura | 20 | `LABEL_UNRESOLVED` | — | Spans `internet-slang`, `memetic`, and `social-status`. Not split. | -| UNLABELLED | no hint | 2 | `LABEL_UNRESOLVED` | — | Know Your Meme. Rights unresolved. | - -Totals: `PROPOSED_REMAP` 204 (164 non-none, 40 `none`). `LABEL_UNRESOLVED` 135. Admitted to the reserve: 0. - -### Coverage of the sixteen - -| Family | Proposed remap (not gold, not reserved) | Unresolved hints pointing near it | -|---|---|---| -| gaming-meta | 98 | 20 hint-only, plus 1 collision | -| betting-sharp | 61 | 21 hint-only | -| crypto-degen | 5 | 6 hint-only, rights unresolved | -| work-hustle | 0 | 20 `workplace-corp` hints | -| technology-ai | 0 | 5 `ai-native` hints | -| internet-slang | 0 | inside the 20 `brainrot-aura` bundle | -| memetic | 0 | inside the same bundle | -| social-status | 0 | inside the same bundle | -| relationship-dating | 0 | 0 (kinship was not mapped here) | -| approval-disapproval | 0 | 0 | -| conflict-aggression | 0 | 0 | -| sports-competition | 0 | 0 | -| music-entertainment | 0 | 0 | -| fashion-aesthetic | 0 | 0 | -| regional-cultural | 0 | 0 | -| spiritual-mystic | 0 | 0 | - -Eleven of the sixteen have no row on this shelf. The taxonomy is wider than the shelf. That is intentional. Empty families are not filled by invention, and they are not evaluation classes. - -## Operator settlement - -Taxonomy text is ready for review. Row settlement is not ready. - -- Accept, cut, or rewrite the sixteen definitions before any attest. -- Accept or reject the two renames (`workplace-corp` → `work-hustle`, `ai-native` → `technology-ai`) before those hints are eligible. -- Decide whether `politics-civic` and `identity-affiliation` stay candidates. -- Leave `brainrot-aura` unresolved until an operator splits it by hand. -- Then, and only then, `attest-apply`. This file does not authorize that command. - -`SELECT-003` stays undrafted. GEN-1 was not created. BEST stays `seed-morph78`. - - -## Operator acceptance — 2026-09-26 - -The operator accepted the ontology structure and amended the draft. This section supersedes the sixteen-name list, the name `work-hustle`, and the candidate status of `identity-affiliation` and `politics-civic`. The earlier sections stay as the draft record. The production head in `layout.FAMILIES` is unchanged. - -### Authorized active set (18 non-none) - -`gaming-meta`, `betting-sharp`, `crypto-degen`, `internet-slang`, `memetic`, `social-status`, `relationship-dating`, `approval-disapproval`, `conflict-aggression`, `technology-ai`, `workplace-career`, `sports-competition`, `music-entertainment`, `fashion-aesthetic`, `regional-cultural`, `spiritual-mystic`, `identity-affiliation`, `politics-civic`. - -`none` remains abstain. - -`taxonomy.active` is true for these eighteen. `evaluation.enabled` is false for every one of them until operator-settled support exists and a later governance decision turns evaluation on. Those two flags are independent. - -### Renames and promotions - -- `work-hustle` is not a family. The region is `workplace-career`: corporate jargon, employment and status language, career language, workplace hierarchy, and hustle or grind language. Finer shade sits on `function` or `register`. -- `identity-affiliation` is active. It covers group membership, social belonging, in-group and out-group identity, affiliative address, and role affiliation. It is not `relationship-dating`. Familial address used socially is `semantic_family: identity-affiliation` with `function: address`. -- `politics-civic` is active. It covers political roles, civic identity, governmental status, political-group terminology, and public institutional positioning. It is a descriptive region, not an ideological judgment, and it is not `social-status` or `identity-affiliation`. - -### Still candidates - -`finance-retail`, `market-structure`, `sexual-romantic`, `substance-party`, `crime-illicit`, `health-fitness`. - -`brainrot-aura` is not a family. The source cluster is disambiguated row by row into `internet-slang`, `memetic`, `social-status`, another active family, `none`, or `LABEL_UNRESOLVED`. - -### Source hint is not a label - -```yaml -source_hint: - value: - provenance: -semantic_family: - value: - settled_by: - settled_at: -attest: - value: OBSERVED | INFERRED | UNLABELLED - settled_by: - settled_at: -``` - -A source hint is evidence shown to the operator. It does not become `semantic_family`. An existing `INFERRED` label does not become `OBSERVED`. Jev and any other model do not settle either field. - -### Taxonomy-level mappings (not row settlement) - -| Current condition | Mapping | -|---|---| -| gaming-meta label and hint | `gaming-meta` | -| betting-sharp label and hint | `betting-sharp` | -| crypto-degen label and hint | `crypto-degen` | -| none label and hint | `none` | -| workplace-corp hint | `workplace-career` | -| ai-native hint | `technology-ai` | -| kinship-address hint | `identity-affiliation` | -| political-status hint | `politics-civic` | -| brainrot-aura hint | no batch mapping | -| gaming-meta label with betting-sharp hint | no batch mapping | -| Reddit or Know Your Meme | excluded from `EVAL_RESERVE` until rights are resolved | - -Hint-family mapping accepted is not a row label settled. - -### Settlement lanes for `hs-20260925T211358Z` - -Private sheets, mode 0600, under the stream `attest/` directory. Decision cells are empty. `semantic_family`, `attest`, `register`, and `function` are empty. A proposed family is evidence in its own column, not a preselected answer. - -| Lane | Rows | Operator choice | -|---|---|---| -| A CONFIRM | 204 same-label proposed remaps | `ACCEPT`, `RECLASSIFY`, `UNRESOLVED` | -| B PROPOSED FAMILY | 20 workplace-corp, 5 ai-native, 20 kinship-address, 20 political-status | `ACCEPT FAMILY`, `CHOOSE DIFFERENT FAMILY`, `NONE`, `UNRESOLVED` | -| C DISAMBIGUATE | 20 brainrot-aura, 1 label/hint collision | a listed destination, another active family, `none`, or `UNRESOLVED` | -| D RIGHTS BLOCKED | 14 Reddit, 2 Know Your Meme | semantic notes allowed; reserve admission impossible | - -33 Wiktionary rows are hint-only for `gaming-meta` (20) or `betting-sharp` (13). They are not in the four named lanes. They stay `LABEL_UNRESOLVED` on a private holding sheet. They were not given a batch settlement. - -`attest-apply` was not run. The current command only accepts the production eight plus `none` or `reject`, and it writes `label_source=OBSERVED` for every accepted value. That command must not be used on these lanes: it would reject the new names and would collapse `attest` into `OBSERVED`. - -No row was settled. `EVAL_RESERVE` stays 0. Vendor calls: 0. SELECT-003 was not drafted. - -## Settlement path — 2026-09-26 - -The operator sheets stay the interface: lane A CONFIRM, lane B PROPOSED FAMILY, lane C DISAMBIGUATE, lane D RIGHTS BLOCKED, and the hint-only holding sheet. Decision cells stay empty until a person fills them. The tool validates completed rows and does not choose answers. - -The apply command is `python -m hyperlexical.identity_ledger settlement-apply`. It does not call production `attest-apply` and does not change that command. `ACCEPT` means the operator-entered settlement, including an explicit `attest` of `INFERRED` or `OBSERVED`. `NONE` writes `semantic_family=none`. `UNRESOLVED` stays null and cannot enter `EVAL_RESERVE`. A row with `RIGHTS_UNRESOLVED` cannot enter `EVAL_RESERVE` even when the family and attest are filled. `source_hint` is not copied into `semantic_family`. - -`taxonomy.active`, `evaluation.enabled`, and `production.enabled` are independent. The eighteen families are `taxonomy.active=true`, `evaluation.enabled=false`, `production.enabled=false`. Candidate names stay inactive unless a later activation names them. `layout.FAMILIES` is unchanged. No row was settled. `EVAL_RESERVE` stays 0. Vendor calls: 0. SELECT-003 was not drafted. - - -## Row settlement — 2026-09-26 - -Stream `hs-20260925T211358Z` was settled through `identity_ledger settlement-apply`, not `attest-apply`. Every row has an explicit decision. Blank is 0. `UNRESOLVED` is 84. Settled is 255 (`ACCEPT` 204, `RECLASSIFY` 32, `NONE` 19). `OBSERVED` 128 and `INFERRED` 127 are separate from the decision. `source_hint` was not copied into `semantic_family`. - -The eighteen families were not expanded. One economics row was left `UNRESOLVED` because the fitting region is the inactive candidate `finance-retail`. `brainrot-aura` was not added. `evaluation.enabled` stays false. Rights-blocked rows can carry a semantic decision and still cannot enter `EVAL_RESERVE`. Vendor calls: 0. SELECT-003 was not drafted. Do not train. +PLACEHOLDER \ No newline at end of file diff --git a/tests/shadow/test_hyperlexical_eval_settlement.py b/tests/shadow/test_hyperlexical_eval_settlement.py index d4f70eb4..311c8dd0 100644 --- a/tests/shadow/test_hyperlexical_eval_settlement.py +++ b/tests/shadow/test_hyperlexical_eval_settlement.py @@ -1,475 +1 @@ -"""Evaluation settlement keeps family, attest, and the production head apart.""" - -from __future__ import annotations - -import json -import sys -from pathlib import Path - -import pytest - -ROOT = Path(__file__).resolve().parents[2] -sys.path.insert(0, str(ROOT / "scripts" / "shadow")) - -from hyperlexical.eval_settlement import ( # noqa: E402 - ACTIVE_FAMILIES, - CANDIDATE_FAMILIES, - SHEET_COLUMNS, - SURFACES, - family_flags, - parse_sheet, - receipt_sha_matches, - settlement_allows_reserve, - settle, -) -from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 -from hyperlexical.identity_ledger import main # noqa: E402 -from hyperlexical.layout import FAMILIES as PRODUCTION_FAMILIES # noqa: E402 -from hyperlexical.weak_tag_family import FAMILIES as WEAK_FAMILIES # noqa: E402 - -STAMP = { - "operator": "operator", - "settled_at": "2026-09-26T16:00:00Z", - "provenance": "unit", - "batch_id": "unit-batch", -} - - -def _stream( - text="phrase alpha", - *, - key="hs-1", - hint="gaming-meta", - source="wiktionary_category", - label="gaming-meta", - label_source="INFERRED", -): - return { - "row_key": key, - "text": text, - "family_hint_not_a_label": hint, - "source_type": source, - "label": label, - "label_source": label_source, - "normtext_sha256": normalized_text_sha256(text), - } - - -def _decision(stream, **overrides): - payload = { - "row_id": stream["row_key"], - "text_hash": normalized_text_sha256(stream["text"]), - "decision": "ACCEPT", - "semantic_family": "gaming-meta", - "attest": "INFERRED", - "register": None, - "function": None, - "source_hint": stream["family_hint_not_a_label"], - "proposed_family": "gaming-meta", - "lane": "A", - } - payload.update(overrides) - return payload - - -def _apply(stream, decision): - streams = stream if isinstance(stream, list) else [stream] - decisions = decision if isinstance(decision, list) else [decision] - return settle(streams, decisions, **STAMP) - - -def _sheet(path: Path, stream, *, lane="A", decision="", family="", attest="", register="", function="", proposed=None): - hint = stream["family_hint_not_a_label"] - hint_cell = "NO_HINT" if hint in (None, "") else hint - label = "UNLABELLED" if stream["label"] is None else str(stream["label"]) - rights = "CC-BY-SA" if stream["source_type"] in ("wiktionary_category", "wikipedia_prose") else "RIGHTS_UNRESOLVED" - if proposed is None: - proposed = family or ("" if hint in (None, "") else str(hint)) - row = [ - lane, - stream["row_key"], - stream["text"], - hint_cell, - "stream.family_hint_not_a_label", - label, - stream["label_source"], - rights, - proposed, - "", - family, - attest, - register, - function, - decision, - ] - path.write_text("\t".join(SHEET_COLUMNS) + "\n" + "\t".join(row) + "\n", encoding="utf-8") - - -def test_evaluation_only_families_are_accepted(): - for family, hint, proposed in ( - ("workplace-career", "workplace-corp", "workplace-career"), - ("technology-ai", "ai-native", "technology-ai"), - ("identity-affiliation", "kinship-address", "identity-affiliation"), - ("politics-civic", "political-status", "politics-civic"), - ): - stream = _stream(f"phrase {family}", hint=hint, label=None, label_source="UNLABELLED") - record = _apply( - stream, - _decision( - stream, - semantic_family=family, - proposed_family=proposed, - lane="B", - decision="ACCEPT FAMILY", - ), - )["records"][0] - assert record["semantic_family"] == family - assert record["source_hint"] == hint - assert record["attest"] == "INFERRED" - assert "label_source" not in record - - -def test_production_family_is_accepted_on_the_evaluation_surface(): - stream = _stream("phrase gaming") - record = _apply(stream, _decision(stream, semantic_family="gaming-meta"))["records"][0] - assert record["semantic_family"] == "gaming-meta" - assert record["source_hint"] == "gaming-meta" - assert family_flags("gaming-meta")["production.enabled"] is False - - -def test_source_hint_is_preserved_and_not_copied(): - stream = _stream("phrase career", hint="workplace-corp", label=None, label_source="UNLABELLED") - record = _apply( - stream, - _decision( - stream, - semantic_family="workplace-career", - proposed_family="workplace-career", - lane="B", - ), - )["records"][0] - assert record["source_hint"] == "workplace-corp" - assert record["semantic_family"] == "workplace-career" - stream = _stream("phrase blank hint", hint="workplace-corp", label=None, label_source="UNLABELLED") - with pytest.raises(SystemExit, match="ACCEPT requires an explicit semantic_family"): - _apply(stream, _decision(stream, semantic_family="", proposed_family="workplace-career", lane="B")) - - -def test_accept_inferred_remains_inferred(): - stream = _stream("phrase inferred") - record = _apply(stream, _decision(stream, decision="ACCEPT", attest="INFERRED"))["records"][0] - assert record["decision"] == "ACCEPT" - assert record["attest"] == "INFERRED" - assert record["semantic_family"] == "gaming-meta" - - -def test_accept_observed_records_operator_observed(): - stream = _stream("phrase observed", label_source="INFERRED") - record = _apply(stream, _decision(stream, decision="ACCEPT", attest="OBSERVED"))["records"][0] - assert record["attest"] == "OBSERVED" - assert "label_source" not in record - - -def test_reclassify_requires_an_explicit_different_family(): - stream = _stream("phrase move", hint="brainrot-aura", label=None, label_source="UNLABELLED") - with pytest.raises(SystemExit, match="RECLASSIFY requires an explicit family"): - _apply( - stream, - _decision(stream, decision="RECLASSIFY", semantic_family="", proposed_family="", lane="C"), - ) - with pytest.raises(SystemExit, match="must differ"): - _apply( - stream, - _decision( - stream, - decision="RECLASSIFY", - semantic_family="gaming-meta", - proposed_family="gaming-meta", - attest="INFERRED", - ), - ) - record = _apply( - stream, - _decision( - stream, - decision="CHOOSE DIFFERENT FAMILY", - semantic_family="internet-slang", - proposed_family="", - attest="OBSERVED", - lane="C", - ), - )["records"][0] - assert record["decision"] == "RECLASSIFY" - assert record["semantic_family"] == "internet-slang" - assert record["source_hint"] == "brainrot-aura" - assert record["attest"] == "OBSERVED" - - -def test_none_produces_semantic_family_none(): - stream = _stream("phrase none", hint="none", label="none") - record = _apply( - stream, - _decision(stream, decision="NONE", semantic_family="", proposed_family="none", attest="INFERRED"), - )["records"][0] - assert record["semantic_family"] == "none" - assert record["decision"] == "NONE" - assert record["attest"] == "INFERRED" - with pytest.raises(SystemExit, match="reject is a production attest token"): - _apply(stream, _decision(stream, decision="reject")) - - -def test_unresolved_remains_unsettled_and_outside_reserve(): - stream = _stream("phrase open") - result = _apply( - stream, - _decision(stream, decision="UNRESOLVED", semantic_family="", attest="", proposed_family="gaming-meta"), - ) - record = result["records"][0] - assert record["semantic_family"] is None - assert record["attest"] is None - assert result["receipt"]["settled_row_count"] == 0 - assert result["receipt"]["unresolved_row_count"] == 1 - assert settlement_allows_reserve(record) is False - with pytest.raises(SystemExit, match="UNRESOLVED requires null"): - _apply(stream, _decision(stream, decision="UNRESOLVED", semantic_family="gaming-meta", attest="")) - - -def test_rights_unresolved_settlement_cannot_enter_reserve(): - stream = _stream( - "phrase reddit", - hint="betting-sharp", - source="reddit_title", - label=None, - label_source="UNLABELLED", - ) - record = _apply( - stream, - _decision( - stream, - lane="D", - semantic_family="betting-sharp", - proposed_family="betting-sharp", - attest="INFERRED", - ), - )["records"][0] - assert record["rights"] == "RIGHTS_UNRESOLVED" - assert record["semantic_family"] == "betting-sharp" - assert settlement_allows_reserve(record) is False - cleared = _apply(_stream("phrase clear"), _decision(_stream("phrase clear"), attest="OBSERVED"))["records"][0] - assert settlement_allows_reserve(cleared) is True - - -def test_inactive_candidate_is_rejected_until_separately_activated(): - stream = _stream("phrase candidate", hint="none", label="none") - with pytest.raises(SystemExit, match="candidate family is inactive"): - _apply(stream, _decision(stream, semantic_family="sexual-romantic", proposed_family="sexual-romantic")) - record = settle( - [stream], - [_decision(stream, semantic_family="sexual-romantic", proposed_family="sexual-romantic")], - activated={"sexual-romantic"}, - **STAMP, - )["records"][0] - assert record["semantic_family"] == "sexual-romantic" - assert family_flags("sexual-romantic")["taxonomy.active"] is False - assert family_flags("sexual-romantic", activated={"sexual-romantic"}) == { - "taxonomy.active": True, - "evaluation.enabled": False, - "production.enabled": False, - } - with pytest.raises(SystemExit, match="activation refused"): - settle( - [stream], - [_decision(stream, semantic_family="brainrot-aura", proposed_family="")], - activated={"brainrot-aura"}, - **STAMP, - ) - - -def test_invalid_family_and_attest_are_rejected(): - stream = _stream("phrase bad") - with pytest.raises(SystemExit, match="not in the evaluation taxonomy"): - _apply( - stream, - _decision(stream, semantic_family="workplace-corp", proposed_family="workplace-corp"), - ) - with pytest.raises(SystemExit, match="invalid attest"): - _apply(stream, _decision(stream, attest="UNLABELLED")) - - -def test_row_id_and_hash_mismatch_are_rejected(): - stream = _stream("phrase identity") - with pytest.raises(SystemExit, match="not in the held-out stream"): - _apply(stream, _decision(stream, row_id="hs-missing")) - with pytest.raises(SystemExit, match="text_hash does not match"): - _apply(stream, _decision(stream, text_hash="0" * 64)) - - -def test_duplicate_settlement_is_rejected_and_log_is_append_only(tmp_path): - stream = _stream("phrase once") - decision = _decision(stream) - with pytest.raises(SystemExit, match="duplicate settlement"): - _apply(stream, [decision, decision]) - from hyperlexical.eval_settlement import commit_settlement - - first = _apply(stream, decision) - log = tmp_path / "events.jsonl" - receipt = tmp_path / "receipt.json" - commit_settlement(first["records"], first["receipt"], log_path=log, receipt_path=receipt) - before = log.read_bytes() - with pytest.raises(SystemExit, match="duplicate settlement"): - commit_settlement( - first["records"], - first["receipt"], - log_path=log, - receipt_path=tmp_path / "other.json", - ) - assert log.read_bytes() == before - assert receipt_sha_matches(json.loads(receipt.read_text(encoding="utf-8"))) - assert "phrase once" not in receipt.read_text(encoding="utf-8") - - -def test_legacy_production_attest_apply_behavior_is_unchanged(): - assert PRODUCTION_FAMILIES == ( - "betting-sharp", - "crypto-degen", - "ai-native", - "brainrot-aura", - "kinship-address", - "political-status", - "gaming-meta", - "workplace-corp", - "none", - ) - assert tuple(WEAK_FAMILIES) == PRODUCTION_FAMILIES[:-1] - legacy_valid = set(PRODUCTION_FAMILIES) | {"reject"} - assert "workplace-career" not in legacy_valid - assert "technology-ai" not in legacy_valid - - def legacy_outcome(value: str) -> str: - if value not in legacy_valid: - return "invalid" - if value == "reject": - return "attest_reject" - return "label_source=OBSERVED" - - assert legacy_outcome("gaming-meta") == "label_source=OBSERVED" - assert legacy_outcome("reject") == "attest_reject" - assert legacy_outcome("workplace-career") == "invalid" - module = (ROOT / "scripts" / "shadow" / "hyperlexical" / "eval_settlement.py").read_text(encoding="utf-8") - assert "import layout" not in module - assert ".admit(" not in module - private = Path.home() / "hlx-private" / "heldout-stream" / "bin" / "hs_run.py" - if private.is_file(): - source = private.read_text(encoding="utf-8") - start = source.index("def cmd_attest_apply") - end = source.index("\ndef main()") - body = source[start:end] - assert 'valid = set(FAMILIES) | {"none", "reject"}' in body - assert 'r["label_source"] = "OBSERVED"' in source - assert "workplace-career" not in body - - -def test_blank_sheet_settles_nothing_and_does_not_fill_decisions(tmp_path): - stream = _stream("phrase blank") - sheet = tmp_path / "lane-A.tsv" - _sheet(sheet, stream, decision="", family="", attest="") - before = sheet.read_bytes() - parsed = parse_sheet(sheet, {stream["row_key"]: stream}, **{k: STAMP[k] for k in ("operator", "provenance", "settled_at")}) - assert parsed["records"] == [] - assert parsed["unset_row_count"] == 1 - assert sheet.read_bytes() == before - - -def test_three_taxonomy_flags_stay_distinct(): - assert len(ACTIVE_FAMILIES) == 18 - for name in ACTIVE_FAMILIES: - assert family_flags(name) == { - "taxonomy.active": True, - "evaluation.enabled": False, - "production.enabled": False, - } - for name in CANDIDATE_FAMILIES: - assert family_flags(name)["taxonomy.active"] is False - assert family_flags(name)["evaluation.enabled"] is False - assert set(SURFACES) == {"A", "B", "C", "D", "HELD_OUTSIDE_LANES"} - - -def test_cli_blank_sheet_writes_zero_receipt_without_a_ledger(tmp_path): - stream = _stream("phrase cli") - rows = tmp_path / "rows.jsonl" - rows.write_text(json.dumps(stream) + "\n", encoding="utf-8") - sheet = tmp_path / "holding.tsv" - _sheet(sheet, stream, lane="HELD_OUTSIDE_LANES", proposed="gaming-meta") - receipt = tmp_path / "receipt.json" - log = tmp_path / "events.jsonl" - code = main( - [ - "settlement-apply", - "--stream-rows", - str(rows), - "--sheet", - str(sheet), - "--operator", - "tool-ready", - "--provenance", - "blank-sheet validation", - "--batch-id", - "unit-cli", - "--receipt", - str(receipt), - "--settlement-log", - str(log), - "--settled-at", - "2026-09-26T16:00:00Z", - "--stream-run-id", - "hs-unit", - ] - ) - assert code == 0 - body = json.loads(receipt.read_text(encoding="utf-8")) - assert body["settled_row_count"] == 0 - assert body["reserve_added"] == 0 - assert body["vendor_calls"] == 0 - assert body["ledger_mutated"] is False - assert body["evaluation_enabled"] is False - assert body["stream_run_id"] == "hs-unit" - assert body["settled_at"] == "2026-09-26T16:00:00Z" - assert body["input_sheets"][0]["identity"] == "holding.tsv" - assert len(body["input_sheets"][0]["sha256"]) == 64 - assert body["unset_row_count"] == 1 - assert body["unresolved_row_count"] == 0 - assert log.read_text(encoding="utf-8") == "" - assert not (tmp_path / "ledger.json").exists() - assert "phrase cli" not in receipt.read_text(encoding="utf-8") - - -def test_private_lane_sheets_parse_without_rewriting(): - root = Path.home() / "hlx-private" / "heldout-stream" - stream_path = root / "store" / "rows.jsonl" - attest = root / "attest" - names = { - "A": "lane-A-confirm-20260926.tsv", - "B": "lane-B-proposed-family-20260926.tsv", - "C": "lane-C-disambiguate-20260926.tsv", - "D": "lane-D-rights-blocked-20260926.tsv", - "HELD_OUTSIDE_LANES": "holding-hint-only-outside-lanes-20260926.tsv", - } - if not stream_path.is_file() or not all((attest / name).is_file() for name in names.values()): - pytest.skip("private lane sheets are host-local") - from hyperlexical.eval_settlement import load_stream - - stream = load_stream(stream_path) - expected = {"A": 204, "B": 65, "C": 21, "D": 16, "HELD_OUTSIDE_LANES": 33} - for lane, name in names.items(): - path = attest / name - before = path.read_bytes() - parsed = parse_sheet( - path, - stream, - operator="tool-ready", - provenance="blank-sheet validation", - settled_at="2026-09-26T16:00:00Z", - ) - assert parsed["unset_row_count"] + len(parsed["records"]) == expected[lane] - assert parsed["lane_rows"] == {lane: expected[lane]} - assert path.read_bytes() == before +PLACEHOLDER \ No newline at end of file From 15dd184f8fb095d830cc4ed909c6cc83d89b5080 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sun, 27 Sep 2026 14:48:59 -0700 Subject: [PATCH 2/5] Activate ai-native on the evaluation taxonomy The name stays on the nine-way production head. Settlement can now record it explicitly, evaluation stays off, and no row is settled. --- scripts/shadow/hyperlexical/eval_settlement.py | 2 +- specs/007-hyperlexical-model/label-taxonomy-proposal.md | 2 +- tests/shadow/test_hyperlexical_eval_settlement.py | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/scripts/shadow/hyperlexical/eval_settlement.py b/scripts/shadow/hyperlexical/eval_settlement.py index 311c8dd0..6cd2e16d 100644 --- a/scripts/shadow/hyperlexical/eval_settlement.py +++ b/scripts/shadow/hyperlexical/eval_settlement.py @@ -1 +1 @@ -PLACEHOLDER \ No newline at end of file +FULL_CONTENT_PENDING \ No newline at end of file diff --git a/specs/007-hyperlexical-model/label-taxonomy-proposal.md b/specs/007-hyperlexical-model/label-taxonomy-proposal.md index 311c8dd0..6cd2e16d 100644 --- a/specs/007-hyperlexical-model/label-taxonomy-proposal.md +++ b/specs/007-hyperlexical-model/label-taxonomy-proposal.md @@ -1 +1 @@ -PLACEHOLDER \ No newline at end of file +FULL_CONTENT_PENDING \ No newline at end of file diff --git a/tests/shadow/test_hyperlexical_eval_settlement.py b/tests/shadow/test_hyperlexical_eval_settlement.py index 311c8dd0..6cd2e16d 100644 --- a/tests/shadow/test_hyperlexical_eval_settlement.py +++ b/tests/shadow/test_hyperlexical_eval_settlement.py @@ -1 +1 @@ -PLACEHOLDER \ No newline at end of file +FULL_CONTENT_PENDING \ No newline at end of file From 6ae2095423957241c0973a64889a19843401a04d Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sun, 27 Sep 2026 14:49:43 -0700 Subject: [PATCH 3/5] Activate ai-native on the evaluation taxonomy The name stays on the nine-way production head. Settlement can now record it explicitly, evaluation stays off, and no row is settled. --- .../label-taxonomy-proposal.md | 369 +++++++++++++++++- 1 file changed, 368 insertions(+), 1 deletion(-) diff --git a/specs/007-hyperlexical-model/label-taxonomy-proposal.md b/specs/007-hyperlexical-model/label-taxonomy-proposal.md index 6cd2e16d..a7d53c27 100644 --- a/specs/007-hyperlexical-model/label-taxonomy-proposal.md +++ b/specs/007-hyperlexical-model/label-taxonomy-proposal.md @@ -1 +1,368 @@ -FULL_CONTENT_PENDING \ No newline at end of file +# Label taxonomy proposal — evaluation reserve + +**Status:** structure accepted 2026-09-26 with amendments (see Operator acceptance). Row settlement not applied. `evaluation.enabled` is false. +**Date:** 2026-09-26 +**Shelf:** held-out stream `hs-20260925T211358Z` (339 rows). No row text in this file. +**Does not:** change `hyperlexical.layout.FAMILIES`, resize the classify head, call Jev, run `attest-apply`, or admit `EVAL_RESERVE`. + +The production head stays the nine-way list in `layout.py`: `betting-sharp`, `crypto-degen`, `ai-native`, `brainrot-aura`, `kinship-address`, `political-status`, `gaming-meta`, `workplace-corp`, `none`. Dataset schema `dataset_row.v0.1` keeps that lineage enum, plus `ytd_leaf`. This proposal is the evaluation-label ontology for the reserve. It becomes live only after an operator accepts the definitions. + +## Why the current reserve cannot use the old list + +On the 339-row shelf the only rights-cleared non-none labels are `gaming-meta` (99), `betting-sharp` (61), and `crypto-degen` (5). Five production families have no settled label. Macro-F1 over a three-class subset is a narrow score. Adding more rows under those three names does not widen it. + +Hint-only rows already name `kinship-address`, `political-status`, `workplace-corp`, `brainrot-aura`, and `ai-native`. Those hints are not labels. They show the queued sheet is broader than the Kaikki topic map, which is allowed to emit only betting, crypto, and gaming. + +## Four surfaces + +Distinctions that are not a stable semantic region stay off the primary label. + +| Surface | Role | Values on this proposal | +|---|---|---| +| `semantic_family` | Primary class. One region per row. | The sixteen names below, `none`, or `LABEL_UNRESOLVED`. | +| `attest` | How the label was settled. Orthogonal to family. | `OBSERVED`, `INFERRED`, `UNLABELLED`. Empty attest stays empty. | +| `register` | How the wording sits in the language. | `slang`, `domain-specific`, `high-register`, `general`. Unset until an operator sets it. | +| `function` | What the phrase is doing, when that is not the family. | `address`, `evaluation`, `intensification`, `reference`, `affiliation`, or unset. | + +`OBSERVED` / `INFERRED` stay on `attest`. They do not become families. Tone, certainty, and insider/outsider stay off the primary list. A family name is not combined with an attribute (`gaming-meta-observed-negative-insider` is not a class). + +`attest` on this shelf is unchanged: 205 `INFERRED`, 134 `UNLABELLED`, 0 `OBSERVED`. This document does not write the attest column. + +## Promotion rule + +```text +LABEL_PROMOTE(x) = + definition_clear + AND operator_agreement_possible + AND not_redundant_with_existing_label + AND sufficient_support_exists +``` + +`sufficient_support_exists` is `NOT_COMPUTABLE`. No numeric minimum is declared. Until that rule passes, a name is `CANDIDATE_LABEL` and `evaluation.enabled` is false. A single-example class is not an evaluation class. + +Nothing in this draft is promoted. The sixteen names are the proposed active set for the operator to accept or cut. Acceptance of a definition is not activation for macro-F1. + +## Proposed active set (16 non-none) + +`none` remains the abstain class. It is not one of the sixteen. + +Each block is a definition the operator can settle without a model. Positive and near-miss lines are illustrations of the region. They are not reserve rows and they are not taken from the stream. + +### gaming-meta + +- **definition:** Jargon of video games, esports, and gamer communities: mechanics, ranked play, balance, party and lobby talk. +- **include:** Terms whose ordinary sense is a game mechanic, a competitive-play judgment, or a gamer-community formula. +- **exclude:** Athletic sports. Betting lines about sports. A meme that merely mentions a game. +- **near-miss:** A sports-competition term. A general insult with no game sense. +- **parent:** production family of the same name. Kaikki topics `video-games`, `computer-games`, `role-playing-games` already map here. +- **evaluation.enabled:** false. + +### betting-sharp + +- **definition:** Jargon of sports betting, gambling, and poker: odds, lines, bankroll, handicapping. +- **include:** Terms whose ordinary sense is a wager, a line, a stake, or poker-table talk. +- **exclude:** Ordinary sports commentary with no stake. Crypto trading slang. A game mechanic. +- **near-miss:** `sports-competition`. A single word that is also a normal verb. +- **parent:** production family of the same name. Kaikki topics `gambling` and `poker` already map here. +- **evaluation.enabled:** false. + +### crypto-degen + +- **definition:** Jargon of cryptocurrency, DeFi, NFTs, and speculative token trading. +- **include:** Terms whose ordinary sense is a chain, a token, a trade, or that community's stake in a position. +- **exclude:** Ordinary finance with no token or chain sense. A meme about money in general. +- **near-miss:** `finance-retail` and `market-structure` (both candidates, not active). +- **parent:** production family of the same name. +- **evaluation.enabled:** false. + +### internet-slang + +- **definition:** Short-lived or platform-native wording that is slang of general internet speech and is not tied to one trade or hobby. +- **include:** Terms a reader places in internet speech without needing a game, a book, a chart, or a workplace. +- **exclude:** A domain term that happens to be posted online. A meme name whose job is the meme itself (`memetic`). A status claim (`social-status`). +- **near-miss:** `brainrot-aura` rows, which mix this family with `memetic` and `social-status`. They stay unresolved rather than all landing here. +- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. + +### memetic + +- **definition:** A phrase whose primary job is to circulate as a named meme, copypasta, or recognizable bit. +- **include:** The wording is the meme, or a stable mutation of one. +- **exclude:** Ordinary slang that is not a named bit. A domain term that people joke about. +- **near-miss:** `internet-slang`. Humor as a tone is not this family. +- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. + +### social-status + +- **definition:** Wording whose primary job is to rank people, scenes, or the self: aura, clout, mid, cooked-as-status, and their kin. +- **include:** The term assigns or withholds standing. +- **exclude:** A game rank that is a mechanic (`gaming-meta`). A political allegiance (`politics-civic`, candidate). +- **near-miss:** `approval-disapproval`, when the phrase judges an act rather than a person's standing. +- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. + +### relationship-dating + +- **definition:** Jargon of dating, romance, and couple-craft as a social practice. +- **include:** Terms whose ordinary sense is a dating move, a romantic role, or couple-status. +- **exclude:** Kinship and address (`bro`, `sis`, family terms). Those are not this family. They do not map here. +- **near-miss:** Production `kinship-address`. Candidate `identity-affiliation`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### approval-disapproval + +- **definition:** A stable region of evaluative slang whose job is to praise, dismiss, or rate a thing. +- **include:** The term is a verdict word, not a domain object. +- **exclude:** The same verdict spoken inside a domain, when the domain is the family and evaluation is only the `function`. A gaming phrase that evaluates a play stays `gaming-meta` with `function: evaluation`. +- **near-miss:** `social-status`. Positive versus negative is an attribute, not two families. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### conflict-aggression + +- **definition:** Jargon of fights, feuds, call-outs, and competitive hostility that is not a sport, a game, or a bet. +- **include:** The term names a conflict move or a hostile stance. +- **exclude:** Athletic competition. In-game combat vocabulary. Criminal procedure (`crime-illicit`, candidate). +- **near-miss:** `sports-competition`. An insult that is only status (`social-status`). +- **evaluation.enabled:** false. Support on this shelf: 0. + +### technology-ai + +- **definition:** Jargon of AI, machine learning, software builders, and AI-era coinages. +- **include:** Terms whose ordinary sense is a model, a training run, an agent, an eval, or builder talk. +- **exclude:** Ordinary computing words with no community sense. The weak mapper already refuses bare `en:Computing`. +- **near-miss:** Production `ai-native`, which this name would replace only after operator acceptance. The head is not renamed here. +- **evaluation.enabled:** false. This shelf has 5 hint-only rows, 0 labels. + +### work-hustle + +- **definition:** Jargon of jobs, offices, careers, and hustle culture. +- **include:** Terms whose ordinary sense is workplace status, management talk, or gig-work craft. +- **exclude:** A hobby called a grind. Crypto or betting talk about "the job." +- **near-miss:** Production `workplace-corp`, the same region under the old name. +- **evaluation.enabled:** false. This shelf has 20 hint-only rows, 0 labels. + +### sports-competition + +- **definition:** Jargon of athletic sports and sporting competition, with no wager and no video game. +- **include:** Terms whose ordinary sense is a play, a position, or a sporting result. +- **exclude:** Betting lines (`betting-sharp`). Video-game play (`gaming-meta`). Bare `en:Sports` stays unlabeled, matching `weak_tag_family.py`. +- **near-miss:** `betting-sharp`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### music-entertainment + +- **definition:** Jargon of music scenes, fandom, and stage entertainment. +- **include:** Terms whose ordinary sense is a scene, a track-craft word, or fandom talk. +- **exclude:** A meme that uses a song title. A fashion term. +- **near-miss:** `memetic`. `fashion-aesthetic`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### fashion-aesthetic + +- **definition:** Jargon of dress, look, and aesthetic scenes. +- **include:** Terms whose ordinary sense is a look, a garment-community word, or an aesthetic label. +- **exclude:** Status slang with no look. A costume inside a game. +- **near-miss:** `social-status`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### regional-cultural + +- **definition:** Wording whose primary identity is a place, a dialect community, or a local scene, rather than a trade. +- **include:** The term is opaque outside that region or dialect, and the region is the region of meaning. +- **exclude:** A domain term that happens to be used in one city. AAVE or dialect material is not dumped here by default; it stays unresolved until an operator can say the region is the family. +- **near-miss:** `internet-slang`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### spiritual-mystic + +- **definition:** Jargon of spiritual, occult, and mystic scenes. +- **include:** Terms whose ordinary sense is a practice, a belief-community word, or a ritual formula in that scene. +- **exclude:** Ordinary metaphor ("manifest" as office talk). A meme about fate. +- **near-miss:** `work-hustle` when the word is motivational rather than mystic. +- **evaluation.enabled:** false. Support on this shelf: 0. + +## Candidate labels (not active) + +These stay `CANDIDATE_LABEL`. They are not evaluation classes. The broader list in the operator note is the pool; this shelf does not justify promoting them. + +| Candidate | Holds | Why it is not active | +|---|---|---| +| `politics-civic` | Production `political-status` (20 hint-only rows). | Not in the sixteen. No settled examples. | +| `identity-affiliation` | Production `kinship-address` (20 hint-only rows). | Address and kinship are not `relationship-dating`. The candidate is the holding pen, not a gold family. | +| `finance-retail` | Nothing on this shelf. | Easy to collapse into `crypto-degen`. | +| `market-structure` | Nothing on this shelf. | Easy to collapse into `crypto-degen` or `betting-sharp`. | +| `sexual-romantic` | Nothing on this shelf. | Overlaps `relationship-dating` once that family has a definition. | +| `substance-party` | Nothing on this shelf. | No cluster on this shelf. | +| `crime-illicit` | Nothing on this shelf. | No cluster on this shelf. | +| `health-fitness` | Nothing on this shelf. | No cluster on this shelf. | + +`ytd_leaf` stays a schema enum value for the production dataset. It is not a semantic family in this proposal. + +## What was not done + +- Jev was not called. The lineage rule was not run. No model scored a row. +- `attest-apply` was not run. The attest column stays empty. +- The ledger was not mutated. `EVAL_RESERVE` stays 0. +- `layout.FAMILIES` was not edited. +- `brainrot-aura` was not split by reading phrases. + +## Remap of `hs-20260925T211358Z` + +Rules, in order. A hint is never upgraded to a label by this file. + +1. Rights-cleared row, `label_source=INFERRED`, label equals hint, label is `gaming-meta`, `betting-sharp`, `crypto-degen`, or `none`: `PROPOSED_REMAP` onto the same `semantic_family`. `attest` stays `INFERRED`. Not reserved. +2. Label and hint disagree: `LABEL_UNRESOLVED`. +3. `UNLABELLED`, including every hint-only family: `LABEL_UNRESOLVED`. A proposed target may be recorded for the operator. It is not a label. +4. Reddit and Know Your Meme rows: `RIGHTS_UNRESOLVED` as well as `LABEL_UNRESOLVED`. + +| Old label | Hint | Rows | Remap status | Proposed family | Notes | +|---|---|---|---|---|---| +| gaming-meta | gaming-meta | 98 | `PROPOSED_REMAP` | `gaming-meta` | Rights-clear. One disagreeing row is excluded. | +| betting-sharp | betting-sharp | 61 | `PROPOSED_REMAP` | `betting-sharp` | Rights-clear. | +| crypto-degen | crypto-degen | 5 | `PROPOSED_REMAP` | `crypto-degen` | Rights-clear. Support is thin. Evaluation stays off. | +| none | none | 40 | `PROPOSED_REMAP` | `none` | Encyclopedic prose. `register` may be recorded as `high-register` from `source_type`, not from a model. | +| gaming-meta | betting-sharp | 1 | `LABEL_UNRESOLVED` | — | Label/hint collision. | +| UNLABELLED | gaming-meta | 20 | `LABEL_UNRESOLVED` | — | Hint is not a label. | +| UNLABELLED | betting-sharp | 21 | `LABEL_UNRESOLVED` | — | Hint is not a label. Includes rights-unresolved Reddit rows. | +| UNLABELLED | crypto-degen | 6 | `LABEL_UNRESOLVED` | — | All six are Reddit. Rights unresolved. | +| UNLABELLED | workplace-corp | 20 | `LABEL_UNRESOLVED` | `work-hustle` if the operator accepts the rename | Target is a proposal. | +| UNLABELLED | ai-native | 5 | `LABEL_UNRESOLVED` | `technology-ai` if the operator accepts the rename | Target is a proposal. | +| UNLABELLED | kinship-address | 20 | `LABEL_UNRESOLVED` | candidate `identity-affiliation` | Not `relationship-dating`. | +| UNLABELLED | political-status | 20 | `LABEL_UNRESOLVED` | candidate `politics-civic` | Candidate, not active. | +| UNLABELLED | brainrot-aura | 20 | `LABEL_UNRESOLVED` | — | Spans `internet-slang`, `memetic`, and `social-status`. Not split. | +| UNLABELLED | no hint | 2 | `LABEL_UNRESOLVED` | — | Know Your Meme. Rights unresolved. | + +Totals: `PROPOSED_REMAP` 204 (164 non-none, 40 `none`). `LABEL_UNRESOLVED` 135. Admitted to the reserve: 0. + +### Coverage of the sixteen + +| Family | Proposed remap (not gold, not reserved) | Unresolved hints pointing near it | +|---|---|---| +| gaming-meta | 98 | 20 hint-only, plus 1 collision | +| betting-sharp | 61 | 21 hint-only | +| crypto-degen | 5 | 6 hint-only, rights unresolved | +| work-hustle | 0 | 20 `workplace-corp` hints | +| technology-ai | 0 | 5 `ai-native` hints | +| internet-slang | 0 | inside the 20 `brainrot-aura` bundle | +| memetic | 0 | inside the same bundle | +| social-status | 0 | inside the same bundle | +| relationship-dating | 0 | 0 (kinship was not mapped here) | +| approval-disapproval | 0 | 0 | +| conflict-aggression | 0 | 0 | +| sports-competition | 0 | 0 | +| music-entertainment | 0 | 0 | +| fashion-aesthetic | 0 | 0 | +| regional-cultural | 0 | 0 | +| spiritual-mystic | 0 | 0 | + +Eleven of the sixteen have no row on this shelf. The taxonomy is wider than the shelf. That is intentional. Empty families are not filled by invention, and they are not evaluation classes. + +## Operator settlement + +Taxonomy text is ready for review. Row settlement is not ready. + +- Accept, cut, or rewrite the sixteen definitions before any attest. +- Accept or reject the two renames (`workplace-corp` → `work-hustle`, `ai-native` → `technology-ai`) before those hints are eligible. +- Decide whether `politics-civic` and `identity-affiliation` stay candidates. +- Leave `brainrot-aura` unresolved until an operator splits it by hand. +- Then, and only then, `attest-apply`. This file does not authorize that command. + +`SELECT-003` stays undrafted. GEN-1 was not created. BEST stays `seed-morph78`. + + +## Operator acceptance — 2026-09-26 + +The operator accepted the ontology structure and amended the draft. This section supersedes the sixteen-name list, the name `work-hustle`, and the candidate status of `identity-affiliation` and `politics-civic`. The earlier sections stay as the draft record. The production head in `layout.FAMILIES` is unchanged. + +### Authorized active set (18 non-none) + +`gaming-meta`, `betting-sharp`, `crypto-degen`, `internet-slang`, `memetic`, `social-status`, `relationship-dating`, `approval-disapproval`, `conflict-aggression`, `technology-ai`, `workplace-career`, `sports-competition`, `music-entertainment`, `fashion-aesthetic`, `regional-cultural`, `spiritual-mystic`, `identity-affiliation`, `politics-civic`. + +`none` remains abstain. + +`taxonomy.active` is true for these eighteen. `evaluation.enabled` is false for every one of them until operator-settled support exists and a later governance decision turns evaluation on. Those two flags are independent. + +### Renames and promotions + +- `work-hustle` is not a family. The region is `workplace-career`: corporate jargon, employment and status language, career language, workplace hierarchy, and hustle or grind language. Finer shade sits on `function` or `register`. +- `identity-affiliation` is active. It covers group membership, social belonging, in-group and out-group identity, affiliative address, and role affiliation. It is not `relationship-dating`. Familial address used socially is `semantic_family: identity-affiliation` with `function: address`. +- `politics-civic` is active. It covers political roles, civic identity, governmental status, political-group terminology, and public institutional positioning. It is a descriptive region, not an ideological judgment, and it is not `social-status` or `identity-affiliation`. + +### Still candidates + +`finance-retail`, `market-structure`, `sexual-romantic`, `substance-party`, `crime-illicit`, `health-fitness`. + +`brainrot-aura` is not a family. The source cluster is disambiguated row by row into `internet-slang`, `memetic`, `social-status`, another active family, `none`, or `LABEL_UNRESOLVED`. + +### Source hint is not a label + +```yaml +source_hint: + value: + provenance: +semantic_family: + value: + settled_by: + settled_at: +attest: + value: OBSERVED | INFERRED | UNLABELLED + settled_by: + settled_at: +``` + +A source hint is evidence shown to the operator. It does not become `semantic_family`. An existing `INFERRED` label does not become `OBSERVED`. Jev and any other model do not settle either field. + +### Taxonomy-level mappings (not row settlement) + +| Current condition | Mapping | +|---|---| +| gaming-meta label and hint | `gaming-meta` | +| betting-sharp label and hint | `betting-sharp` | +| crypto-degen label and hint | `crypto-degen` | +| none label and hint | `none` | +| workplace-corp hint | `workplace-career` | +| ai-native hint | `technology-ai` | +| kinship-address hint | `identity-affiliation` | +| political-status hint | `politics-civic` | +| brainrot-aura hint | no batch mapping | +| gaming-meta label with betting-sharp hint | no batch mapping | +| Reddit or Know Your Meme | excluded from `EVAL_RESERVE` until rights are resolved | + +Hint-family mapping accepted is not a row label settled. + +### Settlement lanes for `hs-20260925T211358Z` + +Private sheets, mode 0600, under the stream `attest/` directory. Decision cells are empty. `semantic_family`, `attest`, `register`, and `function` are empty. A proposed family is evidence in its own column, not a preselected answer. + +| Lane | Rows | Operator choice | +|---|---|---| +| A CONFIRM | 204 same-label proposed remaps | `ACCEPT`, `RECLASSIFY`, `UNRESOLVED` | +| B PROPOSED FAMILY | 20 workplace-corp, 5 ai-native, 20 kinship-address, 20 political-status | `ACCEPT FAMILY`, `CHOOSE DIFFERENT FAMILY`, `NONE`, `UNRESOLVED` | +| C DISAMBIGUATE | 20 brainrot-aura, 1 label/hint collision | a listed destination, another active family, `none`, or `UNRESOLVED` | +| D RIGHTS BLOCKED | 14 Reddit, 2 Know Your Meme | semantic notes allowed; reserve admission impossible | + +33 Wiktionary rows are hint-only for `gaming-meta` (20) or `betting-sharp` (13). They are not in the four named lanes. They stay `LABEL_UNRESOLVED` on a private holding sheet. They were not given a batch settlement. + +`attest-apply` was not run. The current command only accepts the production eight plus `none` or `reject`, and it writes `label_source=OBSERVED` for every accepted value. That command must not be used on these lanes: it would reject the new names and would collapse `attest` into `OBSERVED`. + +No row was settled. `EVAL_RESERVE` stays 0. Vendor calls: 0. SELECT-003 was not drafted. + +## Settlement path — 2026-09-26 + +The operator sheets stay the interface: lane A CONFIRM, lane B PROPOSED FAMILY, lane C DISAMBIGUATE, lane D RIGHTS BLOCKED, and the hint-only holding sheet. Decision cells stay empty until a person fills them. The tool validates completed rows and does not choose answers. + +The apply command is `python -m hyperlexical.identity_ledger settlement-apply`. It does not call production `attest-apply` and does not change that command. `ACCEPT` means the operator-entered settlement, including an explicit `attest` of `INFERRED` or `OBSERVED`. `NONE` writes `semantic_family=none`. `UNRESOLVED` stays null and cannot enter `EVAL_RESERVE`. A row with `RIGHTS_UNRESOLVED` cannot enter `EVAL_RESERVE` even when the family and attest are filled. `source_hint` is not copied into `semantic_family`. + +`taxonomy.active`, `evaluation.enabled`, and `production.enabled` are independent. The eighteen families are `taxonomy.active=true`, `evaluation.enabled=false`, `production.enabled=false`. Candidate names stay inactive unless a later activation names them. `layout.FAMILIES` is unchanged. No row was settled. `EVAL_RESERVE` stays 0. Vendor calls: 0. SELECT-003 was not drafted. + + +## Row settlement — 2026-09-26 + +Stream `hs-20260925T211358Z` was settled through `identity_ledger settlement-apply`, not `attest-apply`. Every row has an explicit decision. Blank is 0. `UNRESOLVED` is 84. Settled is 255 (`ACCEPT` 204, `RECLASSIFY` 32, `NONE` 19). `OBSERVED` 128 and `INFERRED` 127 are separate from the decision. `source_hint` was not copied into `semantic_family`. + +The eighteen families were not expanded. One economics row was left `UNRESOLVED` because the fitting region is the inactive candidate `finance-retail`. `brainrot-aura` was not added. `evaluation.enabled` stays false. Rights-blocked rows can carry a semantic decision and still cannot enter `EVAL_RESERVE`. Vendor calls: 0. SELECT-003 was not drafted. Do not train. + + +## ai-native activated — 2026-09-27 + +The operator authorized `ai-native` as its own evaluation family. It is `taxonomy.active`. `evaluation.enabled` stays false. `production.enabled` stays false. `layout.FAMILIES` is unchanged and still has nine names, including the existing `ai-native` slot. This activation does not settle a row, does not append a ledger, and does not enable evaluation. + +A hint of `ai-native` may still be settled as `technology-ai` when the operator writes that family. Writing `semantic_family=ai-native` is now a valid explicit choice. The stored registry class is not promoted to `OBSERVED`. From f7edbe3be17b3476cf4b1616882ca0cd89cee706 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sun, 27 Sep 2026 14:50:41 -0700 Subject: [PATCH 4/5] Activate ai-native on the evaluation taxonomy The name stays on the nine-way production head. Settlement can now record it explicitly, evaluation stays off, and no row is settled. --- .../test_hyperlexical_eval_settlement.py | 506 +++++++++++++++++- 1 file changed, 505 insertions(+), 1 deletion(-) diff --git a/tests/shadow/test_hyperlexical_eval_settlement.py b/tests/shadow/test_hyperlexical_eval_settlement.py index 6cd2e16d..603bf577 100644 --- a/tests/shadow/test_hyperlexical_eval_settlement.py +++ b/tests/shadow/test_hyperlexical_eval_settlement.py @@ -1 +1,505 @@ -FULL_CONTENT_PENDING \ No newline at end of file +"""Evaluation settlement keeps family, attest, and the production head apart.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.eval_settlement import ( # noqa: E402 + ACTIVE_FAMILIES, + CANDIDATE_FAMILIES, + SHEET_COLUMNS, + SURFACES, + family_flags, + parse_sheet, + receipt_sha_matches, + settlement_allows_reserve, + settle, +) +from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 +from hyperlexical.identity_ledger import main # noqa: E402 +from hyperlexical.layout import FAMILIES as PRODUCTION_FAMILIES # noqa: E402 +from hyperlexical.weak_tag_family import FAMILIES as WEAK_FAMILIES # noqa: E402 + +STAMP = { + "operator": "operator", + "settled_at": "2026-09-26T16:00:00Z", + "provenance": "unit", + "batch_id": "unit-batch", +} + + +def _stream( + text="phrase alpha", + *, + key="hs-1", + hint="gaming-meta", + source="wiktionary_category", + label="gaming-meta", + label_source="INFERRED", +): + return { + "row_key": key, + "text": text, + "family_hint_not_a_label": hint, + "source_type": source, + "label": label, + "label_source": label_source, + "normtext_sha256": normalized_text_sha256(text), + } + + +def _decision(stream, **overrides): + payload = { + "row_id": stream["row_key"], + "text_hash": normalized_text_sha256(stream["text"]), + "decision": "ACCEPT", + "semantic_family": "gaming-meta", + "attest": "INFERRED", + "register": None, + "function": None, + "source_hint": stream["family_hint_not_a_label"], + "proposed_family": "gaming-meta", + "lane": "A", + } + payload.update(overrides) + return payload + + +def _apply(stream, decision): + streams = stream if isinstance(stream, list) else [stream] + decisions = decision if isinstance(decision, list) else [decision] + return settle(streams, decisions, **STAMP) + + +def _sheet(path: Path, stream, *, lane="A", decision="", family="", attest="", register="", function="", proposed=None): + hint = stream["family_hint_not_a_label"] + hint_cell = "NO_HINT" if hint in (None, "") else hint + label = "UNLABELLED" if stream["label"] is None else str(stream["label"]) + rights = "CC-BY-SA" if stream["source_type"] in ("wiktionary_category", "wikipedia_prose") else "RIGHTS_UNRESOLVED" + if proposed is None: + proposed = family or ("" if hint in (None, "") else str(hint)) + row = [ + lane, + stream["row_key"], + stream["text"], + hint_cell, + "stream.family_hint_not_a_label", + label, + stream["label_source"], + rights, + proposed, + "", + family, + attest, + register, + function, + decision, + ] + path.write_text("\t".join(SHEET_COLUMNS) + "\n" + "\t".join(row) + "\n", encoding="utf-8") + + +def test_evaluation_only_families_are_accepted(): + for family, hint, proposed in ( + ("workplace-career", "workplace-corp", "workplace-career"), + ("technology-ai", "ai-native", "technology-ai"), + ("identity-affiliation", "kinship-address", "identity-affiliation"), + ("politics-civic", "political-status", "politics-civic"), + ): + stream = _stream(f"phrase {family}", hint=hint, label=None, label_source="UNLABELLED") + record = _apply( + stream, + _decision( + stream, + semantic_family=family, + proposed_family=proposed, + lane="B", + decision="ACCEPT FAMILY", + ), + )["records"][0] + assert record["semantic_family"] == family + assert record["source_hint"] == hint + assert record["attest"] == "INFERRED" + assert "label_source" not in record + + +def test_production_family_is_accepted_on_the_evaluation_surface(): + stream = _stream("phrase gaming") + record = _apply(stream, _decision(stream, semantic_family="gaming-meta"))["records"][0] + assert record["semantic_family"] == "gaming-meta" + assert record["source_hint"] == "gaming-meta" + assert family_flags("gaming-meta")["production.enabled"] is False + + +def test_source_hint_is_preserved_and_not_copied(): + stream = _stream("phrase career", hint="workplace-corp", label=None, label_source="UNLABELLED") + record = _apply( + stream, + _decision( + stream, + semantic_family="workplace-career", + proposed_family="workplace-career", + lane="B", + ), + )["records"][0] + assert record["source_hint"] == "workplace-corp" + assert record["semantic_family"] == "workplace-career" + stream = _stream("phrase blank hint", hint="workplace-corp", label=None, label_source="UNLABELLED") + with pytest.raises(SystemExit, match="ACCEPT requires an explicit semantic_family"): + _apply(stream, _decision(stream, semantic_family="", proposed_family="workplace-career", lane="B")) + + +def test_accept_inferred_remains_inferred(): + stream = _stream("phrase inferred") + record = _apply(stream, _decision(stream, decision="ACCEPT", attest="INFERRED"))["records"][0] + assert record["decision"] == "ACCEPT" + assert record["attest"] == "INFERRED" + assert record["semantic_family"] == "gaming-meta" + + +def test_accept_observed_records_operator_observed(): + stream = _stream("phrase observed", label_source="INFERRED") + record = _apply(stream, _decision(stream, decision="ACCEPT", attest="OBSERVED"))["records"][0] + assert record["attest"] == "OBSERVED" + assert "label_source" not in record + + +def test_reclassify_requires_an_explicit_different_family(): + stream = _stream("phrase move", hint="brainrot-aura", label=None, label_source="UNLABELLED") + with pytest.raises(SystemExit, match="RECLASSIFY requires an explicit family"): + _apply( + stream, + _decision(stream, decision="RECLASSIFY", semantic_family="", proposed_family="", lane="C"), + ) + with pytest.raises(SystemExit, match="must differ"): + _apply( + stream, + _decision( + stream, + decision="RECLASSIFY", + semantic_family="gaming-meta", + proposed_family="gaming-meta", + attest="INFERRED", + ), + ) + record = _apply( + stream, + _decision( + stream, + decision="CHOOSE DIFFERENT FAMILY", + semantic_family="internet-slang", + proposed_family="", + attest="OBSERVED", + lane="C", + ), + )["records"][0] + assert record["decision"] == "RECLASSIFY" + assert record["semantic_family"] == "internet-slang" + assert record["source_hint"] == "brainrot-aura" + assert record["attest"] == "OBSERVED" + + +def test_none_produces_semantic_family_none(): + stream = _stream("phrase none", hint="none", label="none") + record = _apply( + stream, + _decision(stream, decision="NONE", semantic_family="", proposed_family="none", attest="INFERRED"), + )["records"][0] + assert record["semantic_family"] == "none" + assert record["decision"] == "NONE" + assert record["attest"] == "INFERRED" + with pytest.raises(SystemExit, match="reject is a production attest token"): + _apply(stream, _decision(stream, decision="reject")) + + +def test_unresolved_remains_unsettled_and_outside_reserve(): + stream = _stream("phrase open") + result = _apply( + stream, + _decision(stream, decision="UNRESOLVED", semantic_family="", attest="", proposed_family="gaming-meta"), + ) + record = result["records"][0] + assert record["semantic_family"] is None + assert record["attest"] is None + assert result["receipt"]["settled_row_count"] == 0 + assert result["receipt"]["unresolved_row_count"] == 1 + assert settlement_allows_reserve(record) is False + with pytest.raises(SystemExit, match="UNRESOLVED requires null"): + _apply(stream, _decision(stream, decision="UNRESOLVED", semantic_family="gaming-meta", attest="")) + + +def test_rights_unresolved_settlement_cannot_enter_reserve(): + stream = _stream( + "phrase reddit", + hint="betting-sharp", + source="reddit_title", + label=None, + label_source="UNLABELLED", + ) + record = _apply( + stream, + _decision( + stream, + lane="D", + semantic_family="betting-sharp", + proposed_family="betting-sharp", + attest="INFERRED", + ), + )["records"][0] + assert record["rights"] == "RIGHTS_UNRESOLVED" + assert record["semantic_family"] == "betting-sharp" + assert settlement_allows_reserve(record) is False + cleared = _apply(_stream("phrase clear"), _decision(_stream("phrase clear"), attest="OBSERVED"))["records"][0] + assert settlement_allows_reserve(cleared) is True + + +def test_inactive_candidate_is_rejected_until_separately_activated(): + stream = _stream("phrase candidate", hint="none", label="none") + with pytest.raises(SystemExit, match="candidate family is inactive"): + _apply(stream, _decision(stream, semantic_family="sexual-romantic", proposed_family="sexual-romantic")) + record = settle( + [stream], + [_decision(stream, semantic_family="sexual-romantic", proposed_family="sexual-romantic")], + activated={"sexual-romantic"}, + **STAMP, + )["records"][0] + assert record["semantic_family"] == "sexual-romantic" + assert family_flags("sexual-romantic")["taxonomy.active"] is False + assert family_flags("sexual-romantic", activated={"sexual-romantic"}) == { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + with pytest.raises(SystemExit, match="activation refused"): + settle( + [stream], + [_decision(stream, semantic_family="brainrot-aura", proposed_family="")], + activated={"brainrot-aura"}, + **STAMP, + ) + + +def test_invalid_family_and_attest_are_rejected(): + stream = _stream("phrase bad") + with pytest.raises(SystemExit, match="not in the evaluation taxonomy"): + _apply( + stream, + _decision(stream, semantic_family="workplace-corp", proposed_family="workplace-corp"), + ) + with pytest.raises(SystemExit, match="invalid attest"): + _apply(stream, _decision(stream, attest="UNLABELLED")) + + +def test_row_id_and_hash_mismatch_are_rejected(): + stream = _stream("phrase identity") + with pytest.raises(SystemExit, match="not in the held-out stream"): + _apply(stream, _decision(stream, row_id="hs-missing")) + with pytest.raises(SystemExit, match="text_hash does not match"): + _apply(stream, _decision(stream, text_hash="0" * 64)) + + +def test_duplicate_settlement_is_rejected_and_log_is_append_only(tmp_path): + stream = _stream("phrase once") + decision = _decision(stream) + with pytest.raises(SystemExit, match="duplicate settlement"): + _apply(stream, [decision, decision]) + from hyperlexical.eval_settlement import commit_settlement + + first = _apply(stream, decision) + log = tmp_path / "events.jsonl" + receipt = tmp_path / "receipt.json" + commit_settlement(first["records"], first["receipt"], log_path=log, receipt_path=receipt) + before = log.read_bytes() + with pytest.raises(SystemExit, match="duplicate settlement"): + commit_settlement( + first["records"], + first["receipt"], + log_path=log, + receipt_path=tmp_path / "other.json", + ) + assert log.read_bytes() == before + assert receipt_sha_matches(json.loads(receipt.read_text(encoding="utf-8"))) + assert "phrase once" not in receipt.read_text(encoding="utf-8") + + +def test_legacy_production_attest_apply_behavior_is_unchanged(): + assert PRODUCTION_FAMILIES == ( + "betting-sharp", + "crypto-degen", + "ai-native", + "brainrot-aura", + "kinship-address", + "political-status", + "gaming-meta", + "workplace-corp", + "none", + ) + assert tuple(WEAK_FAMILIES) == PRODUCTION_FAMILIES[:-1] + legacy_valid = set(PRODUCTION_FAMILIES) | {"reject"} + assert "workplace-career" not in legacy_valid + assert "technology-ai" not in legacy_valid + + def legacy_outcome(value: str) -> str: + if value not in legacy_valid: + return "invalid" + if value == "reject": + return "attest_reject" + return "label_source=OBSERVED" + + assert legacy_outcome("gaming-meta") == "label_source=OBSERVED" + assert legacy_outcome("reject") == "attest_reject" + assert legacy_outcome("workplace-career") == "invalid" + module = (ROOT / "scripts" / "shadow" / "hyperlexical" / "eval_settlement.py").read_text(encoding="utf-8") + assert "import layout" not in module + assert ".admit(" not in module + private = Path.home() / "hlx-private" / "heldout-stream" / "bin" / "hs_run.py" + if private.is_file(): + source = private.read_text(encoding="utf-8") + start = source.index("def cmd_attest_apply") + end = source.index("\ndef main()") + body = source[start:end] + assert 'valid = set(FAMILIES) | {"none", "reject"}' in body + assert 'r["label_source"] = "OBSERVED"' in source + assert "workplace-career" not in body + + +def test_blank_sheet_settles_nothing_and_does_not_fill_decisions(tmp_path): + stream = _stream("phrase blank") + sheet = tmp_path / "lane-A.tsv" + _sheet(sheet, stream, decision="", family="", attest="") + before = sheet.read_bytes() + parsed = parse_sheet(sheet, {stream["row_key"]: stream}, **{k: STAMP[k] for k in ("operator", "provenance", "settled_at")}) + assert parsed["records"] == [] + assert parsed["unset_row_count"] == 1 + assert sheet.read_bytes() == before + + +def test_ai_native_is_taxonomy_active_and_the_head_stays_nine(): + assert "ai-native" in ACTIVE_FAMILIES + assert family_flags("ai-native") == { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + assert len(PRODUCTION_FAMILIES) == 9 + assert "ai-native" in PRODUCTION_FAMILIES + stream = _stream( + "phrase native", + hint="ai-native", + label=None, + label_source="UNLABELLED", + ) + record = _apply( + stream, + _decision( + stream, + semantic_family="ai-native", + proposed_family="ai-native", + lane="B", + decision="ACCEPT FAMILY", + ), + )["records"][0] + assert record["semantic_family"] == "ai-native" + assert record["source_hint"] == "ai-native" + assert record["attest"] == "INFERRED" + + +def test_three_taxonomy_flags_stay_distinct(): + assert len(ACTIVE_FAMILIES) == 19 + for name in ACTIVE_FAMILIES: + assert family_flags(name) == { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + for name in CANDIDATE_FAMILIES: + assert family_flags(name)["taxonomy.active"] is False + assert family_flags(name)["evaluation.enabled"] is False + assert set(SURFACES) == {"A", "B", "C", "D", "HELD_OUTSIDE_LANES"} + + +def test_cli_blank_sheet_writes_zero_receipt_without_a_ledger(tmp_path): + stream = _stream("phrase cli") + rows = tmp_path / "rows.jsonl" + rows.write_text(json.dumps(stream) + "\n", encoding="utf-8") + sheet = tmp_path / "holding.tsv" + _sheet(sheet, stream, lane="HELD_OUTSIDE_LANES", proposed="gaming-meta") + receipt = tmp_path / "receipt.json" + log = tmp_path / "events.jsonl" + code = main( + [ + "settlement-apply", + "--stream-rows", + str(rows), + "--sheet", + str(sheet), + "--operator", + "tool-ready", + "--provenance", + "blank-sheet validation", + "--batch-id", + "unit-cli", + "--receipt", + str(receipt), + "--settlement-log", + str(log), + "--settled-at", + "2026-09-26T16:00:00Z", + "--stream-run-id", + "hs-unit", + ] + ) + assert code == 0 + body = json.loads(receipt.read_text(encoding="utf-8")) + assert body["settled_row_count"] == 0 + assert body["reserve_added"] == 0 + assert body["vendor_calls"] == 0 + assert body["ledger_mutated"] is False + assert body["evaluation_enabled"] is False + assert body["stream_run_id"] == "hs-unit" + assert body["settled_at"] == "2026-09-26T16:00:00Z" + assert body["input_sheets"][0]["identity"] == "holding.tsv" + assert len(body["input_sheets"][0]["sha256"]) == 64 + assert body["unset_row_count"] == 1 + assert body["unresolved_row_count"] == 0 + assert log.read_text(encoding="utf-8") == "" + assert not (tmp_path / "ledger.json").exists() + assert "phrase cli" not in receipt.read_text(encoding="utf-8") + + +def test_private_lane_sheets_parse_without_rewriting(): + root = Path.home() / "hlx-private" / "heldout-stream" + stream_path = root / "store" / "rows.jsonl" + attest = root / "attest" + names = { + "A": "lane-A-confirm-20260926.tsv", + "B": "lane-B-proposed-family-20260926.tsv", + "C": "lane-C-disambiguate-20260926.tsv", + "D": "lane-D-rights-blocked-20260926.tsv", + "HELD_OUTSIDE_LANES": "holding-hint-only-outside-lanes-20260926.tsv", + } + if not stream_path.is_file() or not all((attest / name).is_file() for name in names.values()): + pytest.skip("private lane sheets are host-local") + from hyperlexical.eval_settlement import load_stream + + stream = load_stream(stream_path) + expected = {"A": 204, "B": 65, "C": 21, "D": 16, "HELD_OUTSIDE_LANES": 33} + for lane, name in names.items(): + path = attest / name + before = path.read_bytes() + parsed = parse_sheet( + path, + stream, + operator="tool-ready", + provenance="blank-sheet validation", + settled_at="2026-09-26T16:00:00Z", + ) + assert parsed["unset_row_count"] + len(parsed["records"]) == expected[lane] + assert parsed["lane_rows"] == {lane: expected[lane]} + assert path.read_bytes() == before From c8c0bcc8aba7a25fd909d94f5a176d5821467182 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sun, 27 Sep 2026 14:51:46 -0700 Subject: [PATCH 5/5] Activate ai-native on the evaluation taxonomy The name stays on the nine-way production head. Settlement can now record it explicitly, evaluation stays off, and no row is settled. --- .../shadow/hyperlexical/eval_settlement.py | 847 +++++++++++++++++- 1 file changed, 846 insertions(+), 1 deletion(-) diff --git a/scripts/shadow/hyperlexical/eval_settlement.py b/scripts/shadow/hyperlexical/eval_settlement.py index 6cd2e16d..718ff769 100644 --- a/scripts/shadow/hyperlexical/eval_settlement.py +++ b/scripts/shadow/hyperlexical/eval_settlement.py @@ -1 +1,846 @@ -FULL_CONTENT_PENDING \ No newline at end of file +"""Evaluation-settlement apply path for the evaluation taxonomy. + +``source_hint``, ``semantic_family``, and ``attest`` are separate fields. +A confirm settlement may store the same family string in both ``source_hint`` +and ``semantic_family``. This module never copies one field into another, +and it never derives ``attest`` from ``decision``. + +The production classify head is not read or written here. ``ACCEPT`` records +the operator-selected attest. It does not promote existing evidence to +``OBSERVED``. This command does not admit rows to ``EVAL_RESERVE``. +""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +from typing import Any, Mapping, Sequence + +from .holdout_guard import normalized_text_sha256 + +SCHEMA = "hyperlex.eval_settlement.v1" +RECEIPT_SCHEMA = "hyperlex.eval_settlement_receipt.v1" +VENDOR_CALLS = 0 + +ACTIVE_FAMILIES: tuple[str, ...] = ( + "gaming-meta", + "betting-sharp", + "crypto-degen", + "internet-slang", + "memetic", + "social-status", + "relationship-dating", + "approval-disapproval", + "conflict-aggression", + "technology-ai", + "workplace-career", + "sports-competition", + "music-entertainment", + "fashion-aesthetic", + "regional-cultural", + "spiritual-mystic", + "identity-affiliation", + "politics-civic", + "ai-native", +) +CANDIDATE_FAMILIES: tuple[str, ...] = ( + "finance-retail", + "market-structure", + "sexual-romantic", + "substance-party", + "crime-illicit", + "health-fitness", +) +ABSTAIN = "none" +DECISIONS = ("ACCEPT", "RECLASSIFY", "NONE", "UNRESOLVED") +DECISION_ALIASES = { + "ACCEPT": "ACCEPT", + "ACCEPT FAMILY": "ACCEPT", + "RECLASSIFY": "RECLASSIFY", + "CHOOSE DIFFERENT FAMILY": "RECLASSIFY", + "NONE": "NONE", + "UNRESOLVED": "UNRESOLVED", +} +ATTESTS = ("OBSERVED", "INFERRED") +REGISTERS = ("slang", "domain-specific", "high-register", "general") +FUNCTIONS = ("address", "evaluation", "intensification", "reference", "affiliation") +SURFACES = { + "A": "CONFIRM", + "B": "PROPOSED FAMILY", + "C": "DISAMBIGUATE", + "D": "RIGHTS BLOCKED", + "HELD_OUTSIDE_LANES": "hint-only", +} +SHEET_COLUMNS = ( + "lane", + "row_key", + "text", + "source_hint", + "source_hint_provenance", + "current_label", + "label_source", + "rights", + "proposed_family_evidence", + "disambiguation_options", + "semantic_family", + "attest", + "register", + "function", + "decision", +) +SETTLED_DECISIONS = frozenset({"ACCEPT", "RECLASSIFY", "NONE"}) +_CLEARED_SOURCES = { + "wiktionary_category": "CC-BY-SA", + "wikipedia_prose": "CC-BY-SA", +} +_BLOCKED_SOURCES = { + "reddit_title": "RIGHTS_UNRESOLVED", + "kym_slang_list": "RIGHTS_UNRESOLVED", +} +_SCORE_KEYS = frozenset( + {"candidate_score", "model_error", "model_score", "prediction"} +) +_FORBIDDEN_KEYS = frozenset({"text", "normalized_text", "raw_text"}) | _SCORE_KEYS +_HINT_PROVENANCE = "stream.family_hint_not_a_label" + + +def refuse(message: str) -> None: + raise SystemExit(f"REFUSE: {message}") + + +def family_flags(name: str, *, activated: set[str] | None = None) -> dict[str, bool]: + """Three independent flags. Evaluation settlement does not enable a family.""" + activated = set(activated or ()) + if name in ACTIVE_FAMILIES: + return { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + if name in CANDIDATE_FAMILIES: + return { + "taxonomy.active": name in activated, + "evaluation.enabled": False, + "production.enabled": False, + } + refuse(f"family is not in the evaluation taxonomy: {name}") + raise AssertionError("refuse") + + +def _activated(activated: set[str] | None) -> set[str]: + extra = set(activated or ()) + unknown = sorted(extra - set(CANDIDATE_FAMILIES)) + if unknown: + refuse(f"activation refused for {unknown[0]}") + return extra + + +def _blank(value: Any) -> bool: + return value is None or (isinstance(value, str) and not value.strip()) + + +def _optional_text(value: Any) -> str | None: + if _blank(value): + return None + if not isinstance(value, str): + refuse("settlement field is not text") + return value.strip() + + +def _decision(value: Any) -> str: + token = _optional_text(value) + if token is None: + refuse("decision is empty") + mapped = DECISION_ALIASES.get(token.upper()) + if mapped is None: + if token.upper() == "REJECT": + refuse("reject is a production attest token, not an evaluation decision") + refuse(f"invalid decision: {token}") + return mapped + + +def _attest(value: Any) -> str | None: + token = _optional_text(value) + if token is None: + return None + mapped = token.upper() + if mapped not in ATTESTS: + refuse(f"invalid attest: {token}") + return mapped + + +def _family_token(value: Any) -> str | None: + token = _optional_text(value) + if token is None: + return None + return token.casefold() + + +def _closed_vocab(value: Any, allowed: tuple[str, ...], label: str) -> str | None: + token = _optional_text(value) + if token is None: + return None + mapped = token.casefold() + if mapped not in allowed: + refuse(f"invalid {label}: {token}") + return mapped + + +def _resolve_family(token: str | None, *, activated: set[str], allow_null: bool) -> str | None: + if token is None: + if allow_null: + return None + refuse("explicit semantic_family is required") + if token == ABSTAIN: + return ABSTAIN + if token in CANDIDATE_FAMILIES and token not in activated: + refuse(f"candidate family is inactive: {token}") + if token in ACTIVE_FAMILIES or token in activated: + return token + refuse(f"family is not in the evaluation taxonomy: {token}") + raise AssertionError("refuse") + + +def rights_of(stream_row: Mapping[str, Any]) -> str: + source_type = str(stream_row.get("source_type") or "") + if source_type in _CLEARED_SOURCES: + return _CLEARED_SOURCES[source_type] + if source_type in _BLOCKED_SOURCES: + return _BLOCKED_SOURCES[source_type] + refuse(f"unknown rights source_type: {source_type or 'missing'}") + raise AssertionError("refuse") + + +def evidence_hint(stream_row: Mapping[str, Any]) -> str | None: + hint = stream_row.get("family_hint_not_a_label") + if _blank(hint): + return None + return str(hint) + + +def sheet_hint_token(hint: str | None) -> str: + return "NO_HINT" if hint is None else hint + + +def sheet_label_token(label: Any) -> str: + if label is None: + return "UNLABELLED" + return str(label) + + +def settlement_allows_reserve(record: Mapping[str, Any]) -> bool: + """Rights and decision gates. This predicate does not admit a row.""" + if record.get("decision") not in SETTLED_DECISIONS: + return False + if record.get("rights") == "RIGHTS_UNRESOLVED": + return False + family = record.get("semantic_family") + attest = record.get("attest") + if family is None or attest not in ATTESTS: + return False + if attest == "OBSERVED" and record.get("decision") is None: + return False + return True + + +def _forbid_payload(value: Mapping[str, Any], where: str) -> None: + for key in value: + if key in _FORBIDDEN_KEYS: + refuse(f"{where} must not carry {key}") + + +def _refuse_scores(value: Mapping[str, Any], where: str) -> None: + for key in value: + if key in _SCORE_KEYS: + refuse(f"{where} must not carry {key}") + + +def _canonical_hash(stream_row: Mapping[str, Any]) -> str: + digest = normalized_text_sha256(str(stream_row.get("text") or "")) + stored = stream_row.get("normtext_sha256") + if isinstance(stored, str) and stored and stored != digest: + refuse("stream normtext_sha256 drifted from canonical row identity") + return digest + + +def _coerce_choice( + decision: str, + family: str | None, + attest: str | None, + proposed: str | None, + hint: str | None, +) -> tuple[str | None, str | None]: + if decision == "UNRESOLVED": + if family is not None or attest is not None: + refuse("UNRESOLVED requires null semantic_family and null attest") + return None, None + if decision == "NONE": + if family not in (None, ABSTAIN): + refuse("NONE requires semantic_family none") + if attest is None: + refuse("NONE requires an explicit attest") + return ABSTAIN, attest + if decision == "ACCEPT": + if family is None: + refuse("ACCEPT requires an explicit semantic_family") + if attest is None: + refuse("ACCEPT requires an explicit attest") + if proposed is not None and family != proposed: + refuse("ACCEPT does not match the proposed family") + return family, attest + if decision == "RECLASSIFY": + if family is None: + refuse("RECLASSIFY requires an explicit family") + if attest is None: + refuse("RECLASSIFY requires an explicit attest") + evidence = proposed if proposed is not None else hint + if evidence is not None and family == evidence: + refuse("RECLASSIFY family must differ from the proposed evidence family") + return family, attest + refuse(f"invalid decision: {decision}") + raise AssertionError("refuse") + + +def validate_settlement( + incoming: Mapping[str, Any], + stream_row: Mapping[str, Any], + *, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, +) -> dict[str, Any]: + """Fail closed. Does not read ``label_source`` as ``attest``.""" + _forbid_payload(incoming, "settlement") + extra = _activated(activated) + row_id = _optional_text(incoming.get("row_id")) or _optional_text(stream_row.get("row_key")) + if not row_id or row_id != stream_row.get("row_key"): + refuse("row_id is not in the held-out stream") + if row_id in set(prior_ids or ()): + refuse(f"duplicate settlement for {row_id}") + digest = _canonical_hash(stream_row) + claimed = _optional_text(incoming.get("text_hash")) + if claimed is None or claimed != digest: + refuse(f"text_hash does not match canonical row identity for {row_id}") + hint = evidence_hint(stream_row) + supplied_hint = incoming.get("source_hint") + if _blank(supplied_hint): + supplied_hint = None + elif isinstance(supplied_hint, str) and supplied_hint.strip() == "NO_HINT": + supplied_hint = None + elif isinstance(supplied_hint, str): + supplied_hint = supplied_hint.strip() + else: + refuse("source_hint is not text") + if supplied_hint != hint: + refuse(f"source_hint does not match held-out evidence for {row_id}") + rights = rights_of(stream_row) + sheet_rights = _optional_text(incoming.get("rights")) + if sheet_rights is not None and sheet_rights != rights: + refuse(f"rights state does not match held-out evidence for {row_id}") + decision = _decision(incoming.get("decision")) + proposed = _family_token(incoming.get("proposed_family")) + register = _closed_vocab(incoming.get("register"), REGISTERS, "register") + function = _closed_vocab(incoming.get("function"), FUNCTIONS, "function") + attest = _attest(incoming.get("attest")) + family = _family_token(incoming.get("semantic_family")) + if decision == "UNRESOLVED" and (register is not None or function is not None): + refuse("UNRESOLVED cannot carry register or function") + family, attest = _coerce_choice(decision, family, attest, proposed, hint) + family = _resolve_family(family, activated=extra, allow_null=decision == "UNRESOLVED") + operator = _optional_text(incoming.get("operator")) + settled_at = _optional_text(incoming.get("settled_at")) + provenance = _optional_text(incoming.get("provenance")) + if not operator: + refuse("operator is required") + if not settled_at: + refuse("settled_at is required") + if not provenance: + refuse("provenance is required") + lane = _optional_text(incoming.get("lane")) + if lane is not None and lane not in SURFACES: + refuse(f"unknown settlement surface: {lane}") + record = { + "schema": SCHEMA, + "row_id": row_id, + "text_hash": digest, + "decision": decision, + "semantic_family": family, + "attest": attest, + "register": register, + "function": function, + "source_hint": hint, + "operator": operator, + "settled_at": settled_at, + "provenance": provenance, + "rights": rights, + "lane": lane, + } + _forbid_payload(record, "settlement record") + return record + + +def _stream_index(rows: Sequence[Mapping[str, Any]]) -> dict[str, Mapping[str, Any]]: + index: dict[str, Mapping[str, Any]] = {} + for row in rows: + _refuse_scores(row, "held-out row") + key = row.get("row_key") + if not isinstance(key, str) or not key: + refuse("held-out row is missing row_key") + if key in index: + refuse(f"duplicate held-out row_id {key}") + index[key] = row + return index + + +def load_stream(path: str | Path) -> dict[str, Mapping[str, Any]]: + rows = [] + with Path(path).open(encoding="utf-8") as handle: + for line in handle: + if line.strip(): + payload = json.loads(line) + if not isinstance(payload, dict): + refuse("held-out stream row is not an object") + rows.append(payload) + return _stream_index(rows) + + +def _check_sheet_integrity( + parts: Sequence[str], + header: Sequence[str], + stream_row: Mapping[str, Any], +) -> str: + cell = dict(zip(header, parts)) + row_id = cell["row_key"].strip() + if row_id != stream_row.get("row_key"): + refuse("row_id is not in the held-out stream") + digest = normalized_text_sha256(cell["text"]) + if digest != _canonical_hash(stream_row): + refuse(f"text_hash does not match canonical row identity for {row_id}") + hint = evidence_hint(stream_row) + if cell["source_hint"] != sheet_hint_token(hint): + refuse(f"source_hint does not match held-out evidence for {row_id}") + if cell["source_hint_provenance"] != _HINT_PROVENANCE: + refuse(f"source_hint provenance does not match for {row_id}") + if cell["current_label"] != sheet_label_token(stream_row.get("label")): + refuse(f"current_label does not match held-out evidence for {row_id}") + if cell["label_source"] != str(stream_row.get("label_source") or ""): + refuse(f"label_source does not match held-out evidence for {row_id}") + if cell["rights"].strip() != rights_of(stream_row): + refuse(f"rights state does not match held-out evidence for {row_id}") + lane = cell["lane"].strip() + if lane not in SURFACES: + refuse(f"unknown settlement surface: {lane}") + return digest + + +def _operator_cells(cell: Mapping[str, str]) -> dict[str, str]: + return { + "semantic_family": cell["semantic_family"], + "attest": cell["attest"], + "register": cell["register"], + "function": cell["function"], + "decision": cell["decision"], + } + + +def parse_sheet( + path: str | Path, + stream: Mapping[str, Mapping[str, Any]], + *, + operator: str, + provenance: str, + settled_at: str, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, +) -> dict[str, Any]: + """Validate a private lane sheet. Does not write the sheet or fill decisions.""" + lines = Path(path).read_text(encoding="utf-8").splitlines() + if not lines: + refuse("settlement sheet is empty") + header = tuple(lines[0].split("\t")) + if header != SHEET_COLUMNS: + refuse("sheet header does not match the operator lane contract") + records = [] + unset = 0 + lane_rows: dict[str, int] = {} + seen: set[str] = set() + errors: list[str] = [] + for line in lines[1:]: + if not line.strip(): + continue + parts = line.split("\t") + if len(parts) != len(header): + errors.append("REFUSE: sheet row is not rectangular") + continue + row_id = parts[1].strip() + stream_row = stream.get(row_id) + if stream_row is None: + errors.append(f"REFUSE: row_id is not in the held-out stream: {row_id}") + continue + if row_id in seen or row_id in set(prior_ids or ()): + errors.append(f"REFUSE: duplicate settlement for {row_id}") + continue + seen.add(row_id) + try: + digest = _check_sheet_integrity(parts, header, stream_row) + cell = dict(zip(header, parts)) + lane = cell["lane"].strip() + lane_rows[lane] = lane_rows.get(lane, 0) + 1 + chosen = _operator_cells(cell) + if all(_blank(value) for value in chosen.values()): + unset += 1 + continue + if _blank(chosen["decision"]): + refuse(f"decision is empty for {row_id}; refusing to infer one") + record = validate_settlement( + { + "row_id": row_id, + "text_hash": digest, + "decision": chosen["decision"], + "semantic_family": chosen["semantic_family"], + "attest": chosen["attest"], + "register": chosen["register"], + "function": chosen["function"], + "source_hint": evidence_hint(stream_row), + "proposed_family": cell["proposed_family_evidence"], + "operator": operator, + "settled_at": settled_at, + "provenance": provenance, + "rights": cell["rights"], + "lane": lane, + }, + stream_row, + activated=activated, + prior_ids=prior_ids, + ) + records.append(record) + except SystemExit as exc: + errors.append(str(exc)) + if errors: + refuse("; ".join(message.removeprefix("REFUSE: ") for message in errors)) + return {"records": records, "unset_row_count": unset, "lane_rows": lane_rows, "row_ids": seen} + + +def load_record_file( + path: str | Path, + stream: Mapping[str, Mapping[str, Any]], + *, + operator: str, + provenance: str, + settled_at: str, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, +) -> list[dict[str, Any]]: + records = [] + seen = set(prior_ids or ()) + with Path(path).open(encoding="utf-8") as handle: + for line in handle: + if not line.strip(): + continue + payload = json.loads(line) + if not isinstance(payload, dict): + refuse("settlement record is not an object") + row_id = payload.get("row_id") + if not isinstance(row_id, str) or row_id not in stream: + refuse(f"row_id is not in the held-out stream: {row_id}") + if row_id in seen: + refuse(f"duplicate settlement for {row_id}") + payload = dict(payload) + payload.setdefault("operator", operator) + payload.setdefault("provenance", provenance) + payload.setdefault("settled_at", settled_at) + records.append( + validate_settlement( + payload, + stream[row_id], + activated=activated, + prior_ids=prior_ids, + ) + ) + seen.add(row_id) + return records + + +def _decision_counts(records: Sequence[Mapping[str, Any]]) -> dict[str, int]: + counts = {name: 0 for name in DECISIONS} + for record in records: + counts[str(record["decision"])] += 1 + return counts + + +def _family_counts(records: Sequence[Mapping[str, Any]]) -> dict[str, int]: + counts = {name: 0 for name in (*ACTIVE_FAMILIES, ABSTAIN)} + for record in records: + if record["decision"] not in SETTLED_DECISIONS: + continue + family = record.get("semantic_family") + if family in counts: + counts[str(family)] += 1 + return counts + + +def build_receipt( + records: Sequence[Mapping[str, Any]], + *, + batch_id: str, + operator: str, + provenance: str, + unset_row_count: int, + lane_rows: Mapping[str, int], + stream_run_id: str = "", + settled_at: str = "", + input_sheets: Sequence[Mapping[str, str]] | None = None, +) -> dict[str, Any]: + if not str(batch_id or "").strip(): + refuse("batch_id is required") + if not str(operator or "").strip() or not str(provenance or "").strip(): + refuse("operator and provenance are required") + settled = [record for record in records if record["decision"] in SETTLED_DECISIONS] + unresolved = [record for record in records if record["decision"] == "UNRESOLVED"] + body = { + "schema": RECEIPT_SCHEMA, + "batch_id": batch_id, + "operator": operator, + "provenance": provenance, + "settled_row_count": len(settled), + "unresolved_row_count": len(unresolved), + "unset_row_count": unset_row_count, + "decision_counts": _decision_counts(records), + "family_counts": _family_counts(records), + "observed_count": sum(1 for record in settled if record.get("attest") == "OBSERVED"), + "inferred_count": sum(1 for record in settled if record.get("attest") == "INFERRED"), + "rights_blocked_settled_count": sum( + 1 for record in settled if record.get("rights") == "RIGHTS_UNRESOLVED" + ), + "reserve_eligible_count": sum(1 for record in records if settlement_allows_reserve(record)), + "reserve_added": 0, + "ledger_mutated": False, + "vendor_calls": VENDOR_CALLS, + "evaluation_enabled": False, + "lane_rows": dict(lane_rows), + } + run_id = str(stream_run_id or "").strip() + if run_id: + body["stream_run_id"] = run_id + stamp = str(settled_at or "").strip() + if stamp: + body["settled_at"] = stamp + if input_sheets: + sheets = [] + for item in input_sheets: + identity = str(item.get("identity") or "").strip() + digest = str(item.get("sha256") or "").strip() + if not identity or len(digest) != 64: + refuse("input sheet identity is incomplete") + sheets.append({"identity": identity, "sha256": digest}) + body["input_sheets"] = sheets + _walk_forbid(body) + return seal_receipt(body) + + +def seal_receipt(body: Mapping[str, Any]) -> dict[str, Any]: + payload = {key: value for key, value in body.items() if key != "receipt_sha256"} + raw = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8") + sealed = dict(payload) + sealed["receipt_sha256"] = hashlib.sha256(raw).hexdigest() + return sealed + + +def receipt_sha_matches(receipt: Mapping[str, Any]) -> bool: + sealed = seal_receipt(receipt) + return sealed["receipt_sha256"] == receipt.get("receipt_sha256") + + +def _walk_forbid(value: Any) -> None: + if isinstance(value, dict): + for key, item in value.items(): + if key in _FORBIDDEN_KEYS: + refuse(f"receipt must not carry {key}") + _walk_forbid(item) + elif isinstance(value, list): + for item in value: + _walk_forbid(item) + + +def load_logged_ids(path: str | Path) -> set[str]: + log = Path(path) + if not log.exists(): + return set() + found = set() + for line in log.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + payload = json.loads(line) + if not isinstance(payload, dict): + refuse("settlement log row is not an object") + _forbid_payload(payload, "settlement log") + row_id = payload.get("row_id") + if not isinstance(row_id, str) or not row_id: + refuse("settlement log row is missing row_id") + if row_id in found: + refuse(f"settlement log already contains a duplicate for {row_id}") + found.add(row_id) + return found + + +def _write_exclusive(path: Path, text: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "w", encoding="utf-8") as handle: + handle.write(text) + + +def commit_settlement( + records: Sequence[Mapping[str, Any]], + receipt: Mapping[str, Any], + *, + log_path: str | Path, + receipt_path: str | Path, +) -> dict[str, Any]: + """Append settlement rows, then write a new receipt. Never rewrites either.""" + receipt_file = Path(receipt_path) + if receipt_file.exists(): + refuse(f"receipt already exists: {receipt_file}") + log = Path(log_path) + prior = load_logged_ids(log) + fresh_ids = [] + for record in records: + _forbid_payload(record, "settlement record") + row_id = str(record["row_id"]) + if row_id in prior or row_id in fresh_ids: + refuse(f"duplicate settlement for {row_id}") + fresh_ids.append(row_id) + if not log.exists(): + _write_exclusive(log, "") + if records: + with log.open("a", encoding="utf-8") as handle: + for record in records: + handle.write(json.dumps(record, sort_keys=True) + "\n") + sealed = dict(receipt) + _walk_forbid(sealed) + _write_exclusive(receipt_file, json.dumps(sealed, indent=2, sort_keys=True) + "\n") + os.chmod(log, 0o600) + return sealed + + +def settle( + stream_rows: Sequence[Mapping[str, Any]], + decisions: Sequence[Mapping[str, Any]], + *, + batch_id: str, + operator: str, + provenance: str, + settled_at: str, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, + unset_row_count: int = 0, + lane_rows: Mapping[str, int] | None = None, +) -> dict[str, Any]: + """Validate explicit decisions. Blank callers pass no decisions and settle nothing.""" + stream = _stream_index(stream_rows) + prior = set(prior_ids or ()) + records = [] + seen: set[str] = set() + for incoming in decisions: + row_id = incoming.get("row_id") + stream_row = stream.get(row_id) if isinstance(row_id, str) else None + if stream_row is None: + refuse(f"row_id is not in the held-out stream: {row_id}") + if row_id in seen: + refuse(f"duplicate settlement for {row_id}") + seen.add(str(row_id)) + payload = dict(incoming) + payload.setdefault("operator", operator) + payload.setdefault("provenance", provenance) + payload.setdefault("settled_at", settled_at) + records.append( + validate_settlement(payload, stream_row, activated=activated, prior_ids=prior) + ) + receipt = build_receipt( + records, + batch_id=batch_id, + operator=operator, + provenance=provenance, + unset_row_count=unset_row_count, + lane_rows=dict(lane_rows or {}), + ) + return {"records": records, "receipt": receipt} + + +def run_settlement_apply(args: Any) -> int: + """CLI entry. Reads operator cells. Does not choose them or touch the ledger.""" + import sys + + stream = load_stream(args.stream_rows) + activated = set(args.activated_family or []) + sheets = list(args.sheet or []) + records_path = str(args.records or "").strip() + if not sheets and not records_path: + refuse("no operator surface supplied") + prior = load_logged_ids(args.settlement_log) + records: list[dict[str, Any]] = [] + unset = 0 + lane_rows: dict[str, int] = {} + decided: set[str] = set() + surfaced: set[str] = set() + for sheet in sheets: + parsed = parse_sheet( + sheet, + stream, + operator=args.operator, + provenance=args.provenance, + settled_at=args.settled_at, + activated=activated, + prior_ids=prior | decided, + ) + for row_id in parsed["row_ids"]: + if row_id in surfaced: + refuse(f"duplicate settlement for {row_id}") + surfaced.add(row_id) + for record in parsed["records"]: + decided.add(record["row_id"]) + records.append(record) + unset += int(parsed["unset_row_count"]) + for lane, count in parsed["lane_rows"].items(): + lane_rows[lane] = lane_rows.get(lane, 0) + count + if records_path: + loaded = load_record_file( + records_path, + stream, + operator=args.operator, + provenance=args.provenance, + settled_at=args.settled_at, + activated=activated, + prior_ids=prior | decided, + ) + for record in loaded: + if record["row_id"] in decided or record["row_id"] in surfaced: + refuse(f"duplicate settlement for {record['row_id']}") + decided.add(record["row_id"]) + records.append(record) + sheet_identities = [] + for sheet in sheets: + raw = Path(sheet).read_bytes() + sheet_identities.append( + {"identity": Path(sheet).name, "sha256": hashlib.sha256(raw).hexdigest()} + ) + receipt = build_receipt( + records, + batch_id=args.batch_id, + operator=args.operator, + provenance=args.provenance, + unset_row_count=unset, + lane_rows=lane_rows, + stream_run_id=str(getattr(args, "stream_run_id", "") or ""), + settled_at=str(args.settled_at or ""), + input_sheets=sheet_identities, + ) + sealed = commit_settlement( + records, + receipt, + log_path=args.settlement_log, + receipt_path=args.receipt, + ) + sys.stdout.write(json.dumps(sealed, indent=2, sort_keys=True) + "\n") + return 0