Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,14 @@ All notable changes follow Keep a Changelog and Semantic Versioning.

## [Unreleased]

### Fixed

- Reject quality-gate approval when external evaluation is absent or its F1 is invalid,
and reject missing or incomparable critical-entity counts. Report which entities
actually have evaluated support.
- Require every quality criterion to pass explicitly when verifying a release report,
rejecting incomplete reports and legacy skipped external evaluations.

## [1.34.0] - 2026-09-30

### Added
Expand Down
26 changes: 26 additions & 0 deletions HANDOVER.md
Original file line number Diff line number Diff line change
Expand Up @@ -399,3 +399,29 @@ verification unless they are intentionally retained as a user-approved release a
- Push only when explicitly requested.
- A version is published only after its matching tag, successful release workflow, PyPI artifact,
and GitHub release exist. Until then, keep changes under `[Unreleased]`.


## Required quality evidence review (2026-10-02)

Three regression tests reproduced approval with absent external evidence, a
missing critical entity, and an infinite external F1. The gate now rejects those
inputs, compares critical annotation support, and distinguishes represented from
unrepresented entities. The release verifier checks all five criteria and rejects
legacy skipped external evaluations. Per-entity F1 uses the equivalent count
formula, avoiding division by zero when TP is zero and FP/FN are nonzero.

Observed verification using the frozen lockfile:

- Quality gate, comparator and release-verifier tests: 40 passed.
- Suite excluding ONNX/BIO modules with `--no-cov`: 510 passed; 1 Tesseract skip.
- Pre-commit, mypy, strict docs build, package build and release verifier passed.
- Full coverage and supported-platform acceptance remain pending CI.

No detector, held-out sample, published score or quality tolerance was changed.
The remaining difference between the roadmap's strict exact-boundary improvement
protocol and the existing overlap-based tolerance gate requires a separate
methodology change. This patch must not be described as proving that protocol.

Reusable lesson assessment: missing measurements cannot count as passing release
criteria. Explicit negative tests now enforce that invariant at generation and
consumption of gate reports.
16 changes: 4 additions & 12 deletions benchmarks/compare_quality.py
Original file line number Diff line number Diff line change
Expand Up @@ -146,22 +146,14 @@ def compare_results(
b_tp = b_et.get("true_positives", 0)
b_fp = b_et.get("false_positives", 0)
b_fn = b_et.get("false_negatives", 0)
b_f1 = (
(2 * (b_tp / (b_tp + b_fp)) * (b_tp / (b_tp + b_fn)))
/ ((b_tp / (b_tp + b_fp)) + (b_tp / (b_tp + b_fn)))
if (b_tp + b_fp > 0 and b_tp + b_fn > 0)
else 0.0
)
b_denominator = 2 * b_tp + b_fp + b_fn
b_f1 = 2 * b_tp / b_denominator if b_denominator else 0.0

c_tp = c_et.get("true_positives", 0)
c_fp = c_et.get("false_positives", 0)
c_fn = c_et.get("false_negatives", 0)
c_f1 = (
(2 * (c_tp / (c_tp + c_fp)) * (c_tp / (c_tp + c_fn)))
/ ((c_tp / (c_tp + c_fp)) + (c_tp / (c_tp + c_fn)))
if (c_tp + c_fp > 0 and c_tp + c_fn > 0)
else 0.0
)
c_denominator = 2 * c_tp + c_fp + c_fn
c_f1 = 2 * c_tp / c_denominator if c_denominator else 0.0

per_entity_shifts[et] = {
"delta_tp": c_tp - b_tp,
Expand Down
81 changes: 59 additions & 22 deletions benchmarks/quality_gate.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
import argparse
import json
import logging
import math
from dataclasses import dataclass
from pathlib import Path
from typing import Any
Expand Down Expand Up @@ -200,15 +201,27 @@ def evaluate_quality_gate(
base_per_entity = baseline_record.get("per_entity", {})
cand_per_entity = candidate_record.get("per_entity", {})
critical_regressions: list[dict[str, Any]] = []
missing_critical_evidence: list[str] = []
checked_entities: list[str] = []

for ent in HIGH_RISK_ENTITY_TYPES:
for ent in sorted(HIGH_RISK_ENTITY_TYPES):
if ent in base_per_entity:
b_et = base_per_entity[ent]
c_et = cand_per_entity.get(ent, {})
b_tp = b_et.get("true_positives", 0)
b_fn = b_et.get("false_negatives", 0)
c_tp = c_et.get("true_positives", 0)
c_fn = c_et.get("false_negatives", 0)
b_tp = b_et.get("true_positives")
b_fn = b_et.get("false_negatives")
c_tp = c_et.get("true_positives")
c_fn = c_et.get("false_negatives")
counts = (b_tp, b_fn, c_tp, c_fn)
if not all(isinstance(n, int) and not isinstance(n, bool) and n >= 0 for n in counts):
missing_critical_evidence.append(ent)
continue
if b_tp + b_fn != c_tp + c_fn:
missing_critical_evidence.append(ent)
continue
if b_tp + b_fn == 0:
continue
checked_entities.append(ent)

b_rec = b_tp / (b_tp + b_fn) if (b_tp + b_fn) > 0 else 1.0
c_rec = c_tp / (c_tp + c_fn) if (c_tp + c_fn) > 0 else 1.0
Expand All @@ -224,55 +237,79 @@ def evaluate_quality_gate(
}
)

critical_ok = len(critical_regressions) == 0
critical_ok = (
not critical_regressions and not missing_critical_evidence and bool(checked_entities)
)
if not critical_ok:
reason = (
f"Recall regressed for {len(critical_regressions)} high-risk identifiers: "
f"{[r['entity_type'] for r in critical_regressions]}"
)
if missing_critical_evidence or not checked_entities:
reason = "Critical identifier recall evidence is missing, invalid or incomparable."
else:
reason = (
f"Recall regressed for {len(critical_regressions)} high-risk identifiers: "
f"{[r['entity_type'] for r in critical_regressions]}"
)
failure_reasons.append(reason)
results["critical_entities_floor"] = QualityGateCriteriaResult(
criterion_name="critical_entities_floor",
passed=False,
details={"regressed_entities": critical_regressions},
details={
"regressed_entities": critical_regressions,
"missing_or_incomparable_entities": missing_critical_evidence,
},
failure_reason=reason,
)
else:
results["critical_entities_floor"] = QualityGateCriteriaResult(
criterion_name="critical_entities_floor",
passed=True,
details={"protected_entities_checked": list(HIGH_RISK_ENTITY_TYPES)},
details={
"protected_entities_checked": checked_entities,
"unrepresented_entities": sorted(HIGH_RISK_ENTITY_TYPES - set(checked_entities)),
},
)

# 5. Independent External Generalization Track (PIIMB)
if external_eval_record is not None:
ext_micro = external_eval_record.get("character_micro_metrics", {})
ext_f1 = ext_micro.get("f1", 0.0)
ext_ok = ext_f1 >= external_f1_floor
ext_f1 = ext_micro.get("f1") if isinstance(ext_micro, dict) else None
valid_score = (
isinstance(ext_f1, (int, float))
and not isinstance(ext_f1, bool)
and math.isfinite(ext_f1)
and 0 <= ext_f1 <= 1
)
ext_ok = valid_score and isinstance(ext_f1, (int, float)) and ext_f1 >= external_f1_floor
if not ext_ok:
reason = (
f"External benchmark (PIIMB) character F1 {ext_f1:.4f} is below floor "
f"{external_f1_floor:.4f}."
"External benchmark character F1 is missing, invalid or below the required floor."
)
failure_reasons.append(reason)
results["external_generalization"] = QualityGateCriteriaResult(
criterion_name="external_generalization",
passed=False,
details={"external_f1": ext_f1, "required_floor": external_f1_floor},
details={
"external_f1": ext_f1 if valid_score else None,
"required_floor": external_f1_floor,
},
failure_reason=reason,
)
else:
results["external_generalization"] = QualityGateCriteriaResult(
criterion_name="external_generalization",
passed=True,
details={"external_f1": ext_f1, "required_floor": external_f1_floor},
details={
"external_f1": ext_f1 if valid_score else None,
"required_floor": external_f1_floor,
},
)
else:
# Informational pass if external track was not evaluated in this specific run
reason = "Independent external generalization evidence is required for shipment."
failure_reasons.append(reason)
results["external_generalization"] = QualityGateCriteriaResult(
criterion_name="external_generalization",
passed=True,
passed=False,
details={"status": "not_evaluated_in_current_run"},
failure_reason=reason,
)

all_passed = all(r.passed for r in results.values())
Expand Down Expand Up @@ -304,7 +341,7 @@ def main() -> int:
"--external",
type=Path,
default=None,
help="Optional path to external generalization artifact JSON (PIIMB).",
help="External generalization artifact JSON (PIIMB), required for a passing gate.",
)
parser.add_argument(
"--output",
Expand All @@ -317,7 +354,7 @@ def main() -> int:
baseline_data = json.loads(args.baseline.read_text(encoding="utf-8"))
candidate_data = json.loads(args.candidate.read_text(encoding="utf-8"))
external_data = None
if args.external and args.external.exists():
if args.external is not None:
external_data = json.loads(args.external.read_text(encoding="utf-8"))

report = evaluate_quality_gate(
Expand Down
21 changes: 21 additions & 0 deletions docs/quality_benchmarks.md
Original file line number Diff line number Diff line change
Expand Up @@ -146,3 +146,24 @@ Evaluated against the pinned validation split (`a785eb528e28be2693c3718a27e06697
See also the formal [Model Card](model_card.md) for full architecture specifications and hardware profiles.




## Required evidence for a quality-gate report

A release-quality evaluation needs an external evaluation record. Omitting it
produces `REJECT`; a missing file supplied through `--external` is an error.
External character F1 must be a finite number between zero and one and meet the
configured floor. Critical-entity comparisons require matching annotation
support and explicit nonnegative integer TP/FN counts. The report distinguishes
entities with observed support from those not represented in that evaluation.

`verify_release.py --quality-gate-report` requires all five criteria to be
present and explicitly passing. A top-level `passed: true` flag is insufficient,
and legacy reports that treated an unevaluated external track as passing are
rejected. Regenerate those reports with external evidence rather than editing
approval flags.

These checks validate evidence presence and consistency. They do not authenticate
an artifact, prove that a corpus is independent, or replace the roadmap's stricter
experimental protocol. This change does not alter detector behavior or establish
an improvement in the published quality baseline.
42 changes: 40 additions & 2 deletions scripts/verify_release.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@
import email.message
import email.parser
import json
import math
import os
import re
import tarfile
Expand Down Expand Up @@ -159,9 +160,46 @@ def verify_quality_gate_report(report_path: Path) -> None:
if not report_path.is_file():
raise ValueError(f"quality gate report not found at {report_path}")
data = json.loads(report_path.read_text(encoding="utf-8"))
if not data.get("passed", False) or data.get("recommendation") != "SHIP":
summary = data.get("summary_reasons", ["unspecified failure"])
if (
not isinstance(data, dict)
or data.get("passed") is not True
or data.get("recommendation") != "SHIP"
):
summary = (
data.get("summary_reasons", ["unspecified failure"])
if isinstance(data, dict)
else ["invalid report"]
)
raise ValueError(f"quality gate failed: {summary}")
required = {
"pinned_provenance",
"paired_f1_delta",
"precision_protection",
"critical_entities_floor",
"external_generalization",
}
criteria = data.get("criteria")
if not isinstance(criteria, dict) or not required <= criteria.keys():
raise ValueError("quality gate report is missing required criteria")
for name in sorted(required):
criterion = criteria[name]
if not isinstance(criterion, dict) or criterion.get("passed") is not True:
raise ValueError(f"quality gate criterion did not pass: {name}")
external_details = criteria["external_generalization"].get("details")
if (
not isinstance(external_details, dict)
or external_details.get("status") == "not_evaluated_in_current_run"
or "external_f1" not in external_details
):
raise ValueError("quality gate report lacks external evaluation evidence")
external_f1 = external_details["external_f1"]
if (
not isinstance(external_f1, (int, float))
or isinstance(external_f1, bool)
or not math.isfinite(external_f1)
or not 0 <= external_f1 <= 1
):
raise ValueError("quality gate report has invalid external evaluation evidence")
rec = data.get("recommendation")
print(f"verified quality gate report: candidate approved for shipment ({rec})")

Expand Down
58 changes: 57 additions & 1 deletion tests/integration/test_release_verifier.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import email.message
import io
import json
import tarfile
import zipfile
from pathlib import Path
Expand Down Expand Up @@ -152,7 +153,7 @@ def test_verify_quality_gate_report_passing(
tmp_path: Path, capsys: pytest.CaptureFixture[str]
) -> None:
report = tmp_path / "gate.json"
report.write_text('{"passed": true, "recommendation": "SHIP"}', encoding="utf-8")
report.write_text(json.dumps(_complete_gate_report()), encoding="utf-8")
verify_quality_gate_report(report)
assert "approved for shipment" in capsys.readouterr().out

Expand All @@ -171,3 +172,58 @@ def test_verify_quality_gate_report_missing_file(tmp_path: Path) -> None:
missing = tmp_path / "nonexistent.json"
with pytest.raises(ValueError, match="quality gate report not found"):
verify_quality_gate_report(missing)


def _complete_gate_report() -> dict[str, object]:
return {
"passed": True,
"recommendation": "SHIP",
"criteria": {
name: {"passed": True, "details": {"external_f1": 0.82}}
for name in (
"pinned_provenance",
"paired_f1_delta",
"precision_protection",
"critical_entities_floor",
"external_generalization",
)
},
}


@pytest.mark.parametrize(
"criterion",
[
"pinned_provenance",
"paired_f1_delta",
"precision_protection",
"critical_entities_floor",
"external_generalization",
],
)
@pytest.mark.parametrize("state", ["missing", "failed", "string"])
def test_verify_quality_gate_report_checks_each_criterion(
tmp_path: Path, criterion: str, state: str
) -> None:
data = _complete_gate_report()
criteria = data["criteria"]
assert isinstance(criteria, dict)
if state == "missing":
del criteria[criterion]
else:
criteria[criterion]["passed"] = False if state == "failed" else "true"
path = tmp_path / "gate.json"
path.write_text(json.dumps(data), encoding="utf-8")
with pytest.raises(ValueError, match="quality gate"):
verify_quality_gate_report(path)


def test_verify_quality_gate_rejects_legacy_skipped_external_track(tmp_path: Path) -> None:
data = _complete_gate_report()
criteria = data["criteria"]
assert isinstance(criteria, dict)
criteria["external_generalization"]["details"] = {"status": "not_evaluated_in_current_run"}
path = tmp_path / "gate.json"
path.write_text(json.dumps(data), encoding="utf-8")
with pytest.raises(ValueError, match="external evaluation evidence"):
verify_quality_gate_report(path)
Loading
Loading