From 85b7167c2b4ec9542d55d01fa4c40906dfe106bf Mon Sep 17 00:00:00 2001 From: Saverio Mazza Date: Fri, 2 Oct 2026 08:17:13 +0200 Subject: [PATCH 1/2] docs: make README examples match the public token contract --- CHANGELOG.md | 5 +++ HANDOVER.md | 24 +++++++++++++++ README.md | 37 ++++++++++++----------- tests/integration/test_readme_examples.py | 19 ++++++++++++ 4 files changed, 68 insertions(+), 17 deletions(-) create mode 100644 tests/integration/test_readme_examples.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 8600206..e0db199 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,11 @@ All notable changes follow Keep a Changelog and Semantic Versioning. ## [Unreleased] +### Fixed + +- Align README aliases with the emitted token format and execute its Python examples + in an integration test. Clarify overlap benchmark scoring and streaming limitations. + ## [1.34.0] - 2026-09-30 ### Added diff --git a/HANDOVER.md b/HANDOVER.md index 091c165..ea0affa 100644 --- a/HANDOVER.md +++ b/HANDOVER.md @@ -399,3 +399,27 @@ verification unless they are intentionally retained as a user-approved release a - Push only when explicitly requested. - A version is published only after its matching tag, successful release workflow, PyPI artifact, and GitHub release exist. Until then, keep changes under `[Unreleased]`. + + +## Executable README review (2026-10-02) + +The first Python quickstart failed its assertion against the current package: +README examples used long entity names while the implementation emits short codes. +Corrected aliases and added a test that executes all Python fences in order, +providing the documented JSON file in a temporary directory. The asynchronous +example is compiled and defined by that test; existing streaming tests exercise +actual asynchronous processing. No token format or library behavior changed. + +The historical 1.20.0 score is now correctly labeled as an overlap result, and the +streaming introduction no longer promises arbitrary split-entity safety. These +clarifications reflect the existing scorer and heuristic segment splitter. + +Observed verification with the frozen lockfile: + +- README, streaming and session tests: 11 passed. +- Suite excluding ONNX/BIO modules with `--no-cov`: 490 passed; 1 Tesseract skip. +- Pre-commit, mypy (156 source files), strict docs build, package build and release + verifier passed. Full coverage and platform evidence remain CI responsibilities. + +Reusable lesson assessment: executable onboarding assertions catch token-contract + drift that prose review misses. This is enforced by the new integration test. diff --git a/README.md b/README.md index a0fff6d..cbf7182 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@ payloads. ```text Email paolo@example.com from 192.0.2.10. ↓ -Email from . +Email from . ``` Pseudonymize detects structured sensitive values locally and transforms them into numbered, @@ -55,13 +55,13 @@ no telemetry or model downloads, and denies remote-capable backends by default. The engine is strictly gated on detection accuracy against the `ai4privacy/pii-masking-openpii-1.5m` dataset (validation split, 1000 randomly sampled rows). -**Current Baseline (1.20.0, 1000 rows, strict boundaries and type matching, ONNX backend enabled):** +**Historical baseline (1.20.0, 1,000 rows, one-to-one positive boundary overlap and exact entity-type matching, ONNX backend enabled):** - **Precision:** 0.8587 - **Recall:** 0.8016 - **F1 Score:** 0.8292 Each detection is paired with at most one annotation and must agree with it on entity -type. Figures published before the scoring was strictly corrected are not comparable; see +type. These figures measure overlapping spans, not exact-boundary accuracy. Figures published before the scoring was strictly corrected are not comparable; see [docs/benchmarks.md](docs/benchmarks.md). ## Installation @@ -116,7 +116,7 @@ from pseudonymize import pseudonymize, redact safe = pseudonymize("Email paolo@example.com") hidden = redact("Email paolo@example.com") -assert safe == "Email " +assert safe == "Email " assert hidden == "Email [REDACTED]" ``` @@ -129,7 +129,7 @@ from pseudonymize import Pseudonymizer result = Pseudonymizer().process_with_report("Email paolo@example.com from 192.0.2.10.") -assert result.output == "Email from ." +assert result.output == "Email from ." assert result.statistics.detections_found == 2 assert result.detections[0].backend == "rules" assert "paolo@example.com" not in repr(result) @@ -155,8 +155,8 @@ payload = { result = Pseudonymizer(policy=Policy.llm()).process_data_with_report(payload) -assert result.output["messages"][0]["content"] == "Email " -assert result.output["messages"][1]["content"] == "Use again" +assert result.output["messages"][0]["content"] == "Email " +assert result.output["messages"][1]["content"] == "Use again" assert result.output["model"] == "example-model" ``` @@ -164,7 +164,11 @@ The input is not mutated. Dictionary keys and non-string values are preserved. ### Stream LLM responses -Real-time WebSocket chunks from OpenAI or Anthropic can be processed seamlessly without risking split-entity leakage across chunks. Both `process_stream` and `process_stream_async` are available: +Both `process_stream` and `process_stream_async` buffer incoming text chunks and retain +context between emitted segments. Segment boundaries are heuristic: punctuation and long +unbroken input can split an entity, so streaming does not guarantee the same detection as +processing the complete text. Use `process()` on a complete bounded message when that +equivalence is required. ```python import asyncio @@ -183,20 +187,19 @@ async def handle_stream(socket): ## Transformation modes -| Mode | Example | Identity behavior | -| --- | --- | --- | -| `numbered` | `` | Stable inside one explicit scope | -| `generic` | `` | Does not distinguish values of the same type | -| `deterministic` | `` | Stable for the same key, namespace, type, and normalized value | -| `redacted` | `[REDACTED]` | Removes type and identity distinction | +- `numbered`: ``, stable inside one explicit scope. +- `generic`: ``, without distinguishing values of the same type. +- `deterministic`: an HMAC-derived `` token, stable for the same key, + namespace, entity type, and normalized value. +- `redacted`: `[REDACTED]`, without type or identity distinctions. ```python from pseudonymize import Pseudonymizer scope = Pseudonymizer().new_scope() -assert scope.process("paolo@example.com").text == "" -assert scope.process("maria@example.com and paolo@example.com").text == (" and ") +assert scope.process("paolo@example.com").text == "" +assert scope.process("maria@example.com and paolo@example.com").text == (" and ") ``` Deterministic mode uses HMAC-SHA256 and requires a key of at least 32 bytes: @@ -284,7 +287,7 @@ result = Pseudonymizer().process( include_mapping=True, ) -assert result.restore("Reply to .") == "Reply to paolo@example.com." +assert result.restore("Reply to .") == "Reply to paolo@example.com." ``` Mappings contain sensitive source values. They are hidden from `repr`, never persisted by the diff --git a/tests/integration/test_readme_examples.py b/tests/integration/test_readme_examples.py new file mode 100644 index 0000000..e385683 --- /dev/null +++ b/tests/integration/test_readme_examples.py @@ -0,0 +1,19 @@ +"""Execute the README quickstart against the installed library.""" + +import re +from pathlib import Path +from typing import Any + +import pytest + + +def test_readme_python_examples(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + readme = Path(__file__).resolve().parents[2] / "README.md" + examples = re.findall(r"^```python\n(.*?)^```", readme.read_text(encoding="utf-8"), re.M | re.S) + assert examples, "README must contain executable Python examples" + monkeypatch.chdir(tmp_path) + (tmp_path / "requests.json").write_text('{"email": "reader@example.com"}', encoding="utf-8") + namespace: dict[str, Any] = {} + for index, example in enumerate(examples, start=1): + exec(compile(example, f"README.md:python-example-{index}", "exec"), namespace) # noqa: S102 + assert "reader@example.com" not in (tmp_path / "requests.safe.json").read_text(encoding="utf-8") From c9d43edf95dc174567ed16c45c916b8f7396bea7 Mon Sep 17 00:00:00 2001 From: Paolo Mazza Date: Fri, 9 Oct 2026 11:58:06 +0000 Subject: [PATCH 2/2] fix: verify and audit the optional service extra --- CHANGELOG.md | 4 ++++ scripts/audit_extras.py | 2 +- scripts/audit_install.py | 4 ++++ scripts/verify_release.py | 4 +++- tests/integration/test_release_verifier.py | 13 +++++++++++++ 5 files changed, 25 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 93ec177..979c70b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,10 @@ All notable changes follow Keep a Changelog and Semantic Versioning. - Optional bounded ONNX case recovery for lowercase PERSON candidates using high-confidence re-inference and constrained BIO decoding. Disabled by default; no name dictionaries or fabricated spans. - Python-owned Lambda HTTP service with verified model artifacts, required gateway authentication, ONNX-only person detection, and explicit inference failure. Hosted processing disables coreference expansion so each name occurrence requires model confirmation. +### Fixed +- Include the hosted-service extra in release verification and isolated installation + audits, while keeping service frameworks out of the dependency-free base import. + ## [1.36.0] - 2026-10-09 ### Added diff --git a/scripts/audit_extras.py b/scripts/audit_extras.py index 4c2cf20..7f8d028 100644 --- a/scripts/audit_extras.py +++ b/scripts/audit_extras.py @@ -6,7 +6,7 @@ import tempfile from pathlib import Path -DOCUMENTED_EXTRAS = ("html", "ml", "ocr", "office", "pdf", "remote") +DOCUMENTED_EXTRAS = ("html", "ml", "ocr", "office", "pdf", "remote", "service") def audit_extras(wheel_path: Path, python_executable: str | None = None) -> None: diff --git a/scripts/audit_install.py b/scripts/audit_install.py index c51b2c7..41fe382 100644 --- a/scripts/audit_install.py +++ b/scripts/audit_install.py @@ -11,6 +11,9 @@ "aiohttp", "docling", "docx", + "fastapi", + "pydantic", + "uvicorn", "httpx", "onnxruntime", "openpyxl", @@ -19,6 +22,7 @@ } EXPECTED_BASE_REQUIREMENTS: frozenset[str] = frozenset() EXTRA_CHECKS: dict[str, tuple[str, frozenset[str]]] = { + "service": ("pseudonymize.service", frozenset({"fastapi", "pydantic", "uvicorn"})), "remote": ("pseudonymize.backends.remote", frozenset({"httpx"})), "html": ("pseudonymize.html_xml", frozenset()), "office": ("pseudonymize.inspection.office", frozenset({"docx", "openpyxl"})), diff --git a/scripts/verify_release.py b/scripts/verify_release.py index 8b21d44..847f1f8 100644 --- a/scripts/verify_release.py +++ b/scripts/verify_release.py @@ -23,7 +23,9 @@ } EXPECTED_DEVELOPMENT_CLASSIFIER = "Development Status :: 5 - Production/Stable" EXPECTED_BASE_REQUIREMENTS: frozenset[str] = frozenset() -EXPECTED_EXTRAS: frozenset[str] = frozenset({"ml", "office", "pdf", "ocr", "remote", "html"}) +EXPECTED_EXTRAS: frozenset[str] = frozenset( + {"ml", "office", "pdf", "ocr", "remote", "html", "service"} +) REQUIRED_SDIST_FILES = frozenset( { "CHANGELOG.md", diff --git a/tests/integration/test_release_verifier.py b/tests/integration/test_release_verifier.py index 19aa4c3..5647aba 100644 --- a/tests/integration/test_release_verifier.py +++ b/tests/integration/test_release_verifier.py @@ -1,6 +1,7 @@ import email.message import io import tarfile +import tomllib import zipfile from pathlib import Path @@ -171,3 +172,15 @@ def test_verify_quality_gate_report_missing_file(tmp_path: Path) -> None: missing = tmp_path / "nonexistent.json" with pytest.raises(ValueError, match="quality gate report not found"): verify_quality_gate_report(missing) + + +def test_declared_extras_have_release_and_install_audits() -> None: + from scripts.audit_extras import DOCUMENTED_EXTRAS + from scripts.audit_install import EXTRA_CHECKS + + project_file = Path(__file__).resolve().parents[2] / "pyproject.toml" + with project_file.open("rb") as stream: + declared = set(tomllib.load(stream)["project"]["optional-dependencies"]) + assert declared == EXPECTED_EXTRAS + assert declared == set(DOCUMENTED_EXTRAS) + assert declared == set(EXTRA_CHECKS)