diff --git a/CHANGELOG.md b/CHANGELOG.md index 8600206..6612b48 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,11 @@ All notable changes follow Keep a Changelog and Semantic Versioning. ## [Unreleased] +### Fixed + +- Invalidate Bloom filter membership caches after insertion so earlier misses cannot hide + newly added entries. Validate filter parameters and allocate at least one bit. + ## [1.34.0] - 2026-09-30 ### Added diff --git a/HANDOVER.md b/HANDOVER.md index 091c165..955ccf4 100644 --- a/HANDOVER.md +++ b/HANDOVER.md @@ -399,3 +399,28 @@ verification unless they are intentionally retained as a user-approved release a - Push only when explicitly requested. - A version is published only after its matching tag, successful release workflow, PyPI artifact, and GitHub release exist. Until then, keep changes under `[Unreleased]`. + + +## Bloom filter correctness review (2026-10-02) + +The mutation regression was reproduced before fixing it: a cached negative lookup +remained negative after insertion. Small filters could also allocate zero bits. +The change clears membership cache entries after insertion, validates parameters, +and rounds the bit count upward. No dependency or detection threshold changes. + +Observed verification on Python 3.14.2 with the frozen lockfile: + +- Focused Bloom and gazetteer tests: 11 passed. +- `uv run --frozen pre-commit run --all-files`: passed. +- `uv run --frozen mypy`: passed, 155 source files. +- `uv run --frozen mkdocs build --strict`: passed. +- `uv build` and `uv run --frozen python scripts/verify_release.py`: passed. +- `uv run --frozen pytest --ignore=tests/unit/backends/test_onnx.py + --ignore=tests/unit/backends/test_bio_decoder.py --no-cov`: 493 passed, + 1 skipped because Tesseract was unavailable. +- The full suite terminated with exit 137 during ONNX tests. Full coverage and the + cross-platform matrix remain unverified locally; the partial run is not a + substitute for either gate. + +Reusable lesson assessment: mutation must invalidate cached negative membership +results. The regression test records that invariant in its owning implementation. diff --git a/src/pseudonymize/memory/bloom.py b/src/pseudonymize/memory/bloom.py index d394e61..a790fa5 100644 --- a/src/pseudonymize/memory/bloom.py +++ b/src/pseudonymize/memory/bloom.py @@ -8,6 +8,10 @@ class BloomFilter: __slots__ = ("_bit_array", "_contains_cached", "_hash_count", "_size") def __init__(self, capacity: int, error_rate: float = 0.001) -> None: + if capacity <= 0: + raise ValueError("capacity must be positive") + if not 0 < error_rate < 1: + raise ValueError("error_rate must be between 0 and 1") self._size = self._get_size(capacity, error_rate) self._hash_count = self._get_hash_count(self._size, capacity) self._bit_array = bytearray((self._size + 7) // 8) @@ -41,6 +45,9 @@ def add(self, item: str) -> None: digest = int.from_bytes(h.digest(), "big") % self._size self._bit_array[digest // 8] |= 1 << (digest % 8) + # Cached misses become stale whenever the bit array changes. + self._contains_cached.cache_clear() + def _raw_contains(self, item: str) -> bool: item_bytes = item.encode("utf-8") base = hashlib.sha256(item_bytes) @@ -58,7 +65,7 @@ def __contains__(self, item: str) -> bool: @staticmethod def _get_size(n: int, p: float) -> int: m = -(n * math.log(p)) / (math.log(2) ** 2) - return int(m) + return max(1, math.ceil(m)) @staticmethod def _get_hash_count(m: int, n: int) -> int: diff --git a/tests/unit/test_bloom.py b/tests/unit/test_bloom.py index ace3c63..3e45ad7 100644 --- a/tests/unit/test_bloom.py +++ b/tests/unit/test_bloom.py @@ -10,3 +10,39 @@ def test_bloom_filter() -> None: assert "Jonathan" not in bf assert "Doe" not in bf + + +def test_cached_miss_is_invalidated_after_add() -> None: + bf = BloomFilter(100) + for word in ("new-entry", "Élodie", "東京", ""): + assert word not in bf + bf.add(word) + assert word in bf + bf.add(word) + assert word in bf + + +def test_mutation_preserves_previous_members() -> None: + bf = BloomFilter.from_words(iter(["alpha", "beta"])) + assert "alpha" in bf + assert "gamma" not in bf + bf.add("gamma") + assert all(word in bf for word in ("alpha", "beta", "gamma")) + + +def test_small_filter_has_at_least_one_bit() -> None: + bf = BloomFilter(1, error_rate=0.99) + assert "entry" not in bf + bf.add("entry") + assert "entry" in bf + + +def test_invalid_filter_parameters() -> None: + import pytest + + for capacity in (0, -1): + with pytest.raises(ValueError, match="capacity must be positive"): + BloomFilter(capacity) + for rate in (0.0, -0.1, 1.0, 2.0, float("nan"), float("inf")): + with pytest.raises(ValueError, match="error_rate must be between 0 and 1"): + BloomFilter(10, rate)