Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 5 additions & 5 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -20,15 +20,15 @@ jobs:
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: "3.12"
python-version: "3.11"

- name: Install uv
uses: astral-sh/setup-uv@v5
with:
enable-cache: true

- name: Install dependencies
run: uv sync --extra dev
run: uv sync --all-extras

- name: Run ty type checking
run: uv run ty check
Expand Down Expand Up @@ -86,7 +86,7 @@ jobs:
# Patch coverage (new/changed lines ≥ 80%) is enforced by Codecov below.
run: |
EXTRA_ARGS=""
if [ "${{ matrix.python-version }}" = "3.12" ]; then
if [ "${{ matrix.python-version }}" = "3.11" ]; then
EXTRA_ARGS="--cov=src/risk_assessment --cov-report=xml --cov-report=html --cov-report=term-missing"
fi
uv run pytest tests/ -v --tb=short --maxfail=5 \
Expand All @@ -102,7 +102,7 @@ jobs:
retention-days: 7

- name: Upload coverage to Codecov
if: matrix.python-version == '3.12'
if: matrix.python-version == '3.11'
uses: codecov/codecov-action@v5
with:
token: ${{ secrets.CODECOV_TOKEN }}
Expand All @@ -111,7 +111,7 @@ jobs:
verbose: true

- name: Upload HTML coverage report as artifact
if: always() && matrix.python-version == '3.12'
if: always() && matrix.python-version == '3.11'
uses: actions/upload-artifact@v4
with:
name: coverage-html-report
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,7 @@ dependencies = [
"word2number==1.1",

# To fix dependabot
"setuptools>=83.0.0",
# "setuptools>=83.0.0",
]
[project.optional-dependencies]
rest = [
Expand Down
5 changes: 2 additions & 3 deletions src/risk_assessment/anonymization/mondrian.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@

from __future__ import annotations

from contextlib import suppress
from dataclasses import dataclass
from enum import Enum, auto
from typing import Any
Expand Down Expand Up @@ -112,12 +113,10 @@ def _find_common_ancestor(values: npt.NDArray[Any], hierarchy: GeneralizationHie
node = node.parent
else:
while node is not None:
try:
with suppress(ValueError):
index = ancestors.index(node)
ancestors = ancestors[index:]
break
except ValueError:
pass
node = node.parent

if ancestors is None:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -148,7 +148,7 @@ def _check_constraints(self, dataset: DataFrame) -> bool:
return True


@dataclass(eq=True)
@dataclass
class LatticeNode:
"""A single node in the generalization lattice.

Expand Down Expand Up @@ -189,6 +189,12 @@ def __str__(self) -> str:
def __hash__(self) -> int:
return hash(str(self.values))

def __eq__(self, other: Any) -> bool:
if isinstance(other, LatticeNode):
return self.__hash__() == other.__hash__()

return False


def _calculate_product(levels: list[list[int]]) -> list[list[int]]:
product_result: list[list[int]] = []
Expand Down
3 changes: 2 additions & 1 deletion src/risk_assessment/classification/unstructured/nltk.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,8 @@


class NLTKPoSTagger(EntityExtractor):
def __init__(self, tokenizer: TokenizerI, tagger: TaggerI) -> None:
def __init__(self, tokenizer: TokenizerI, tagger: TaggerI, type_mapping: dict[str, str] | None = None) -> None:
super().__init__(type_mapping if type_mapping is not None else {})
self.tokenizer = tokenizer
self.tagger = tagger

Expand Down
6 changes: 3 additions & 3 deletions src/risk_assessment/metrics/informationloss/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -199,8 +199,8 @@ def _get_loss_categorical(value: Any, column_information: ColumnInformation) ->
def _report_for_column_without_transformation_level(
original: Series | DataFrame, anonymized: Series | DataFrame, column_information: ColumnInformation
) -> float:
# if column_information.column_class != ColumnClass.CATEGORICAL:
# raise ValueError("Cannot process non categorical colums")
if column_information.column_class != ColumnClass.CATEGORICAL:
raise ValueError("Cannot process non categorical colums")

weight = column_information.weight
precision = 0.0
Expand Down Expand Up @@ -241,7 +241,7 @@ def _categorical_precision_report_per_quasi_column(
if c_i.column_type == ColumnType.QUASI:
report: float = _report_for_column(
original.iloc[:, index], anonymized.iloc[:, index], column_information[index], transformation_levels
) # for index, c_i in enumerate(column_information) if c_i.column_type == ColumnType.QUASI
)
results.append(report)

return results
Expand Down
1 change: 1 addition & 0 deletions src/risk_assessment/readi/text_tokenizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -158,6 +158,7 @@ def __init__(self, model_name: str = "FacebookAI/roberta-base", device: str = "c
)

def span_tokenize(self, text: str) -> list[tuple[int, int]]:
assert self.tokenizer is not None
tokenized_text = self.tokenizer(text)
num_of_tokens = len(tokenized_text["input_ids"])
token_spans: list[tuple[int, int]] = []
Expand Down
2 changes: 0 additions & 2 deletions tests/anonymization/test_anonymity_checker.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,6 @@
import importlib.resources
import math

import pandas as pd
import pytest

from risk_assessment.anonymization import KAnonymity, PrivacyConstraint
from risk_assessment.anonymization.optimal_lattice_anonymization import AnonymityChecker, LatticeNode
Expand Down
1 change: 0 additions & 1 deletion tests/anonymization/test_privacy_constraints.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,6 @@
from risk_assessment.anonymization import DistinctLDiversity, EntropyLDiversity, KAnonymity, TCloseness
from risk_assessment.metrics.informationloss import ColumnClass, ColumnInformation, ColumnType
from risk_assessment.utility import extract_histograms
from risk_assessment.utility.hierarchy import MaterializedHierarchy


def test_kanonymity():
Expand Down
9 changes: 6 additions & 3 deletions tests/classification/unstructured/test_presidio.py
Original file line number Diff line number Diff line change
Expand Up @@ -76,7 +76,10 @@ def _make_extractor(**kwargs) -> PresidioEntityExtractor:
return PresidioEntityExtractor(type_mapping=TYPE_MAPPING, **kwargs)


def _stub_result(entity_type: str, start: int, end: int, score: float = 0.85) -> _RecognizerResult:
RecognizerResult = _RecognizerResult


def _stub_result(entity_type: str, start: int, end: int, score: float = 0.85) -> RecognizerResult:
return _RecognizerResult(entity_type=entity_type, start=start, end=end, score=score)


Expand Down Expand Up @@ -255,7 +258,7 @@ def test_empty_type_mapping_returns_raw_types():
def test_build_entity_returns_entity_dataclass():
extractor = _make_extractor()
result = _stub_result("PERSON", 3, 11, score=0.88)
entity = extractor._build_entity(result)
entity = extractor._build_entity(result) # ty: ignore[invalid-argument-type]
assert isinstance(entity, Entity)
assert entity.start == 3
assert entity.end == 11
Expand All @@ -267,7 +270,7 @@ def test_build_entity_returns_entity_dataclass():
def test_build_entity_score_zero():
extractor = _make_extractor()
result = _stub_result("PERSON", 0, 4, score=0.0)
entity = extractor._build_entity(result)
entity = extractor._build_entity(result) # ty: ignore[invalid-argument-type]
assert entity.confidence == pytest.approx(0.0)


Expand Down
Loading
Loading