diff --git a/CHANGELOG.md b/CHANGELOG.md index 206e279..972f5d1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -69,3 +69,8 @@ This file records what changes **in the product** – process and session state - Database roles `app_owner` (migrations) and `app_rw` (runtime, no RLS bypass); schema `app`. - Verify commands `pnpm verify:changed`, `pnpm verify`, `pnpm verify:full`; CI runs integration tests against real PostgreSQL + SeaweedFS. + +### Fixed +- AI verifier: a unit quoted together with the neighbouring table cell (e.g. `60 | Stk.`) is now + confirmed as `found` when one cell of the quote is exactly the unit; quotes that differ from the + source in real characters are still rejected (#50). diff --git a/services/ai/README.md b/services/ai/README.md index b0aa2b4..cea9c4a 100644 --- a/services/ai/README.md +++ b/services/ai/README.md @@ -373,7 +373,7 @@ the legitimate value (`tests/test_evals_run.py`). The verbatim-quote limitation grounding proves provenance, not intent. The eval gate catches it instead – a test replays a model that quotes the injected sentence and asserts the gate fails at any threshold. -**Baseline** (`evals/baseline.json`, replay of the hand-written responses, 2026-09-23): +**Baseline** (`evals/baseline.json`, replay of the hand-written responses, 2026-09-23, `line_items.unit` updated 2026-09-24 in #50): | Key field | acc | miss P | miss R | grounding | false-found | |---|---|---|---|---|---| @@ -385,13 +385,18 @@ that quotes the injected sentence and asserts the gate fails at any threshold. | additional_requirements | 100 | 100 | 87.5 | 100 | 0 | | line_items.description | 96.55 | 0 | – | 100 | 0 | | line_items.quantity | 89.66 | 0 | – | 92.86 | 0 | -| line_items.unit | 86.21 | 0 | – | 89.29 | 0 | +| line_items.unit | 96.55 | 0 | – | 100 | 0 | | line_items.material | 89.29 | 0 | 0 | 96.43 | 3.85 | | line_items.dimensions | 92.86 | 50 | 100 | 96.3 | 0 | -These numbers describe the hand-written responses, not real model quality. Finding from `t03`: a -quote that spans a pipe-table cell border (`60 | Stk.`) fails the unit check, because a unit -counts only directly after a number; the verifier is unchanged here (open point). +These numbers describe the hand-written responses, not real model quality. Finding from `t03` +(fixed in #50, baseline updated 2026-09-24): a quote that spans a pipe-table cell border +(`60 | Stk.`) failed the unit check, because a unit counted only directly after a number or as +the whole quote. A known unit now also counts when one cell of the quote (split on `|` and tabs) is +exactly the unit and no other cell names a different unit (a whole row like `60 | Stk. | 12 | kg` +proves neither); the quote-in-segment check is unchanged. Known limit: the verifier does not know +table columns, so a quote of only the wrong column's cell (`12 | kg`) still passes – as a bare `kg` +quote always did. ## Verified facts (2026-09-22, in this environment) diff --git a/services/ai/evals/baseline.json b/services/ai/evals/baseline.json index 57106d4..c30c832 100644 --- a/services/ai/evals/baseline.json +++ b/services/ai/evals/baseline.json @@ -75,10 +75,10 @@ "false_found_rate": 0.0 }, "line_items.unit": { - "found_accuracy": 86.21, + "found_accuracy": 96.55, "missing_precision": 0.0, "missing_recall": null, - "grounding_pass_rate": 89.29, + "grounding_pass_rate": 100.0, "false_found_rate": 0.0 }, "line_items.material": { diff --git a/services/ai/src/requestflow_ai/grounding/values.py b/services/ai/src/requestflow_ai/grounding/values.py index 9a72bd3..c1dd3b2 100644 --- a/services/ai/src/requestflow_ai/grounding/values.py +++ b/services/ai/src/requestflow_ai/grounding/values.py @@ -8,7 +8,8 @@ only accepted as a week value (returned as ``KW 42`` / ``KW 42/2026``) and flagged, so the verifier caps it at ``uncertain``. A date computed from a week is not in the quote and is rejected. Units: a small canonical set (``mm``, ``cm``, ``m``, ``kg``, ``t``, ``pcs``) with German and -English spellings (``Stk.``, ``Stück``, ``Meter``, ...); unknown units are checked as text and +English spellings (``Stk.``, ``Stück``, ``Meter``, ...); a known unit must follow a number or fill a +whole table cell of the quote (cells split on ``|`` and tabs); unknown units are checked as text and returned trimmed. E-mail: case-insensitive match on address boundaries; returned lowercased. Phone: the digits (with a leading ``+``) must equal one phone-like number in the quote; returned @@ -66,9 +67,11 @@ for canonical, spellings in _UNIT_SPELLINGS.items() for spelling in spellings } -# A known unit counts only right after a number ("250mm", "1.250 Stk.") or when the whole quote is -# the unit (a table cell). A bare "St 37-2" (steel grade) or "t=5" (thickness) is not a unit. +# A known unit counts only right after a number ("250mm", "1.250 Stk.") or when a whole table cell +# of the quote is the unit ("Stk.", "60 | Stk.", "Stk.\tKugelhahn"; cells split on "|" and tabs, +# #50). A bare "St 37-2" (steel grade) or "t=5" (thickness) is not a unit. _UNIT_AFTER_NUMBER = re.compile(r"\d[\s\u00a0]*(?P[^\W\d_]+)\.?(?![^\W_])") +_CELL_BORDER = re.compile(r"[|\t]") _EMAIL_SHAPE = re.compile(r"[^\s@<>]+@[^\s@<>]+\.[^\s@<>]+") # A phone-like number: digits with spaces, "/", "-" or parentheses between them. No dots: a date @@ -251,7 +254,12 @@ def _check_unit(value: str, quote: str) -> ValueCheck: canonical = canonical_unit(value) if canonical is None: return _check_text(value, quote) - if canonical_unit(quote.strip()) == canonical: + # Exact match per cell (the whole quote is one cell when it has no border): no fuzziness. + # A quote whose cells name different units (a whole row like `60 | Stk. | 12 | kg`) is + # ambiguous: it does not prove which unit belongs to the item, so no cell counts (#50 review). + cells = (canonical_unit(cell) for cell in _CELL_BORDER.split(quote)) + cell_units = {unit for unit in cells if unit is not None} + if cell_units == {canonical}: return ValueCheck(ok=True, normalized=canonical) for match in _UNIT_AFTER_NUMBER.finditer(quote): if canonical_unit(match.group("unit")) == canonical: diff --git a/services/ai/tests/test_grounding_values.py b/services/ai/tests/test_grounding_values.py index dacb9e6..f413a87 100644 --- a/services/ai/tests/test_grounding_values.py +++ b/services/ai/tests/test_grounding_values.py @@ -233,6 +233,10 @@ def test_unknown_unit_has_no_canonical_form(raw: str) -> None: # A table cell holding only the unit. ("Stk.", "Stk.", "pcs"), ("kg", " kg ", "kg"), + # A quote across a table cell border: one cell of it holds only the unit (#50). + ("Stk.", "60 | Stk.", "pcs"), + ("Stk.", "Stk. | Kugelhahn", "pcs"), + ("kg", "kg\tStahlblech", "kg"), ], ) def test_unit_value_is_checked_and_canonicalised(value: str, quote: str, normalized: str) -> None: @@ -257,6 +261,12 @@ def test_unit_value_is_checked_and_canonicalised(value: str, quote: str, normali ("Stk.", "Werkstoff St 52"), ("t", "Blech t=5"), ("t", "t 5 mm"), + # Across a cell border only a cell that is exactly the unit counts (#50). + ("pcs", "1 | St 37-2"), + ("Stk.", "60 | Stückliste"), + ("t", "5 | t=5"), + ("pcs", "| 60 | Stk. | 12 | kg |"), + ("kg", "| 60 | Stk. | 12 | kg |"), ], ) def test_unit_not_in_quote_is_not_ok(value: str, quote: str) -> None: diff --git a/services/ai/tests/test_grounding_verifier.py b/services/ai/tests/test_grounding_verifier.py index 01f2f20..fa9f188 100644 --- a/services/ai/tests/test_grounding_verifier.py +++ b/services/ai/tests/test_grounding_verifier.py @@ -458,6 +458,59 @@ def test_line_item_quantity_is_never_promoted_from_uncertain() -> None: assert result.fields["quantity"].value == "1250" +TABLE_ROW_SEGMENTS: dict[str, Segment] = { + s.id: s + for s in [ + body_segment( + "eml-l7", "| 1 | 60 | Stk. | Kugelhahn | 1.4408 | DN80 |" + ), + body_segment( + "eml-l8", "| 2 | 25 | Stck. | Rueckschlagventil | 1.0619 | DN50 |" + ), + ] +} + + +def test_unit_quoted_across_a_table_cell_border_is_found() -> None: + # Issue #50 (eval t03): the model quotes quantity and unit cell together. + extraction = extraction_with_items( + item( + quantity=field("60", "found", "eml-l7", "| 60 | Stk."), + unit=field("Stk.", "found", "eml-l7", "60 | Stk."), + ), + ) + (result,) = verify_line_items(extraction, list(TABLE_ROW_SEGMENTS.values())) + assert result.fields["quantity"].status == "found" + assert result.fields["unit"].status == "found" + assert result.fields["unit"].value == "pcs" + + +@pytest.mark.parametrize( + "quote", + [ + # Real characters dropped ("Stck." -> "Stk."): not a separator difference. + "25 | Stk.", + # Cell border dropped as well as padding: still not the segment text. + "25 Stk.", + ], +) +def test_cross_cell_quote_that_drops_real_characters_is_unverified(quote: str) -> None: + extraction = extraction_with_items(item(unit=field("Stk.", "found", "eml-l8", quote))) + (result,) = verify_line_items(extraction, list(TABLE_ROW_SEGMENTS.values())) + assert result.fields["unit"].status == "unverified" + assert result.fields["unit"].reason == "quote_not_in_segment" + + +def test_cross_cell_quote_whose_unit_cell_is_not_a_unit_is_unverified() -> None: + # The quote is in the segment, but no cell of it is the unit: "Kugelhahn" proves no "Stk.". + extraction = extraction_with_items( + item(unit=field("Stk.", "found", "eml-l7", "Kugelhahn | 1.4408")) + ) + (result,) = verify_line_items(extraction, list(TABLE_ROW_SEGMENTS.values())) + assert result.fields["unit"].status == "unverified" + assert result.fields["unit"].reason == "value_not_in_quote" + + def test_no_line_items_is_an_empty_list() -> None: assert verify_line_items(extraction_with_items(), list(V2_SEGMENTS.values())) == []