Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -69,3 +69,8 @@ This file records what changes **in the product** – process and session state
- Database roles `app_owner` (migrations) and `app_rw` (runtime, no RLS bypass); schema `app`.
- Verify commands `pnpm verify:changed`, `pnpm verify`, `pnpm verify:full`; CI runs integration
tests against real PostgreSQL + SeaweedFS.

### Fixed
- AI verifier: a unit quoted together with the neighbouring table cell (e.g. `60 | Stk.`) is now
confirmed as `found` when one cell of the quote is exactly the unit; quotes that differ from the
source in real characters are still rejected (#50).
15 changes: 10 additions & 5 deletions services/ai/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -373,7 +373,7 @@ the legitimate value (`tests/test_evals_run.py`). The verbatim-quote limitation
grounding proves provenance, not intent. The eval gate catches it instead – a test replays a model
that quotes the injected sentence and asserts the gate fails at any threshold.

**Baseline** (`evals/baseline.json`, replay of the hand-written responses, 2026-09-23):
**Baseline** (`evals/baseline.json`, replay of the hand-written responses, 2026-09-23, `line_items.unit` updated 2026-09-24 in #50):

| Key field | acc | miss P | miss R | grounding | false-found |
|---|---|---|---|---|---|
Expand All @@ -385,13 +385,18 @@ that quotes the injected sentence and asserts the gate fails at any threshold.
| additional_requirements | 100 | 100 | 87.5 | 100 | 0 |
| line_items.description | 96.55 | 0 | – | 100 | 0 |
| line_items.quantity | 89.66 | 0 | – | 92.86 | 0 |
| line_items.unit | 86.21 | 0 | – | 89.29 | 0 |
| line_items.unit | 96.55 | 0 | – | 100 | 0 |
| line_items.material | 89.29 | 0 | 0 | 96.43 | 3.85 |
| line_items.dimensions | 92.86 | 50 | 100 | 96.3 | 0 |

These numbers describe the hand-written responses, not real model quality. Finding from `t03`: a
quote that spans a pipe-table cell border (`60 | Stk.`) fails the unit check, because a unit
counts only directly after a number; the verifier is unchanged here (open point).
These numbers describe the hand-written responses, not real model quality. Finding from `t03`
(fixed in #50, baseline updated 2026-09-24): a quote that spans a pipe-table cell border
(`60 | Stk.`) failed the unit check, because a unit counted only directly after a number or as
the whole quote. A known unit now also counts when one cell of the quote (split on `|` and tabs) is
exactly the unit and no other cell names a different unit (a whole row like `60 | Stk. | 12 | kg`
proves neither); the quote-in-segment check is unchanged. Known limit: the verifier does not know
table columns, so a quote of only the wrong column's cell (`12 | kg`) still passes – as a bare `kg`
quote always did.

## Verified facts (2026-09-22, in this environment)

Expand Down
4 changes: 2 additions & 2 deletions services/ai/evals/baseline.json
Original file line number Diff line number Diff line change
Expand Up @@ -75,10 +75,10 @@
"false_found_rate": 0.0
},
"line_items.unit": {
"found_accuracy": 86.21,
"found_accuracy": 96.55,
"missing_precision": 0.0,
"missing_recall": null,
"grounding_pass_rate": 89.29,
"grounding_pass_rate": 100.0,
"false_found_rate": 0.0
},
"line_items.material": {
Expand Down
16 changes: 12 additions & 4 deletions services/ai/src/requestflow_ai/grounding/values.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,8 @@
only accepted as a week value (returned as ``KW 42`` / ``KW 42/2026``) and flagged, so the verifier
caps it at ``uncertain``. A date computed from a week is not in the quote and is rejected.
Units: a small canonical set (``mm``, ``cm``, ``m``, ``kg``, ``t``, ``pcs``) with German and
English spellings (``Stk.``, ``Stück``, ``Meter``, ...); unknown units are checked as text and
English spellings (``Stk.``, ``Stück``, ``Meter``, ...); a known unit must follow a number or fill a
whole table cell of the quote (cells split on ``|`` and tabs); unknown units are checked as text and
returned trimmed.
E-mail: case-insensitive match on address boundaries; returned lowercased.
Phone: the digits (with a leading ``+``) must equal one phone-like number in the quote; returned
Expand Down Expand Up @@ -66,9 +67,11 @@
for canonical, spellings in _UNIT_SPELLINGS.items()
for spelling in spellings
}
# A known unit counts only right after a number ("250mm", "1.250 Stk.") or when the whole quote is
# the unit (a table cell). A bare "St 37-2" (steel grade) or "t=5" (thickness) is not a unit.
# A known unit counts only right after a number ("250mm", "1.250 Stk.") or when a whole table cell
# of the quote is the unit ("Stk.", "60 | Stk.", "Stk.\tKugelhahn"; cells split on "|" and tabs,
# #50). A bare "St 37-2" (steel grade) or "t=5" (thickness) is not a unit.
_UNIT_AFTER_NUMBER = re.compile(r"\d[\s\u00a0]*(?P<unit>[^\W\d_]+)\.?(?![^\W_])")
_CELL_BORDER = re.compile(r"[|\t]")

_EMAIL_SHAPE = re.compile(r"[^\s@<>]+@[^\s@<>]+\.[^\s@<>]+")
# A phone-like number: digits with spaces, "/", "-" or parentheses between them. No dots: a date
Expand Down Expand Up @@ -251,7 +254,12 @@ def _check_unit(value: str, quote: str) -> ValueCheck:
canonical = canonical_unit(value)
if canonical is None:
return _check_text(value, quote)
if canonical_unit(quote.strip()) == canonical:
# Exact match per cell (the whole quote is one cell when it has no border): no fuzziness.
# A quote whose cells name different units (a whole row like `60 | Stk. | 12 | kg`) is
# ambiguous: it does not prove which unit belongs to the item, so no cell counts (#50 review).
cells = (canonical_unit(cell) for cell in _CELL_BORDER.split(quote))
cell_units = {unit for unit in cells if unit is not None}
if cell_units == {canonical}:
return ValueCheck(ok=True, normalized=canonical)
for match in _UNIT_AFTER_NUMBER.finditer(quote):
if canonical_unit(match.group("unit")) == canonical:
Expand Down
10 changes: 10 additions & 0 deletions services/ai/tests/test_grounding_values.py
Original file line number Diff line number Diff line change
Expand Up @@ -233,6 +233,10 @@ def test_unknown_unit_has_no_canonical_form(raw: str) -> None:
# A table cell holding only the unit.
("Stk.", "Stk.", "pcs"),
("kg", " kg ", "kg"),
# A quote across a table cell border: one cell of it holds only the unit (#50).
("Stk.", "60 | Stk.", "pcs"),
("Stk.", "Stk. | Kugelhahn", "pcs"),
("kg", "kg\tStahlblech", "kg"),
],
)
def test_unit_value_is_checked_and_canonicalised(value: str, quote: str, normalized: str) -> None:
Expand All @@ -257,6 +261,12 @@ def test_unit_value_is_checked_and_canonicalised(value: str, quote: str, normali
("Stk.", "Werkstoff St 52"),
("t", "Blech t=5"),
("t", "t 5 mm"),
# Across a cell border only a cell that is exactly the unit counts (#50).
("pcs", "1 | St 37-2"),
("Stk.", "60 | Stückliste"),
("t", "5 | t=5"),
("pcs", "| 60 | Stk. | 12 | kg |"),
("kg", "| 60 | Stk. | 12 | kg |"),
],
)
def test_unit_not_in_quote_is_not_ok(value: str, quote: str) -> None:
Expand Down
53 changes: 53 additions & 0 deletions services/ai/tests/test_grounding_verifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -458,6 +458,59 @@ def test_line_item_quantity_is_never_promoted_from_uncertain() -> None:
assert result.fields["quantity"].value == "1250"


TABLE_ROW_SEGMENTS: dict[str, Segment] = {
s.id: s
for s in [
body_segment(
"eml-l7", "| 1 | 60 | Stk. | Kugelhahn | 1.4408 | DN80 |"
),
body_segment(
"eml-l8", "| 2 | 25 | Stck. | Rueckschlagventil | 1.0619 | DN50 |"
),
]
}


def test_unit_quoted_across_a_table_cell_border_is_found() -> None:
# Issue #50 (eval t03): the model quotes quantity and unit cell together.
extraction = extraction_with_items(
item(
quantity=field("60", "found", "eml-l7", "| 60 | Stk."),
unit=field("Stk.", "found", "eml-l7", "60 | Stk."),
),
)
(result,) = verify_line_items(extraction, list(TABLE_ROW_SEGMENTS.values()))
assert result.fields["quantity"].status == "found"
assert result.fields["unit"].status == "found"
assert result.fields["unit"].value == "pcs"


@pytest.mark.parametrize(
"quote",
[
# Real characters dropped ("Stck." -> "Stk."): not a separator difference.
"25 | Stk.",
# Cell border dropped as well as padding: still not the segment text.
"25 Stk.",
],
)
def test_cross_cell_quote_that_drops_real_characters_is_unverified(quote: str) -> None:
extraction = extraction_with_items(item(unit=field("Stk.", "found", "eml-l8", quote)))
(result,) = verify_line_items(extraction, list(TABLE_ROW_SEGMENTS.values()))
assert result.fields["unit"].status == "unverified"
assert result.fields["unit"].reason == "quote_not_in_segment"


def test_cross_cell_quote_whose_unit_cell_is_not_a_unit_is_unverified() -> None:
# The quote is in the segment, but no cell of it is the unit: "Kugelhahn" proves no "Stk.".
extraction = extraction_with_items(
item(unit=field("Stk.", "found", "eml-l7", "Kugelhahn | 1.4408"))
)
(result,) = verify_line_items(extraction, list(TABLE_ROW_SEGMENTS.values()))
assert result.fields["unit"].status == "unverified"
assert result.fields["unit"].reason == "value_not_in_quote"


def test_no_line_items_is_an_empty_list() -> None:
assert verify_line_items(extraction_with_items(), list(V2_SEGMENTS.values())) == []

Expand Down
Loading