From 7b06dcce531f59cf96665a5bcc53c8b51d1ef4eb Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 21:13:48 -0300 Subject: [PATCH 01/18] Carry examples and counterexamples on a rubric A choice option and a score level could only say what they cover. Callers want to show inputs that belong there and, for a choice, inputs that belong to some other option: measured against live Jev, a deliberately ambiguous ticket goes from a 0.51/0.49 coin flip on the wrong option to 0.91 on the right one once the options carry examples. The parts ride on the rubric string and are composed into it at the four places a rubric becomes wire text, so an option cannot reach the wire having quietly lost them. The string value stays the bare text, and a rubric with no parts renders to that text byte for byte, which is what leaves every 0.1.0 caller's request unchanged. They are flattened into the text rather than sent as the API's structured criteria because a score answer echoes its criteria back in `legend`, which is `dict[str, str]` here: objects would not deserialise, so a documented API feature would cost a breaking change to a field nothing reads, for no measured gain and about 11% more billed input tokens. `level(...)` takes no counterexamples: "not this option" means nothing on an ordered scale, where the level above or below is what an input that does not belong here scores. An `option(...)` written where a level belongs is a `ConfigError` rather than a clause silently dropped. --- src/guideme/__init__.py | 4 +- src/guideme/enums.py | 195 +++++++++++++++++++++++++++++++++++----- src/guideme/question.py | 14 ++- tests/test_enums.py | 48 +++++++++- tests/test_live.py | 45 ++++++++++ tests/test_rubrics.py | 158 ++++++++++++++++++++++++++++++++ tests/test_surface.py | 2 + tests/test_wire.py | 10 +++ 8 files changed, 451 insertions(+), 25 deletions(-) create mode 100644 tests/test_rubrics.py diff --git a/src/guideme/__init__.py b/src/guideme/__init__.py index 17cae35..7c34762 100644 --- a/src/guideme/__init__.py +++ b/src/guideme/__init__.py @@ -1,6 +1,6 @@ """Type-safe inline judgments from TypeSafe Jev.""" -from guideme.enums import Choice, Levels, fallback +from guideme.enums import Choice, Levels, fallback, level, option from guideme.errors import ( AuthError, ConfigError, @@ -57,7 +57,9 @@ "choose", "choose_among", "fallback", + "level", "noul", + "option", "score", "score_levels", ] diff --git a/src/guideme/enums.py b/src/guideme/enums.py index 9d86269..a47918d 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -1,11 +1,15 @@ """The two enum bases a caller declares their options and levels with. A member's name is its wire key and its value is its rubric, the sentence the -model reads. Docstrings on members are not the rubric; the value is. Both bases -validate at class-definition time, so a rubric that could not be asked is an -error where it is written rather than on the first request. +model reads. Docstrings on members are not the rubric; the value is. A rubric is +a bare string, or an `option(...)`, `level(...)` or `fallback(...)` that carries +examples beside the text and has them composed into it at the one point the +rubric becomes wire text. Both bases validate at class-definition time, so a +rubric that could not be asked is an error where it is written rather than on +the first request. """ +from collections.abc import Sequence from enum import Enum from typing import Self, final @@ -15,26 +19,165 @@ MIN_OPTIONS = 1 """Fewest options a choice may carry.""" +EXAMPLES = "Examples: " +"""Opens the clause naming inputs that belong to this option or level.""" + +NOT_THIS = "Not this option: " +"""Opens the clause naming inputs that belong to some other option.""" + +_UNSET: tuple[str, ...] = () +"""The `examples` and `counterexamples` default, and how an empty sequence written on +purpose is told from an argument left out: CPython hands out one empty tuple, so the +`examples=[]` this refuses is never this object.""" + @final -class _Fallback(str): - """A rubric marked as the member to fall back to when the policy says unsure. +class _Rubric(str): # noqa: SLOT000 -- a str subclass may not carry non-empty __slots__ + """A rubric held as its parts: the text, its examples, its counterexamples. + + A `str` subclass whose `str` value is the bare text, so a member's value reads + exactly like a bare string's and nothing downstream meets a new type. The parts + ride along as attributes until `render` composes them. `str` is a variable-length + built-in, so its subclasses cannot have slots and the parts live in the instance + dictionary; they are set in `__new__` because a `str` is immutable and carries its + text from there. + """ - A `str` subclass, so the member's value is its rubric exactly like every - other member's and the marking lives in the type rather than in a second - attribute the caller could see. + examples: tuple[str, ...] = () + counterexamples: tuple[str, ...] = () + is_fallback: bool = False + + def __new__( + cls, + rubric: str, + examples: tuple[str, ...], + counterexamples: tuple[str, ...], + *, + is_fallback: bool, + ) -> Self: + """Carry the parts on the string without changing what the string says.""" + self = super().__new__(cls, rubric) + self.examples = examples + self.counterexamples = counterexamples + self.is_fallback = is_fallback + return self + + +def _items(what: str, items: Sequence[str], named: str) -> tuple[str, ...]: + """Check one clause's items: written means non-empty, each says something, no repeats.""" + if items is _UNSET: + return () + if not items: + detail = f"{what!r}: {named}= is empty; a clause written on purpose must say something" + raise ConfigError(detail) + values = tuple(items) + blank = [item for item in values if not item.strip()] + if blank: + detail = f"{what!r}: every entry in {named} must say something, got {blank[0]!r}" + raise ConfigError(detail) + if len(set(values)) != len(values): + detail = f"{what!r}: {named} repeats an entry; each must be distinct" + raise ConfigError(detail) + return values + + +def _rubric( + rubric: str, + examples: Sequence[str], + counterexamples: Sequence[str], + *, + is_fallback: bool, +) -> str: + """Check the parts and hold them on the rubric, unrendered.""" + if not rubric.strip(): + detail = f"a rubric must say something, got {rubric!r}" + raise ConfigError(detail) + return _Rubric( + rubric, + _items(rubric, examples, "examples"), + _items(rubric, counterexamples, "counterexamples"), + is_fallback=is_fallback, + ) + + +def option( + rubric: str, + *, + examples: Sequence[str] = _UNSET, + counterexamples: Sequence[str] = _UNSET, +) -> str: + """A choice option: what it covers, inputs that belong to it, inputs that do not. + + The value is still the rubric text; the examples are composed into it where the + rubric goes on the wire, so an option written as a bare string and one written as + `option("…")` send the same bytes. A string may be an example of one option and a + counterexample of another: that is how two confusable options are told apart. + + An empty rubric, an empty entry, an `examples=[]` written out, or a repeat within + one clause is a `ConfigError` where the option is written. """ + return _rubric(rubric, examples, counterexamples, is_fallback=False) + - __slots__ = () +def level(rubric: str, *, examples: Sequence[str] = _UNSET) -> str: + """A score level: what it means, and inputs that score here. + + An example listed under a level is the statement that such an input scores that + level; its place in the ordered scale is what says which score. There are no + counterexamples: "not this option" means nothing on an ordered scale, so the + level below or above is what an input that does not belong here scores. + """ + return _rubric(rubric, examples, _UNSET, is_fallback=False) -def fallback(rubric: str) -> str: +def fallback( + rubric: str, + *, + examples: Sequence[str] = _UNSET, + counterexamples: Sequence[str] = _UNSET, +) -> str: """Mark the member to use when the policy says unsure. At most one per `Choice`. - A `.otherwise(...)` on the question beats it; with neither, an unsure answer - raises `UnsureError`. + An `option(...)` in every other respect, examples included. A `.otherwise(...)` on + the question beats it; with neither, an unsure answer raises `UnsureError`. """ - return _Fallback(rubric) + return _rubric(rubric, examples, counterexamples, is_fallback=True) + + +def render(rubric: str) -> str: + """Compose a rubric's parts into the text the model reads. + + Clauses are joined with a newline and the items of one with `"; "`, and the text + is used verbatim: never trimmed, never re-punctuated, so an example ending in `?` + does not become `Where is my refund?.`. A bare string, and a rubric carrying no + parts, come back byte for byte, which is what leaves an existing caller's request + unchanged. The rendered string is a cross-SDK contract item; `guideme-rust` + composes the same one in its derive macro. + """ + if not isinstance(rubric, _Rubric): + return rubric + if not rubric.examples and not rubric.counterexamples: + return str(rubric) + lines = [str(rubric)] + if rubric.examples: + lines.append(EXAMPLES + "; ".join(rubric.examples)) + if rubric.counterexamples: + lines.append(NOT_THIS + "; ".join(rubric.counterexamples)) + return "\n".join(lines) + + +def require_no_counterexamples(where: str, rubric: str) -> None: + """Refuse counterexamples on a level, rather than rendering or dropping them. + + `level(...)` does not offer them, so this is an `option(...)` written where a level + belongs. Silently losing what it carries is the failure this raises instead. + """ + if isinstance(rubric, _Rubric) and rubric.counterexamples: + detail = ( + f"{where}: counterexamples are not allowed on a level, which is a position on " + f"an ordered scale; use level(...), which takes examples only" + ) + raise ConfigError(detail) def _require_distinct(cls: type[Enum]) -> None: @@ -81,21 +224,29 @@ def __init_subclass__(cls) -> None: ) raise ConfigError(detail) _require_text(cls, "choice option") - marked = [member.name for member in cls if isinstance(member.value, _Fallback)] + marked = [ + member.name + for member in cls + if isinstance(member.value, _Rubric) and member.value.is_fallback + ] if len(marked) > 1: detail = f"{cls.__name__}: only one member may be marked fallback, got {marked}" raise ConfigError(detail) @classmethod def rubric(cls) -> tuple[tuple[str, str], ...]: - """`(key, rubric)` pairs in declaration order.""" - return tuple((member.name, member.value) for member in cls) + """`(key, rendered rubric)` pairs in declaration order. + + This is where an `option(...)`'s examples are composed into its text, so a + member cannot reach the wire having quietly lost them. + """ + return tuple((member.name, render(member.value)) for member in cls) @classmethod def fallback_member(cls) -> Self | None: """The `fallback(...)` member, if the rubric marks one.""" for member in cls: - if isinstance(member.value, _Fallback): + if isinstance(member.value, _Rubric) and member.value.is_fallback: return member return None @@ -131,12 +282,13 @@ def __init_subclass__(cls) -> None: raise ConfigError(detail) _require_text(cls, "level") for member in cls: - if isinstance(member.value, _Fallback): + if isinstance(member.value, _Rubric) and member.value.is_fallback: detail = ( f"{cls.__name__}.{member.name}: fallback(...) is not allowed on Levels; " f"use .otherwise(level) on the question" ) raise ConfigError(detail) + require_no_counterexamples(f"{cls.__name__}.{member.name}", member.value) @property def index(self) -> int: @@ -145,8 +297,11 @@ def index(self) -> int: @classmethod def levels(cls) -> tuple[str, ...]: - """Level descriptions, low to high.""" - return tuple(member.value for member in cls) + """Rendered level descriptions, low to high. + + This is where a `level(...)`'s examples are composed into its text. + """ + return tuple(render(member.value) for member in cls) @classmethod def from_index(cls, index: int) -> Self | None: diff --git a/src/guideme/question.py b/src/guideme/question.py index 1c9e8f5..53c8aab 100644 --- a/src/guideme/question.py +++ b/src/guideme/question.py @@ -10,7 +10,7 @@ from typing import Self, final from guideme._json import Json -from guideme.enums import Choice, Levels +from guideme.enums import Choice, Levels, render, require_no_counterexamples from guideme.errors import ConfigError, ProtocolError, UnsureError from guideme.policy import ( MAX_LEVELS, @@ -432,10 +432,13 @@ def choose_among(instructions: Json, options: Mapping[str, str | None]) -> Choic outside it is a `ConfigError` raised here, where the options are written, never later at `ask`. Keys cannot collide: a `Mapping` has already made them unique. + + A rubric is a string, `None`, or an `option(...)` carrying examples, which + are composed into it here: the same text a `Choice` member would send. """ return _choice( instructions, - tuple(options.items()), + tuple((key, None if text is None else render(text)) for key, text in options.items()), Key, lambda: None, ) @@ -468,6 +471,9 @@ def score_levels(instructions: Json, levels: Sequence[str]) -> ScoreQuestion[Ran Takes 2 to 10 levels, the same range a `Levels` enum takes; outside it is a `ConfigError`. + A level is a string or a `level(...)` carrying examples, which are composed + into it here: the same text a `Levels` member would send. + A `str` is a `Sequence[str]` of its own characters, so `score_levels("…", "abc")` would quietly ask about a three-letter scale. It is a `ConfigError` instead. """ @@ -476,4 +482,6 @@ def score_levels(instructions: Json, levels: Sequence[str]) -> ScoreQuestion[Ran f"levels must be a sequence of level descriptions, got a single {type(levels).__name__}" ) raise ConfigError(detail) - return _score(instructions, tuple(levels), Rank) + for index, text in enumerate(levels): + require_no_counterexamples(f"level {index}", text) + return _score(instructions, tuple(render(text) for text in levels), Rank) diff --git a/tests/test_enums.py b/tests/test_enums.py index 4565dfa..7b31320 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -6,7 +6,7 @@ from hypothesis import strategies as st from guideme import ConfigError -from guideme.enums import Choice, Levels, fallback +from guideme.enums import Choice, Levels, fallback, level, option def _compare(low: Levels, high: Levels) -> bool: @@ -71,6 +71,42 @@ class OneLevel(Levels): return OneLevel +def _a_blank_rubric() -> type[Choice]: + class Blank(Choice): + a = option(" ") + + return Blank + + +def _an_examples_clause_written_empty() -> type[Choice]: + class Empty(Choice): + a = option("Payments", examples=[]) + + return Empty + + +def _a_blank_example() -> type[Choice]: + class Blank(Choice): + a = option("Payments", examples=["My card was charged twice", " "]) + + return Blank + + +def _a_repeated_example() -> type[Choice]: + class Repeated(Choice): + a = option("Payments", examples=["Where is my refund?", "Where is my refund?"]) + + return Repeated + + +def _a_counterexample_on_a_level() -> type[Levels]: + class Marked(Levels): + cosmetic = option("No impact", counterexamples=["cannot log in"]) + blocking = level("No workaround exists") + + return Marked + + def _eleven_levels() -> type[Levels]: class ElevenLevels(Levels): l0 = "a" @@ -136,6 +172,11 @@ class Urgency(Levels): _duplicate_choice_rubric, _duplicate_level_rubric, _fallback_on_a_level, + _a_blank_rubric, + _an_examples_clause_written_empty, + _a_blank_example, + _a_repeated_example, + _a_counterexample_on_a_level, ] REFUSED_IDS = [ @@ -147,6 +188,11 @@ class Urgency(Levels): "two_options_with_the_same_rubric", "two_levels_with_the_same_rubric", "a_fallback_marker_on_a_levels", + "a_blank_rubric", + "an_examples_clause_written_empty", + "a_blank_example", + "a_repeated_example", + "a_counterexample_on_a_level", ] diff --git a/tests/test_live.py b/tests/test_live.py index 01290f5..2c99894 100644 --- a/tests/test_live.py +++ b/tests/test_live.py @@ -7,11 +7,14 @@ AsyncGuide, Choice, Guide, + Key, Levels, Scored, choose, + choose_among, fallback, noul, + option, score, ) from guideme.question import Question @@ -110,3 +113,45 @@ async def run() -> tuple[bool, Department, Scored[Frustration], dict[str, bool]] urgent, dept, mood, flags = asyncio.run(run()) check(urgent, dept, mood, flags) assert billed(spans) > 0 + + +AMBIGUOUS = "About those shoes - what is the situation with the money side of things?" +"""A ticket two options both half fit, where the examples are what tells them apart.""" + +POLICY = "The shop's rules" +STATUS = "One customer's open case" + +POLICY_EXAMPLES = ["How many days do I have to send it back?", "Can I return a sale item?"] +STATUS_EXAMPLES = ["Where is my refund?", "I posted the shoes back last week and heard nothing"] + +BARE = {"return_policy": POLICY, "return_status": STATUS} +"""The rubrics as bare strings. Deliberately vague: the ticket is near a coin flip.""" + +DESCRIBED = { + "return_policy": option(POLICY, examples=POLICY_EXAMPLES, counterexamples=[STATUS_EXAMPLES[1]]), + "return_status": option(STATUS, examples=STATUS_EXAMPLES, counterexamples=[POLICY_EXAMPLES[1]]), +} +"""The same two rubrics, with the examples that tell the two apart.""" + + +def _return_status(guide: Guide, options: dict[str, str]) -> float: + ranked = guide.ask( + choose_among("What is the customer asking about?", options).detail(), AMBIGUOUS + ) + return dict(ranked.probabilities)[Key("return_status")] + + +def test_examples_move_the_distribution_towards_the_option_they_describe() -> None: + """The invariant the feature exists for, not a number the model is not stable to. + + The rubric text is the same in both asks, so the examples are the only thing that + changed. Measured on 2026-09-21 against `jev-1.13.0`: 0.50 bare, 0.88 described, + over three runs each. + """ + guide = Guide.from_env() + try: + bare = _return_status(guide, BARE) + described = _return_status(guide, DESCRIBED) + finally: + guide.close() + assert described > bare diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py new file mode 100644 index 0000000..348d775 --- /dev/null +++ b/tests/test_rubrics.py @@ -0,0 +1,158 @@ +import pytest +from hypothesis import given +from hypothesis import strategies as st +from pytest_httpserver import HTTPServer + +from guideme import Choice, Levels, choose, fallback, level, option, score +from guideme.enums import render + +from .conftest import ( + JSON, + TICKET, + Json, + Runner, + as_object, + expect_post, + narrow, + reply, + validator, +) + +check_request = validator("request") + +BILLING = "Payments, invoicing, refunds" +TECHNICAL = "Bugs, outages, integrations" +COSMETIC = "No impact to functionality" + +# The golden table of the cross-SDK design: the inputs, and the exact bytes both SDKs +# render them to. `guideme-rust` reproduces this table from its derive macro, and +# spec/vectors/rubric.json is where the two are held to each other. +GOLDEN: list[tuple[str, str]] = [ + (option(BILLING), BILLING), + ( + option(TECHNICAL, examples=["502 on every request"]), + f"{TECHNICAL}\nExamples: 502 on every request", + ), + ( + option(BILLING, examples=["My card was charged twice", "Where is my refund?"]), + f"{BILLING}\nExamples: My card was charged twice; Where is my refund?", + ), + ( + option( + BILLING, + examples=["My card was charged twice"], + counterexamples=["The dashboard is down"], + ), + f"{BILLING}\nExamples: My card was charged twice\nNot this option: The dashboard is down", + ), + ( + option(BILLING, counterexamples=["The dashboard is down"]), + f"{BILLING}\nNot this option: The dashboard is down", + ), + ( + level(COSMETIC, examples=["typo in a label", "misaligned icon"]), + f"{COSMETIC}\nExamples: typo in a label; misaligned icon", + ), +] + +GOLDEN_IDS = [ + "no_parts", + "one_example", + "two_examples", + "examples_and_a_counterexample", + "a_counterexample_only", + "a_level_with_examples", +] + + +@pytest.mark.parametrize(("rubric", "expected"), GOLDEN, ids=GOLDEN_IDS) +def test_a_rubric_renders_the_bytes_the_contract_names(rubric: str, expected: str) -> None: + assert render(rubric) == expected + # The value itself stays the bare text: a rubric only expands where it becomes + # wire text, so a member's value reads as it is written. + assert str(rubric) in {BILLING, TECHNICAL, COSMETIC} + + +@given(st.text(min_size=1).filter(lambda text: bool(text.strip()))) +def test_a_rubric_with_no_parts_renders_byte_for_byte(what: str) -> None: + # The load-bearing invariant: 0.1.0's bytes do not move. A bare string and an + # `option(...)` with nothing attached both render to the text itself. + assert render(what) == what + assert render(option(what)) == what + assert render(level(what)) == what + assert render(fallback(what)) == what + + +class Department(Choice): + """A choice whose every member carries examples, one of them the fallback.""" + + billing = option( + BILLING, + examples=["My card was charged twice", "Where is my refund?"], + counterexamples=["The dashboard is down"], + ) + technical = option(TECHNICAL, examples=["502 on every request"]) + sales = fallback("Pricing, upgrades, new accounts", examples=["Do you have a team plan?"]) + + +class Severity(Levels): + """Ordered levels, each with the inputs that score there.""" + + cosmetic = level(COSMETIC, examples=["typo in a label"]) + degraded = level("Broken feature, workaround exists", examples=["export fails in one browser"]) + blocking = level("No workaround exists", examples=["cannot log in", "data loss"]) + + +ANSWERS: dict[str, Json] = { + "q0": { + "type": "choice", + "choice": "technical", + "probabilities": {"billing": 0.1, "technical": 0.8, "sales": 0.1}, + "confidence": 0.9, + }, + "q1": { + "type": "score", + "score": 1.0, + "legend": {"0": COSMETIC, "1": "Broken feature", "2": "No workaround"}, + "probabilities": {"0": 0.1, "1": 0.8, "2": 0.1}, + "confidence": 0.9, + }, +} +"""One answer per question of the batch below, over its own keys and levels.""" + +CHOICE_CRITERIA: Json = { + "billing": ( + f"{BILLING}\nExamples: My card was charged twice; Where is my refund?" + f"\nNot this option: The dashboard is down" + ), + "technical": f"{TECHNICAL}\nExamples: 502 on every request", + "sales": "Pricing, upgrades, new accounts\nExamples: Do you have a team plan?", +} +"""What `Department` must put on the wire, key by key.""" + +SCORE_CRITERIA: Json = [ + f"{COSMETIC}\nExamples: typo in a label", + "Broken feature, workaround exists\nExamples: export fails in one browser", + "No workaround exists\nExamples: cannot log in; data loss", +] +"""What `Severity` must put on the wire, low to high.""" + + +def test_examples_reach_the_wire_as_the_rendered_criteria( + httpserver: HTTPServer, runner: Runner +) -> None: + expect_post(httpserver).respond_with_data(reply(ANSWERS), content_type=JSON) + batch = ( + choose(Department, "Which team should handle this?"), + score(Severity, "How bad is it?"), + ) + assert runner.ask(batch, TICKET) == (Department.technical, Severity.degraded) + + request, _ = httpserver.log[-1] + body = as_object(narrow(request.get_json())) + check_request(body) + questions = as_object(body["questions"]) + assert as_object(questions["q0"])["criteria"] == CHOICE_CRITERIA + assert as_object(questions["q1"])["criteria"] == SCORE_CRITERIA + # A fallback marked with examples is still the fallback. + assert Department.fallback_member() is Department.sales diff --git a/tests/test_surface.py b/tests/test_surface.py index dd5624e..966e471 100644 --- a/tests/test_surface.py +++ b/tests/test_surface.py @@ -38,7 +38,9 @@ "choose", "choose_among", "fallback", + "level", "noul", + "option", "score", "score_levels", } diff --git a/tests/test_wire.py b/tests/test_wire.py index 3898d92..7cb7a69 100644 --- a/tests/test_wire.py +++ b/tests/test_wire.py @@ -37,6 +37,7 @@ UnsureError, choose, fallback, + option, score, ) from guideme.api import NoulAnswer, Request, Response, question_to_wire, request_to_wire @@ -559,6 +560,13 @@ def _levels_given_as_one_string(_monkeypatch: pytest.MonkeyPatch) -> None: _ = score_levels("How cross?", "abc") +def _a_runtime_level_with_counterexamples(_monkeypatch: pytest.MonkeyPatch) -> None: + # `level(...)` offers no counterexamples, so this is an `option(...)` written where a + # level belongs. Rendering "Not this option" onto an ordered scale is meaningless and + # dropping what it carries is silent, so the constructor refuses it. + _ = score_levels("How cross?", [option("Calm", counterexamples=["shouting"]), "Cross"]) + + def _events_log_without_the_logs_api(monkeypatch: pytest.MonkeyPatch) -> None: # Asking for log records where `opentelemetry-api` has no logs API is refused where # it is asked for. The alternative is a guide that emits none and never says so. @@ -597,6 +605,7 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: _an_api_key_of_spaces, _an_empty_model, _levels_given_as_one_string, + _a_runtime_level_with_counterexamples, _events_log_without_the_logs_api, _events_both_without_the_logs_api, _events_log_with_a_drifted_log_record, @@ -610,6 +619,7 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: "api_key_of_spaces", "empty_model", "levels_given_as_one_string", + "a_runtime_level_with_counterexamples", "events_log_without_the_logs_api", "events_both_without_the_logs_api", "events_log_with_a_drifted_log_record", From be48eaccd3d0d89a18511486ea1990a246320aee Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 21:16:26 -0300 Subject: [PATCH 02/18] Document rubric examples and why they are flattened The next reader will ask why a documented API feature was ignored, so docs/design.md records the measurement: object criteria come back in a score answer's legend and would not parse against a public type nothing reads, for no gain and about 11% more billed input tokens. docs/contract.md states the rendered string as a contract item and names the one deliberate asymmetry with Rust, whose renderer lives in the derive macro. The README shows option, level and fallback in code, which is also what tests/test_surface.py asks of every exported name, and AGENTS.md gains the invariant and the new test ceiling. --- AGENTS.md | 24 +++++++++++++++++++----- README.md | 45 +++++++++++++++++++++++++++++++++++++++++++++ docs/contract.md | 10 ++++++++++ docs/design.md | 28 ++++++++++++++++++++++++++++ 4 files changed, 102 insertions(+), 5 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index faaed3b..2546c1f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -8,7 +8,7 @@ A Python package that makes a TypeSafe Jev judgment usable as control flow: a ye `if`, a choice is an exhaustive `match`, a score is a comparison. One distribution, `guideme`, published to PyPI under `MIT OR Apache-2.0`. The public surface has two tiers: -- the 33 names in `__all__` in `src/guideme/__init__.py`, imported from `guideme` itself; +- the 35 names in `__all__` in `src/guideme/__init__.py`, imported from `guideme` itself; - `guideme.api` and `guideme.policy` as whole modules, imported by their own path and not re-exported at the top level: `guideme.api` is the wire mirror and `guideme.api.client` holds `Client` and `AsyncClient`, and `guideme.policy` holds `resolve`. @@ -40,7 +40,7 @@ two pages: `https://docs.typesafe.ai/api.md` covers `POST /v1/systemone` and | `src/guideme/scalars.py` | `Probability`, `Confidence`, `Key`, `Rank`, `ApiKey`, `Model` | validation happens once, here; `ApiKey` never prints | | `src/guideme/errors.py` | the `GuidemeError` tree and `kind` | `kind` is the cross-SDK name and the `error.type` value; imports nothing from `guideme` | | `src/guideme/policy.py` | `resolve`, `Policy`, `Thresholds`, `Verdict`, the answer and outcome dataclasses | pure: no I/O, no caller enums, keys and level indices only | -| `src/guideme/enums.py` | `Choice`, `Levels`, `fallback` | a member's name is its wire key and its value is its rubric; both validate at class definition, and a repeated rubric text is refused there | +| `src/guideme/enums.py` | `Choice`, `Levels`, `option`, `level`, `fallback`, `render` | a member's name is its wire key and its value is its rubric; both validate at class definition, and a repeated rubric text is refused there. `render` is the one place a rubric's examples become wire text, and its output is a cross-SDK contract item | | `src/guideme/question.py` | question kinds, constructors, `Ranked`, `Scored`, the unsure ladder | a question is inert until asked; the reader travels with it | | `src/guideme/ask.py` | shapes: `encode`, `decode`, `Plan` | ids are `q0..qN` in encounter order, insertion order for a dict | | `src/guideme/_ask_overloads.py` | the typed `ask` surfaces | GENERATED; edit `scripts/gen_ask_overloads.py` and run `mise run gen` | @@ -106,6 +106,16 @@ the rest are checked in review. is a `ConfigError` on the class statement; so is `fallback(…)` on a `Levels`, which has `.otherwise(level)` instead. Both fire where the enum is written, like the Rust derive's compile errors. +- **A rubric's examples are composed into its text at the wire, and nowhere else.** A rubric's + value stays the bare text; `enums.render` is what turns an `option(…)`, `level(…)` or + `fallback(…)` into what the model reads, and every site that puts a rubric on the wire — + `Choice.rubric()`, `Levels.levels()`, `choose_among`, `score_levels` — goes through it, so a + rubric cannot arrive having quietly lost what it carries. A bare string and a rubric with no + examples render to their own bytes, so an existing caller's request does not move. The + rendered string is a contract item shared with every other SDK, stated in `docs/contract.md` + and pinned by `spec/vectors/rubric.json`. A blank rubric, a blank or repeated example, an + `examples=[]` written out, and a counterexample on a level are each a `ConfigError` where the + rubric is written. - **Nothing outside `api/` may see a `pydantic` exception.** A caller's state or instructions that pydantic refuses leaves `api/` as a `ConfigError`, and a `NaN` or an infinity is refused rather than serialised as `null`. @@ -159,13 +169,17 @@ is made in `guideme-rust` first, not here. ## Tests -Few tests, high grade. The ceiling is 43 test functions; a parametrised function counts once. +Few tests, high grade. The ceiling is 47 test functions; a parametrised function counts once. It was 40 before the logs signal, which is user-requested scope that the span assertions could not cover: correlation, severity and routing each need a record to look at. The forty-third is the pre-publish proof that a log sink which raises reaches neither the caller nor the ask span: it asserts the absence of a failure on a path where every other test asserts a presence, so no -existing test could carry it. Everything else that pass added went into a parameter of a test -that was already there. A new test must be one of: +existing test could carry it. Rubric examples added the last four, also user-requested scope: +the golden rendering table, the passthrough property that proves 0.1.0's bytes have not moved, +the wire proof that a rendered rubric reaches the request, and a live proof that examples move +the distribution. Each asserts a different thing about a string no existing test looks at. +Everything else those two passes added went into a parameter of a test that was already there. +A new test must be one of: - a property test (`hypothesis`) over a law of `policy.resolve`, the shapes, or the wire types; - a wire or contract check through a real local HTTP server (`pytest-httpserver`), asserting on diff --git a/README.md b/README.md index 71a79d3..d71287b 100644 --- a/README.md +++ b/README.md @@ -104,6 +104,51 @@ A noul can carry `.criteria("what yes means", "what no means")`. Instructions ac or any JSON-shaped value, so a question can reference structured data by field name the way the TypeSafe docs describe. +### Examples in a rubric + +Two options that read alike are told apart by showing inputs rather than by describing harder. +`option(…)` takes the inputs that belong to an option and the ones that belong somewhere else, +`level(…)` takes the inputs that score at that level, and `fallback(…)` is an `option(…)` that +also marks the unsure member. + +```python +from guideme import Choice, Levels, fallback, level, option + + +class Department(Choice): + billing = option( + "Payments, invoicing, refunds", + examples=["My card was charged twice", "Where is my refund?"], + counterexamples=["The dashboard is down"], + ) + technical = option("Bugs, outages, integrations", examples=["502 on every request"]) + sales = fallback("Pricing, upgrades, new accounts", examples=["Do you have a team plan?"]) + + +class Severity(Levels): + cosmetic = level("No impact to functionality", examples=["typo in a label"]) + degraded = level("Broken feature, workaround exists", examples=["export fails in one browser"]) + blocking = level("No workaround exists", examples=["cannot log in", "data loss"]) +``` + +The member's value is still the bare rubric; the examples are composed into it only in the +request, as + +```text +Payments, invoicing, refunds +Examples: My card was charged twice; Where is my refund? +Not this option: The dashboard is down +``` + +So a rubric with no examples sends exactly what it sent before, and the same strings work in +`choose_among("…", {"billing": option(…)})` and `score_levels("…", [level(…), …])`. A string +may be an example of one option and a counterexample of another: that is the point when two +options are confusable. `level(…)` has no counterexamples, because "not this option" means +nothing on an ordered scale — an input that does not belong at one level scores at another. + +A blank rubric, a blank or repeated example, an `examples=[]` written out, and a counterexample +on a level are each a `ConfigError` where the rubric is written. + The state is anything JSON-shaped: a text literal, a `dict`, a list of them. A dataclass goes through `dataclasses.asdict`, a pydantic model through `.model_dump()`. diff --git a/docs/contract.md b/docs/contract.md index 26ff89e..73f9601 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -16,6 +16,16 @@ every one of which is a parametrised case of `tests/test_policy_vectors.py`: an and another SDK is either a bug here or an ambiguity to resolve upstream; it is never fixed by changing the copy. +The rendered rubric string is a contract item too. An option or a level written with +`option(…)`, `level(…)` or `fallback(…)` carries its examples beside its text, and the string +those compose into — clauses joined with a newline, items within one joined with `"; "`, the +text verbatim, and the text alone when there are no examples — is what goes on the wire. Every +guideme SDK composes the same string from the same parts, and `spec/vectors/rubric.json` is +where the renderers are held to each other. One asymmetry is deliberate: `choose_among` and +`score_levels` here take an `option(…)` or a `level(…)` value, while Rust's equivalents take a +string its derive macro has already composed, because that is where Rust's renderer lives. +Equivalent inputs put identical bytes on the wire. + Drift is caught rather than trusted. `mise run spec-check` clones guideme-rust, diffs its `spec/` against this one and fails on any difference except `spec/SOURCE`, which is provenance and has no counterpart upstream. It runs on every pull request as the `spec-drift` job of diff --git a/docs/design.md b/docs/design.md index 4385cf2..0501a6b 100644 --- a/docs/design.md +++ b/docs/design.md @@ -97,6 +97,29 @@ Each module survives the test. an error span status rather than an error-level record. Anything without a convention is namespaced `guideme.`. The package depends on `opentelemetry-api` only and installs no provider; `docs/observability.md` shows the exporter side. +- **A rubric's examples are flattened into its text, not sent as structured criteria.** The + TypeSafe API takes structured `criteria`, and `docs.typesafe.ai/primitives/choice.md` + documents exactly the `what` / `not_for` / `examples` object this surface wants. It is not + used, for three measured reasons. A score answer echoes its criteria back in `legend`, which + is `dict[str, str]` here and `BTreeMap` in Rust; object criteria come back as + objects and fail to parse, so sending them means a breaking change to a public type — in a + field neither SDK reads beyond its length. Flattening is as good: on the docs' own worked + example, flattened scored 1.01 against structured's 1.03 at a higher confidence, and on an + ambiguous choice both reached the option that bare strings miss, inside run-to-run variance. + And flattening is cheaper: identical content billed 400 input tokens flattened against 450 + structured. The gain comes from the examples being present, not from the JSON shape. So + `option(…)`, `level(…)` and `fallback(…)` carry the parts on a `str` subclass and `render` + composes them where the rubric becomes wire text — the wire schema, `spec/`, and every + existing golden vector untouched. +- **The rendered rubric is a contract item, and the renderer has one entry per wire site.** + Clauses join with a newline and items with `"; "`, and the text is used verbatim: a newline + rather than a space is what removes the need for a punctuation rule, since an example ending + in `?` would otherwise render as `Where is my refund?.`. `Choice.rubric()`, `Levels.levels()`, + `choose_among` and `score_levels` all render, so a value that reached a runtime constructor + cannot silently lose its examples. Rust renders the same string inside its derive macro, + which is the one deliberate asymmetry: `choose_among` here takes an `option(…)`, while Rust's + equivalent takes a string the macro already composed. Equivalent inputs put identical bytes + on the wire. - **An answer is a span event and a log record, and the caller picks.** Rust emits one `tracing` event and lets the subscriber fan it out, so its example filters events off the span exporter to store each one once. There is no subscriber here, so the library makes both @@ -123,6 +146,11 @@ Each module survives the test. statement, naming the members that repeat. - **`Key` and `Rank`** are only meaningful through `choose_among` and `score_levels`. They are `NewType`s over `str` and `int`, so nothing else hands you one. +- **A rubric's value is its bare text, examples or not.** `option("x", examples=[…])` still + equals `"x"`, so two options whose text matches are still one member however their examples + differ, and a member's `.value` still reads as it was written. The expansion happens only in + the request. That also means an `examples=[]` written out is a `ConfigError`: it cannot be + told from the default by its effect, so it is refused as the mistake it is. - **The four scalars are brands, not validated types.** `Probability`, `Confidence`, `Key` and `Rank` are `NewType`s, so `Probability(2.0)` and `Rank(99)` are accepted by the checker and by the interpreter alike. What makes them trustworthy is that only the wire mints them, and it From b12df1d5af8aeaa1760bd6a9e77a6124ce48062e Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 21:16:28 -0300 Subject: [PATCH 03/18] Bump to 0.1.1 Purely additive, so a patch. Both lock files record the version and both are resolved with --locked, so the root and examples/otlp are relocked here. The observability capture carried the version on its InstrumentationScope lines and would now be stale. It was taken once and says that nothing in it is invented, so the version is elided rather than edited to a number no run produced. --- CHANGELOG.md | 19 +++++++++++++++++++ docs/observability.md | 12 +++++++----- examples/otlp/uv.lock | 2 +- pyproject.toml | 2 +- uv.lock | 2 +- 5 files changed, 29 insertions(+), 8 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index aedd5d8..9c01433 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,25 @@ Nothing yet. +## 0.1.1 — 2026-09-21 + +Additive. Nothing that worked in 0.1.0 sends different bytes. + +- `option(rubric, examples=…, counterexamples=…)` and `level(rubric, examples=…)` join + `fallback(…)`, which now takes the same keywords. All three are exported from `guideme`, + bringing `__all__` to 35 names. A rubric written as a bare string keeps working everywhere. +- The parts are composed into the rubric where it becomes wire text: clauses joined with a + newline, items within one joined with `"; "`, the text verbatim. A rubric with no examples + renders to its own text, byte for byte, so an existing request is unchanged. The rendered + string is a cross-SDK contract item, stated in `docs/contract.md` and pinned by + `spec/vectors/rubric.json`. +- Every site that puts a rubric on the wire renders — `Choice.rubric()`, `Levels.levels()`, + `choose_among` and `score_levels` — so examples cannot be silently dropped by reaching a + runtime constructor. +- `level(…)` takes no counterexamples: "not this option" means nothing on an ordered scale. An + `option(…)` carrying counterexamples written where a level belongs is a `ConfigError`, as is + a blank rubric, a blank or repeated example, and an `examples=[]` written out. + ## 0.1.0 — 2026-09-21 First release. Everything below is new, so this entry lists the surface rather than the diff --git a/docs/observability.md b/docs/observability.md index 065015c..4136919 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -328,7 +328,9 @@ a different backend. Captured from `otel/opentelemetry-collector-contrib` with the debug exporter and the `collector.yaml` above, running `examples/otlp` under `events("log")` against the live API. -Timestamps, `Flags` and the resource block are trimmed throughout. The first two blocks are +Timestamps, `Flags`, the version on each `InstrumentationScope` line and the resource block are +trimmed throughout — the version because a capture is taken once and every release would +otherwise leave a number here that a reader cannot tell from a real one. The first two blocks are otherwise complete; the later ones are excerpts, cut to the lines each is making a point about, so a missing `Parent ID`, `Kind`, scope line or attribute there means it was cut, not that it was absent. Nothing is reworded, and no value is invented. @@ -336,7 +338,7 @@ was absent. Nothing is reworded, and no value is invented. The three-question batch is one ask span with one attempt under it: ``` -InstrumentationScope guideme 0.1.0 +InstrumentationScope guideme Span #1 Trace ID : 164107043647c42bc827fff022ec4308 Parent ID : @@ -356,7 +358,7 @@ Attributes: -> gen_ai.usage.input_tokens: Int(422) -> gen_ai.usage.output_tokens: Int(71) -InstrumentationScope guideme.api 0.1.0 +InstrumentationScope guideme.api Span #1 Trace ID : 164107043647c42bc827fff022ec4308 Parent ID : 53c4b9ada6417408 @@ -376,7 +378,7 @@ Attributes: Its three answers arrive on the logs pipeline, each carrying the ask span's ids: ``` -InstrumentationScope guideme 0.1.0 +InstrumentationScope guideme LogRecord #1 SeverityText: INFO SeverityNumber: Info(9) @@ -441,7 +443,7 @@ throttled attempt's rather than the ask's. The live API does not throttle on dem block is from a local server that answers `429` once: ``` -InstrumentationScope guideme.api 0.1.0 +InstrumentationScope guideme.api LogRecord #0 SeverityText: WARN SeverityNumber: Warn(13) diff --git a/examples/otlp/uv.lock b/examples/otlp/uv.lock index 5820788..88d00b9 100644 --- a/examples/otlp/uv.lock +++ b/examples/otlp/uv.lock @@ -103,7 +103,7 @@ wheels = [ [[package]] name = "guideme" -version = "0.1.0" +version = "0.1.1" source = { editable = "../../" } dependencies = [ { name = "httpx" }, diff --git a/pyproject.toml b/pyproject.toml index 1cb67d9..0fc7092 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "guideme" -version = "0.1.0" +version = "0.1.1" description = "Type-safe inline judgments from TypeSafe Jev: a yes/no is an if, a choice is an exhaustive match, a score is a comparison." readme = "README.md" requires-python = ">=3.12" diff --git a/uv.lock b/uv.lock index 399cd96..8562ffc 100644 --- a/uv.lock +++ b/uv.lock @@ -421,7 +421,7 @@ wheels = [ [[package]] name = "guideme" -version = "0.1.0" +version = "0.1.1" source = { editable = "." } dependencies = [ { name = "httpx" }, From b4fd732bd16719ca3d742f27a44f1e0e6c1c94b2 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 21:25:47 -0300 Subject: [PATCH 04/18] Give noul criteria examples, and refuse contradictory ones A yes and a no are two described alternatives of one question, as confusable as two options of a choice. Measured against live Jev: asked whether a nightly export job with a manual workaround is urgent, plain `Urgent` / `Not urgent` answers yes at 0.75 four runs running, and the same criteria carrying examples answer no at 0.17 four runs running. The plain criteria are wrong, and one of the no examples is exactly what the state describes. So `.criteria(...)` renders like every other wire site and `option(...)` serves it, rather than a fourth constructor for the same idea. Two contradictions are now refused where the rubric is written: one string as an example of two alternatives of the same question, which says an input belongs to both, and one string as both an example and a counterexample of the same alternative. The overlap that reads alike and is the whole point stays legal -- an example of one alternative and a counterexample of another -- and the wire test now sends exactly that, so a future tightening cannot take it away unnoticed. Rendering order was already declaration order; it is now stated as contract and pinned by a golden case whose clauses are written out of alphabetical order, so a renderer that sorted or took a set would fail it. --- AGENTS.md | 23 +++++++++++----- CHANGELOG.md | 22 +++++++++++----- README.md | 29 ++++++++++++++------- docs/contract.md | 26 +++++++++++------- docs/design.md | 32 +++++++++++++++++------ src/guideme/enums.py | 58 ++++++++++++++++++++++++++++++++--------- src/guideme/question.py | 22 +++++++++++++--- tests/test_enums.py | 33 +++++++++++++++++++++++ tests/test_rubrics.py | 46 +++++++++++++++++++++++++++----- tests/test_wire.py | 24 +++++++++++++++++ 10 files changed, 252 insertions(+), 63 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 2546c1f..831fe18 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -109,13 +109,22 @@ the rest are checked in review. - **A rubric's examples are composed into its text at the wire, and nowhere else.** A rubric's value stays the bare text; `enums.render` is what turns an `option(…)`, `level(…)` or `fallback(…)` into what the model reads, and every site that puts a rubric on the wire — - `Choice.rubric()`, `Levels.levels()`, `choose_among`, `score_levels` — goes through it, so a - rubric cannot arrive having quietly lost what it carries. A bare string and a rubric with no - examples render to their own bytes, so an existing caller's request does not move. The - rendered string is a contract item shared with every other SDK, stated in `docs/contract.md` - and pinned by `spec/vectors/rubric.json`. A blank rubric, a blank or repeated example, an - `examples=[]` written out, and a counterexample on a level are each a `ConfigError` where the - rubric is written. + `Choice.rubric()`, `Levels.levels()`, `choose_among`, `score_levels` and + `NoulQuestion.criteria` — goes through it, so a rubric cannot arrive having quietly lost what + it carries. All three kinds of question take examples: a yes and a no are as confusable as + two options, and `option(…)` is the one constructor for a described alternative, so there is + no fourth. A bare string and a rubric with no examples render to their own bytes, so an + existing caller's request does not move. The rendered string is a contract item shared with + every other SDK, stated in `docs/contract.md` and pinned by `spec/vectors/rubric.json`, and + so is the order: examples and counterexamples render in the order written, never sorted and + never de-duplicated into a set. +- **A rubric's examples must be consistent, and that is checked where it is written.** A blank + or whitespace-only rubric or entry, a repeat within one clause, an `examples=[]` written out, + and a counterexample on a level are each a `ConfigError`. So are the two contradictions: one + string as an example of two alternatives of the same question, and one string as both an + example and a counterexample of the same alternative. One string as an example of one + alternative and a counterexample of another is **legal and required** — it is the confusable + pattern the feature exists for, and `tests/test_rubrics.py` sends it on the wire. - **Nothing outside `api/` may see a `pydantic` exception.** A caller's state or instructions that pydantic refuses leaves `api/` as a `ConfigError`, and a `NaN` or an infinity is refused rather than serialised as `null`. diff --git a/CHANGELOG.md b/CHANGELOG.md index 9c01433..85cbfff 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,16 +12,24 @@ Additive. Nothing that worked in 0.1.0 sends different bytes. `fallback(…)`, which now takes the same keywords. All three are exported from `guideme`, bringing `__all__` to 35 names. A rubric written as a bare string keeps working everywhere. - The parts are composed into the rubric where it becomes wire text: clauses joined with a - newline, items within one joined with `"; "`, the text verbatim. A rubric with no examples - renders to its own text, byte for byte, so an existing request is unchanged. The rendered - string is a cross-SDK contract item, stated in `docs/contract.md` and pinned by - `spec/vectors/rubric.json`. + newline, items within one joined with `"; "`, in the order written, the text verbatim. A + rubric with no examples renders to its own text, byte for byte, so an existing request is + unchanged. The rendered string and its order are cross-SDK contract items, stated in + `docs/contract.md` and pinned by `spec/vectors/rubric.json`. +- All three kinds of question take them. `noul("…").criteria(yes, no)` accepts an `option(…)` + for either side: a yes and a no are as confusable as two options, and there is no fourth + constructor for them. - Every site that puts a rubric on the wire renders — `Choice.rubric()`, `Levels.levels()`, - `choose_among` and `score_levels` — so examples cannot be silently dropped by reaching a - runtime constructor. + `choose_among`, `score_levels` and `NoulQuestion.criteria` — so examples cannot be silently + dropped by reaching a runtime constructor. - `level(…)` takes no counterexamples: "not this option" means nothing on an ordered scale. An `option(…)` carrying counterexamples written where a level belongs is a `ConfigError`, as is - a blank rubric, a blank or repeated example, and an `examples=[]` written out. + a blank or whitespace-only rubric or entry, a repeat within one clause, and an `examples=[]` + written out. +- Contradictory examples are a `ConfigError` too: one string as an example of two alternatives + of the same question, or as both an example and a counterexample of the same alternative. One + string as an example of one alternative and a counterexample of another stays legal — that is + the confusable-options pattern the feature exists for. ## 0.1.0 — 2026-09-21 diff --git a/README.md b/README.md index d71287b..5aed7c3 100644 --- a/README.md +++ b/README.md @@ -106,10 +106,11 @@ the TypeSafe docs describe. ### Examples in a rubric -Two options that read alike are told apart by showing inputs rather than by describing harder. -`option(…)` takes the inputs that belong to an option and the ones that belong somewhere else, -`level(…)` takes the inputs that score at that level, and `fallback(…)` is an `option(…)` that -also marks the unsure member. +Two alternatives that read alike are told apart by showing inputs rather than by describing +harder. `option(…)` takes the inputs that belong to an alternative and the ones that belong +somewhere else, `level(…)` takes the inputs that score at that level, and `fallback(…)` is an +`option(…)` that also marks the unsure member. All three kinds of question take them: a noul's +`.criteria(…)` accepts an `option(…)` for the yes and for the no. ```python from guideme import Choice, Levels, fallback, level, option @@ -141,13 +142,21 @@ Not this option: The dashboard is down ``` So a rubric with no examples sends exactly what it sent before, and the same strings work in -`choose_among("…", {"billing": option(…)})` and `score_levels("…", [level(…), …])`. A string -may be an example of one option and a counterexample of another: that is the point when two -options are confusable. `level(…)` has no counterexamples, because "not this option" means -nothing on an ordered scale — an input that does not belong at one level scores at another. +`choose_among("…", {"billing": option(…)})`, `score_levels("…", [level(…), …])` and +`noul("…").criteria(option(…), option(…))`. Examples and counterexamples render in the order +they are written, always: that order is part of the published contract. -A blank rubric, a blank or repeated example, an `examples=[]` written out, and a counterexample -on a level are each a `ConfigError` where the rubric is written. +A string may be an example of one option and a counterexample of another. That is the point +when two options are confusable, and it is the one overlap that stays legal. Offering the same +string as an example of two options, or as both an example and a counterexample of the same +option, says an input belongs where it cannot, so each is refused. + +`level(…)` has no counterexamples, because "not this option" means nothing on an ordered +scale — an input that does not belong at one level scores at another. + +A blank or whitespace-only rubric or entry, a repeat within one clause, an `examples=[]` +written out, a counterexample on a level, and the two contradictions above are each a +`ConfigError` where the rubric is written. The state is anything JSON-shaped: a text literal, a `dict`, a list of them. A dataclass goes through `dataclasses.asdict`, a pydantic model through `.model_dump()`. diff --git a/docs/contract.md b/docs/contract.md index 73f9601..2d9ce23 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -16,15 +16,23 @@ every one of which is a parametrised case of `tests/test_policy_vectors.py`: an and another SDK is either a bug here or an ambiguity to resolve upstream; it is never fixed by changing the copy. -The rendered rubric string is a contract item too. An option or a level written with -`option(…)`, `level(…)` or `fallback(…)` carries its examples beside its text, and the string -those compose into — clauses joined with a newline, items within one joined with `"; "`, the -text verbatim, and the text alone when there are no examples — is what goes on the wire. Every -guideme SDK composes the same string from the same parts, and `spec/vectors/rubric.json` is -where the renderers are held to each other. One asymmetry is deliberate: `choose_among` and -`score_levels` here take an `option(…)` or a `level(…)` value, while Rust's equivalents take a -string its derive macro has already composed, because that is where Rust's renderer lives. -Equivalent inputs put identical bytes on the wire. +The rendered rubric string is a contract item too. An alternative written with `option(…)`, +`level(…)` or `fallback(…)` carries its examples beside its text, and the string those compose +into — clauses joined with a newline, items within one joined with `"; "`, the text verbatim, +and the text alone when there are no examples — is what goes on the wire. Every guideme SDK +composes the same string from the same parts, and `spec/vectors/rubric.json` is where the +renderers are held to each other. All three kinds of question carry them, a noul's yes and no +included. + +**Order is part of it.** Examples and counterexamples render in the order they were written, +never sorted and never collapsed into a set, because two SDKs ordering differently would send +different bytes for the same declaration. + +One asymmetry is deliberate. `choose_among` and `score_levels` here take an `option(…)` or a +`level(…)` value; Rust's equivalents keep taking a plain string, because widening their +signatures risks inference breakage for existing callers on a path that can already pass a +string its own renderer composed. It is revisited at 0.2.0. Equivalent inputs put identical +bytes on the wire either way, which is what the contract actually promises. Drift is caught rather than trusted. `mise run spec-check` clones guideme-rust, diffs its `spec/` against this one and fails on any difference except `spec/SOURCE`, which is provenance and has diff --git a/docs/design.md b/docs/design.md index 0501a6b..c8a992d 100644 --- a/docs/design.md +++ b/docs/design.md @@ -112,14 +112,30 @@ Each module survives the test. composes them where the rubric becomes wire text — the wire schema, `spec/`, and every existing golden vector untouched. - **The rendered rubric is a contract item, and the renderer has one entry per wire site.** - Clauses join with a newline and items with `"; "`, and the text is used verbatim: a newline - rather than a space is what removes the need for a punctuation rule, since an example ending - in `?` would otherwise render as `Where is my refund?.`. `Choice.rubric()`, `Levels.levels()`, - `choose_among` and `score_levels` all render, so a value that reached a runtime constructor - cannot silently lose its examples. Rust renders the same string inside its derive macro, - which is the one deliberate asymmetry: `choose_among` here takes an `option(…)`, while Rust's - equivalent takes a string the macro already composed. Equivalent inputs put identical bytes - on the wire. + Clauses join with a newline and items with `"; "`, in the order written, and the text is used + verbatim: a newline rather than a space is what removes the need for a punctuation rule, + since an example ending in `?` would otherwise render as `Where is my refund?.`. The + `Not this option` label was measured against `Not` and `Counterexamples` and won. + `Choice.rubric()`, `Levels.levels()`, `choose_among`, `score_levels` and + `NoulQuestion.criteria` all render, so a value that reached a runtime constructor cannot + silently lose its examples. Rust renders the same string, which leaves one deliberate + asymmetry: `choose_among` here takes an `option(…)`, while Rust's equivalent keeps taking a + plain string rather than risk inference breakage for existing callers. Equivalent inputs put + identical bytes on the wire. +- **A noul's criteria take examples too, and through the same `option(…)`.** A yes and a no are + two described alternatives of one question, exactly as confusable as two options of a choice, + and measurement says so: asked whether a nightly export job with a manual workaround is + urgent, plain `Urgent` / `Not urgent` answers yes at 0.75 four times running, and the same + criteria carrying examples answer no at 0.17. The examples move it to the correct answer, + because one of the no examples is the situation the state describes. So there is no fourth + constructor — `option(…)` is "a described alternative" and serves both — and `.criteria(…)` + renders like every other wire site. +- **Contradictory examples are refused where they are written.** One string offered as an + example of two alternatives of the same question says an input belongs to both, which cannot + be true; one string offered as both an example and a counterexample of the same alternative + says it does and does not belong. Both are a `ConfigError`. The overlap that looks similar + and is the whole point stays legal: the same string as an example of one alternative and a + counterexample of another is how two confusable ones are told apart. - **An answer is a span event and a log record, and the caller picks.** Rust emits one `tracing` event and lets the subscriber fan it out, so its example filters events off the span exporter to store each one once. There is no subscriber here, so the library makes both diff --git a/src/guideme/enums.py b/src/guideme/enums.py index a47918d..6525c59 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -9,7 +9,7 @@ the first request. """ -from collections.abc import Sequence +from collections.abc import Iterable, Sequence from enum import Enum from typing import Self, final @@ -92,12 +92,17 @@ def _rubric( if not rubric.strip(): detail = f"a rubric must say something, got {rubric!r}" raise ConfigError(detail) - return _Rubric( - rubric, - _items(rubric, examples, "examples"), - _items(rubric, counterexamples, "counterexamples"), - is_fallback=is_fallback, - ) + shown = _items(rubric, examples, "examples") + excluded = _items(rubric, counterexamples, "counterexamples") + both = [item for item in shown if item in set(excluded)] + if both: + detail = ( + f"{rubric!r}: {both[0]!r} is both an example and a counterexample of it, which " + f"says the input does and does not belong here; it may be an example of one " + f"option and a counterexample of another, but not of the same one" + ) + raise ConfigError(detail) + return _Rubric(rubric, shown, excluded, is_fallback=is_fallback) def option( @@ -106,15 +111,20 @@ def option( examples: Sequence[str] = _UNSET, counterexamples: Sequence[str] = _UNSET, ) -> str: - """A choice option: what it covers, inputs that belong to it, inputs that do not. + """A described alternative: what it covers, inputs that belong to it, inputs that do not. + + Use it for a `Choice` member, for a runtime option of `choose_among`, and for either + side of a noul's `.criteria(...)`; a yes and a no are as confusable as two options. The value is still the rubric text; the examples are composed into it where the rubric goes on the wire, so an option written as a bare string and one written as - `option("…")` send the same bytes. A string may be an example of one option and a - counterexample of another: that is how two confusable options are told apart. + `option("…")` send the same bytes, and both clauses keep the order they were + written in. A string may be an example of one alternative and a counterexample of + another: that is how two confusable ones are told apart. - An empty rubric, an empty entry, an `examples=[]` written out, or a repeat within - one clause is a `ConfigError` where the option is written. + A blank rubric or entry, an `examples=[]` written out, a repeat within one clause, + and a string given as both an example and a counterexample of this one alternative + are each a `ConfigError` where the option is written. """ return _rubric(rubric, examples, counterexamples, is_fallback=False) @@ -166,6 +176,28 @@ def render(rubric: str) -> str: return "\n".join(lines) +def require_unshared_examples(where: str, rubrics: Iterable[tuple[str, str | None]]) -> None: + """Refuse one string offered as an example of two alternatives of the same question. + + It would say the input belongs to both, which cannot be true. The reverse is legal + and is the point of the feature: the same string as an example of one alternative + and a counterexample of another is how two confusable ones are told apart. + """ + seen: dict[str, str] = {} + for name, rubric in rubrics: + if not isinstance(rubric, _Rubric): + continue + for example in rubric.examples: + first = seen.setdefault(example, name) + if first != name: + detail = ( + f"{where}: {example!r} is an example of both {first} and {name}, so it " + f"says one input belongs to two alternatives; make it a counterexample " + f"of one of them instead" + ) + raise ConfigError(detail) + + def require_no_counterexamples(where: str, rubric: str) -> None: """Refuse counterexamples on a level, rather than rendering or dropping them. @@ -232,6 +264,7 @@ def __init_subclass__(cls) -> None: if len(marked) > 1: detail = f"{cls.__name__}: only one member may be marked fallback, got {marked}" raise ConfigError(detail) + require_unshared_examples(cls.__name__, ((m.name, m.value) for m in cls)) @classmethod def rubric(cls) -> tuple[tuple[str, str], ...]: @@ -289,6 +322,7 @@ def __init_subclass__(cls) -> None: ) raise ConfigError(detail) require_no_counterexamples(f"{cls.__name__}.{member.name}", member.value) + require_unshared_examples(cls.__name__, ((m.name, m.value) for m in cls)) @property def index(self) -> int: diff --git a/src/guideme/question.py b/src/guideme/question.py index 53c8aab..db0d13d 100644 --- a/src/guideme/question.py +++ b/src/guideme/question.py @@ -10,7 +10,13 @@ from typing import Self, final from guideme._json import Json -from guideme.enums import Choice, Levels, render, require_no_counterexamples +from guideme.enums import ( + Choice, + Levels, + render, + require_no_counterexamples, + require_unshared_examples, +) from guideme.errors import ConfigError, ProtocolError, UnsureError from guideme.policy import ( MAX_LEVELS, @@ -325,8 +331,14 @@ class NoulQuestion(_Binary[bool], _Fallible[bool]): """A yes/no question read as a `bool`.""" def criteria(self, yes: str, no: str) -> Self: - """Describe what a yes and a no mean.""" - return replace(self, spec=NoulSpec(NoulCriteria(yes, no))) + """Describe what a yes and a no mean. + + Either may be an `option(...)` carrying examples, which are composed into it + here: a yes and a no are as confusable as two options of a choice, and showing + an input that belongs to each is what tells them apart. + """ + require_unshared_examples("criteria", (("yes", yes), ("no", no))) + return replace(self, spec=NoulSpec(NoulCriteria(render(yes), render(no)))) def detail(self) -> DetailedNoul: """Read the full `Verdict` instead. Any `.otherwise(...)` is dropped.""" @@ -436,6 +448,7 @@ def choose_among(instructions: Json, options: Mapping[str, str | None]) -> Choic A rubric is a string, `None`, or an `option(...)` carrying examples, which are composed into it here: the same text a `Choice` member would send. """ + require_unshared_examples("choose_among", options.items()) return _choice( instructions, tuple((key, None if text is None else render(text)) for key, text in options.items()), @@ -484,4 +497,7 @@ def score_levels(instructions: Json, levels: Sequence[str]) -> ScoreQuestion[Ran raise ConfigError(detail) for index, text in enumerate(levels): require_no_counterexamples(f"level {index}", text) + require_unshared_examples( + "score_levels", ((f"level {index}", text) for index, text in enumerate(levels)) + ) return _score(instructions, tuple(render(text) for text in levels), Rank) diff --git a/tests/test_enums.py b/tests/test_enums.py index 7b31320..76d07ab 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -99,6 +99,33 @@ class Repeated(Choice): return Repeated +def _one_example_of_two_options() -> type[Choice]: + class Shared(Choice): + billing = option("Payments", examples=["Where is my refund?"]) + technical = option("Bugs", examples=["Where is my refund?"]) + + return Shared + + +def _one_example_of_two_levels() -> type[Levels]: + class Shared(Levels): + cosmetic = level("No impact", examples=["a typo in a label"]) + blocking = level("No workaround exists", examples=["a typo in a label"]) + + return Shared + + +def _an_example_that_is_also_a_counterexample() -> type[Choice]: + class Both(Choice): + billing = option( + "Payments", + examples=["Where is my refund?"], + counterexamples=["Where is my refund?"], + ) + + return Both + + def _a_counterexample_on_a_level() -> type[Levels]: class Marked(Levels): cosmetic = option("No impact", counterexamples=["cannot log in"]) @@ -177,6 +204,9 @@ class Urgency(Levels): _a_blank_example, _a_repeated_example, _a_counterexample_on_a_level, + _one_example_of_two_options, + _one_example_of_two_levels, + _an_example_that_is_also_a_counterexample, ] REFUSED_IDS = [ @@ -193,6 +223,9 @@ class Urgency(Levels): "a_blank_example", "a_repeated_example", "a_counterexample_on_a_level", + "one_example_of_two_options", + "one_example_of_two_levels", + "an_example_that_is_also_a_counterexample", ] diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py index 348d775..e2b75c3 100644 --- a/tests/test_rubrics.py +++ b/tests/test_rubrics.py @@ -3,7 +3,7 @@ from hypothesis import strategies as st from pytest_httpserver import HTTPServer -from guideme import Choice, Levels, choose, fallback, level, option, score +from guideme import Choice, Levels, choose, fallback, level, noul, option, score from guideme.enums import render from .conftest import ( @@ -53,6 +53,20 @@ level(COSMETIC, examples=["typo in a label", "misaligned icon"]), f"{COSMETIC}\nExamples: typo in a label; misaligned icon", ), + # Both clauses are written out of alphabetical order, so a renderer that sorted or + # took a set would fail here. Declaration order is contract: two SDKs ordering + # differently would send different bytes for the same declaration. + ( + option( + BILLING, + examples=["Where is my refund?", "My card was charged twice"], + counterexamples=["The dashboard is down", "A 502 on every request"], + ), + ( + f"{BILLING}\nExamples: Where is my refund?; My card was charged twice" + f"\nNot this option: The dashboard is down; A 502 on every request" + ), + ), ] GOLDEN_IDS = [ @@ -62,6 +76,7 @@ "examples_and_a_counterexample", "a_counterexample_only", "a_level_with_examples", + "declaration_order_is_kept", ] @@ -83,15 +98,19 @@ def test_a_rubric_with_no_parts_renders_byte_for_byte(what: str) -> None: assert render(fallback(what)) == what +DASHBOARD = "The dashboard is down" +"""An example of one option and a counterexample of another: the confusable pattern.""" + + class Department(Choice): """A choice whose every member carries examples, one of them the fallback.""" billing = option( BILLING, examples=["My card was charged twice", "Where is my refund?"], - counterexamples=["The dashboard is down"], + counterexamples=[DASHBOARD], ) - technical = option(TECHNICAL, examples=["502 on every request"]) + technical = option(TECHNICAL, examples=[DASHBOARD, "502 on every request"]) sales = fallback("Pricing, upgrades, new accounts", examples=["Do you have a team plan?"]) @@ -117,15 +136,16 @@ class Severity(Levels): "probabilities": {"0": 0.1, "1": 0.8, "2": 0.1}, "confidence": 0.9, }, + "q2": {"type": "noul", "noul": 0.2}, } """One answer per question of the batch below, over its own keys and levels.""" CHOICE_CRITERIA: Json = { "billing": ( f"{BILLING}\nExamples: My card was charged twice; Where is my refund?" - f"\nNot this option: The dashboard is down" + f"\nNot this option: {DASHBOARD}" ), - "technical": f"{TECHNICAL}\nExamples: 502 on every request", + "technical": f"{TECHNICAL}\nExamples: {DASHBOARD}; 502 on every request", "sales": "Pricing, upgrades, new accounts\nExamples: Do you have a team plan?", } """What `Department` must put on the wire, key by key.""" @@ -137,6 +157,12 @@ class Severity(Levels): ] """What `Severity` must put on the wire, low to high.""" +NOUL_CRITERIA: Json = { + "true": "Needs a person now\nExamples: the whole site is down", + "false": "Can wait\nExamples: a broken job someone has a manual workaround for", +} +"""What a noul's described criteria must put on the wire, under the wire's own names.""" + def test_examples_reach_the_wire_as_the_rendered_criteria( httpserver: HTTPServer, runner: Runner @@ -145,8 +171,12 @@ def test_examples_reach_the_wire_as_the_rendered_criteria( batch = ( choose(Department, "Which team should handle this?"), score(Severity, "How bad is it?"), + noul("Is this urgent?").criteria( + option("Needs a person now", examples=["the whole site is down"]), + option("Can wait", examples=["a broken job someone has a manual workaround for"]), + ), ) - assert runner.ask(batch, TICKET) == (Department.technical, Severity.degraded) + assert runner.ask(batch, TICKET) == (Department.technical, Severity.degraded, False) request, _ = httpserver.log[-1] body = as_object(narrow(request.get_json())) @@ -154,5 +184,7 @@ def test_examples_reach_the_wire_as_the_rendered_criteria( questions = as_object(body["questions"]) assert as_object(questions["q0"])["criteria"] == CHOICE_CRITERIA assert as_object(questions["q1"])["criteria"] == SCORE_CRITERIA - # A fallback marked with examples is still the fallback. + assert as_object(questions["q2"])["criteria"] == NOUL_CRITERIA + # A fallback marked with examples is still the fallback, and `The dashboard is down` + # went out as an example of one option and a counterexample of another. assert Department.fallback_member() is Department.sales diff --git a/tests/test_wire.py b/tests/test_wire.py index 7cb7a69..2856e58 100644 --- a/tests/test_wire.py +++ b/tests/test_wire.py @@ -567,6 +567,26 @@ def _a_runtime_level_with_counterexamples(_monkeypatch: pytest.MonkeyPatch) -> N _ = score_levels("How cross?", [option("Calm", counterexamples=["shouting"]), "Cross"]) +def _one_example_of_two_runtime_options(_monkeypatch: pytest.MonkeyPatch) -> None: + # One input cannot belong to two options. The reverse -- an example of one and a + # counterexample of another -- is legal and is what tells confusable options apart. + _ = choose_among( + "Which team?", + { + "billing": option("Money", examples=["Where is my refund?"]), + "technical": option("Bugs", examples=["Where is my refund?"]), + }, + ) + + +def _one_example_of_both_noul_criteria(_monkeypatch: pytest.MonkeyPatch) -> None: + # A yes and a no are two alternatives of one question, so the same rule holds there. + _ = noul("Urgent?").criteria( + option("Needs a person now", examples=["the export is broken"]), + option("Can wait", examples=["the export is broken"]), + ) + + def _events_log_without_the_logs_api(monkeypatch: pytest.MonkeyPatch) -> None: # Asking for log records where `opentelemetry-api` has no logs API is refused where # it is asked for. The alternative is a guide that emits none and never says so. @@ -606,6 +626,8 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: _an_empty_model, _levels_given_as_one_string, _a_runtime_level_with_counterexamples, + _one_example_of_two_runtime_options, + _one_example_of_both_noul_criteria, _events_log_without_the_logs_api, _events_both_without_the_logs_api, _events_log_with_a_drifted_log_record, @@ -620,6 +642,8 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: "empty_model", "levels_given_as_one_string", "a_runtime_level_with_counterexamples", + "one_example_of_two_runtime_options", + "one_example_of_both_noul_criteria", "events_log_without_the_logs_api", "events_both_without_the_logs_api", "events_log_with_a_drifted_log_record", From 4566bd8dbf566706fd6737d675a6070a9a2ffa9d Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 21:30:45 -0300 Subject: [PATCH 05/18] Default the example clauses to None, not an empty tuple Telling `examples=[]` from an argument left out used to be identity against the `()` default, which works only because CPython hands out one empty tuple. That is an implementation detail rather than a language guarantee, and the failure mode is silent: on a runtime that does not intern it, the caller mistake this refuses stops being caught and nothing says so. `None` is now the clause left out, so the distinction is a plain `is None` and rests on nothing. Any empty sequence that arrives was typed by the caller, which makes `examples=()` refused as well -- consistent, because an empty clause written out says nothing whatever type it is, and now pinned by its own case in the refusal list rather than left as an accident of the rule. --- AGENTS.md | 3 ++- CHANGELOG.md | 5 +++-- README.md | 7 ++++--- docs/design.md | 10 ++++++++-- src/guideme/enums.py | 34 +++++++++++++++++----------------- tests/test_enums.py | 11 +++++++++++ 6 files changed, 45 insertions(+), 25 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 831fe18..969c6da 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -119,7 +119,8 @@ the rest are checked in review. so is the order: examples and counterexamples render in the order written, never sorted and never de-duplicated into a set. - **A rubric's examples must be consistent, and that is checked where it is written.** A blank - or whitespace-only rubric or entry, a repeat within one clause, an `examples=[]` written out, + or whitespace-only rubric or entry, a repeat within one clause, an empty clause written out + (`examples` and `counterexamples` default to `None`, so any empty sequence was typed), and a counterexample on a level are each a `ConfigError`. So are the two contradictions: one string as an example of two alternatives of the same question, and one string as both an example and a counterexample of the same alternative. One string as an example of one diff --git a/CHANGELOG.md b/CHANGELOG.md index 85cbfff..9299d6f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,8 +24,9 @@ Additive. Nothing that worked in 0.1.0 sends different bytes. dropped by reaching a runtime constructor. - `level(…)` takes no counterexamples: "not this option" means nothing on an ordered scale. An `option(…)` carrying counterexamples written where a level belongs is a `ConfigError`, as is - a blank or whitespace-only rubric or entry, a repeat within one clause, and an `examples=[]` - written out. + a blank or whitespace-only rubric or entry, a repeat within one clause, and an empty clause + written out: `examples` and `counterexamples` default to `None`, so any empty sequence that + arrives was typed on purpose and says nothing. - Contradictory examples are a `ConfigError` too: one string as an example of two alternatives of the same question, or as both an example and a counterexample of the same alternative. One string as an example of one alternative and a counterexample of another stays legal — that is diff --git a/README.md b/README.md index 5aed7c3..d996fd8 100644 --- a/README.md +++ b/README.md @@ -154,9 +154,10 @@ option, says an input belongs where it cannot, so each is refused. `level(…)` has no counterexamples, because "not this option" means nothing on an ordered scale — an input that does not belong at one level scores at another. -A blank or whitespace-only rubric or entry, a repeat within one clause, an `examples=[]` -written out, a counterexample on a level, and the two contradictions above are each a -`ConfigError` where the rubric is written. +Leave a clause out to say there is none. An empty one written out — `examples=[]` or +`examples=()` — says nothing, so it is refused as the mistake it is, along with a blank or +whitespace-only rubric or entry, a repeat within one clause, a counterexample on a level, and +the two contradictions above. Each is a `ConfigError` where the rubric is written. The state is anything JSON-shaped: a text literal, a `dict`, a list of them. A dataclass goes through `dataclasses.asdict`, a pydantic model through `.model_dump()`. diff --git a/docs/design.md b/docs/design.md index c8a992d..dfeb60b 100644 --- a/docs/design.md +++ b/docs/design.md @@ -165,8 +165,14 @@ Each module survives the test. - **A rubric's value is its bare text, examples or not.** `option("x", examples=[…])` still equals `"x"`, so two options whose text matches are still one member however their examples differ, and a member's `.value` still reads as it was written. The expansion happens only in - the request. That also means an `examples=[]` written out is a `ConfigError`: it cannot be - told from the default by its effect, so it is refused as the mistake it is. + the request. +- **A clause left out is `None`; an empty one was written on purpose.** `examples` and + `counterexamples` default to `None`, which is how "there are none" is said, so any empty + sequence that arrives was typed by the caller and says nothing — a `ConfigError`, whatever + its type, `[]` and `()` alike. The alternative, defaulting to `()` and telling the two apart + by identity, would have rested on CPython interning the empty tuple: correct today, and + silently wrong on a runtime that does not, with a real caller mistake quietly no longer + caught. `is None` needs no such assumption. - **The four scalars are brands, not validated types.** `Probability`, `Confidence`, `Key` and `Rank` are `NewType`s, so `Probability(2.0)` and `Rank(99)` are accepted by the checker and by the interpreter alike. What makes them trustworthy is that only the wire mints them, and it diff --git a/src/guideme/enums.py b/src/guideme/enums.py index 6525c59..a4e787b 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -25,11 +25,6 @@ NOT_THIS = "Not this option: " """Opens the clause naming inputs that belong to some other option.""" -_UNSET: tuple[str, ...] = () -"""The `examples` and `counterexamples` default, and how an empty sequence written on -purpose is told from an argument left out: CPython hands out one empty tuple, so the -`examples=[]` this refuses is never this object.""" - @final class _Rubric(str): # noqa: SLOT000 -- a str subclass may not carry non-empty __slots__ @@ -63,9 +58,13 @@ def __new__( return self -def _items(what: str, items: Sequence[str], named: str) -> tuple[str, ...]: - """Check one clause's items: written means non-empty, each says something, no repeats.""" - if items is _UNSET: +def _items(what: str, items: Sequence[str] | None, named: str) -> tuple[str, ...]: + """Check one clause's items: written means non-empty, each says something, no repeats. + + `None` is the clause left out. Anything else that is empty was written on purpose + and says nothing, which is the caller's mistake rather than a default. + """ + if items is None: return () if not items: detail = f"{what!r}: {named}= is empty; a clause written on purpose must say something" @@ -83,8 +82,8 @@ def _items(what: str, items: Sequence[str], named: str) -> tuple[str, ...]: def _rubric( rubric: str, - examples: Sequence[str], - counterexamples: Sequence[str], + examples: Sequence[str] | None, + counterexamples: Sequence[str] | None, *, is_fallback: bool, ) -> str: @@ -108,8 +107,8 @@ def _rubric( def option( rubric: str, *, - examples: Sequence[str] = _UNSET, - counterexamples: Sequence[str] = _UNSET, + examples: Sequence[str] | None = None, + counterexamples: Sequence[str] | None = None, ) -> str: """A described alternative: what it covers, inputs that belong to it, inputs that do not. @@ -122,14 +121,15 @@ def option( written in. A string may be an example of one alternative and a counterexample of another: that is how two confusable ones are told apart. - A blank rubric or entry, an `examples=[]` written out, a repeat within one clause, + Leave a clause out to say there is none. An empty one written out says nothing and + is refused: a blank rubric or entry, an `examples=[]`, a repeat within one clause, and a string given as both an example and a counterexample of this one alternative are each a `ConfigError` where the option is written. """ return _rubric(rubric, examples, counterexamples, is_fallback=False) -def level(rubric: str, *, examples: Sequence[str] = _UNSET) -> str: +def level(rubric: str, *, examples: Sequence[str] | None = None) -> str: """A score level: what it means, and inputs that score here. An example listed under a level is the statement that such an input scores that @@ -137,14 +137,14 @@ def level(rubric: str, *, examples: Sequence[str] = _UNSET) -> str: counterexamples: "not this option" means nothing on an ordered scale, so the level below or above is what an input that does not belong here scores. """ - return _rubric(rubric, examples, _UNSET, is_fallback=False) + return _rubric(rubric, examples, None, is_fallback=False) def fallback( rubric: str, *, - examples: Sequence[str] = _UNSET, - counterexamples: Sequence[str] = _UNSET, + examples: Sequence[str] | None = None, + counterexamples: Sequence[str] | None = None, ) -> str: """Mark the member to use when the policy says unsure. At most one per `Choice`. diff --git a/tests/test_enums.py b/tests/test_enums.py index 76d07ab..118f772 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -85,6 +85,15 @@ class Empty(Choice): return Empty +def _an_empty_tuple_written_out() -> type[Choice]: + # `None` is how a clause is left out, so an empty sequence is always a clause + # written on purpose that says nothing -- whatever type the caller reached for. + class Empty(Choice): + a = option("Payments", counterexamples=()) + + return Empty + + def _a_blank_example() -> type[Choice]: class Blank(Choice): a = option("Payments", examples=["My card was charged twice", " "]) @@ -201,6 +210,7 @@ class Urgency(Levels): _fallback_on_a_level, _a_blank_rubric, _an_examples_clause_written_empty, + _an_empty_tuple_written_out, _a_blank_example, _a_repeated_example, _a_counterexample_on_a_level, @@ -220,6 +230,7 @@ class Urgency(Levels): "a_fallback_marker_on_a_levels", "a_blank_rubric", "an_examples_clause_written_empty", + "an_empty_tuple_written_out", "a_blank_example", "a_repeated_example", "a_counterexample_on_a_level", From 0211d5c893a15330a71ee22803f4e36bc1dc198b Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 21:35:44 -0300 Subject: [PATCH 06/18] Prove the noul half of the invariant in the live test The noul criteria path was proven to reach the wire by the wire assertion, but its effect on the answer was only ever measured by hand. It is the same invariant as the choice half -- examples move the distribution towards the alternative they describe -- so it folds into the existing live function rather than becoming a second one, and the entry count does not move. It is also the stronger half: a vague `Urgent` / `Not urgent` calls a broken nightly job with a manual workaround urgent at 0.75, and the same criteria carrying examples call it 0.17, because one of the not-urgent examples is what the ticket describes. Asserted as `told < plain`, never a float. --- AGENTS.md | 4 +++- tests/test_live.py | 35 +++++++++++++++++++++++++++++++---- 2 files changed, 34 insertions(+), 5 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 969c6da..50413ca 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -187,7 +187,9 @@ it asserts the absence of a failure on a path where every other test asserts a p existing test could carry it. Rubric examples added the last four, also user-requested scope: the golden rendering table, the passthrough property that proves 0.1.0's bytes have not moved, the wire proof that a rendered rubric reaches the request, and a live proof that examples move -the distribution. Each asserts a different thing about a string no existing test looks at. +the distribution — one function covering a choice and a noul, because both are the same +invariant and a second function would buy nothing. Each asserts a different thing about a +string no existing test looks at. Everything else those two passes added went into a parameter of a test that was already there. A new test must be one of: diff --git a/tests/test_live.py b/tests/test_live.py index 2c99894..56132c0 100644 --- a/tests/test_live.py +++ b/tests/test_live.py @@ -134,6 +134,18 @@ async def run() -> tuple[bool, Department, Scored[Frustration], dict[str, bool]] """The same two rubrics, with the examples that tell the two apart.""" +WORKAROUND = ( + "Our nightly export job has been failing since Tuesday. We pull the numbers by hand for now." +) +"""A ticket a vague `Urgent` / `Not urgent` calls urgent, and the examples call otherwise.""" + +URGENT = "Urgent" +NOT_URGENT = "Not urgent" + +URGENT_EXAMPLES = ["customers cannot log in", "money is moving to the wrong place"] +NOT_URGENT_EXAMPLES = ["a broken job with a manual workaround", "a cosmetic bug"] + + def _return_status(guide: Guide, options: dict[str, str]) -> float: ranked = guide.ask( choose_among("What is the customer asking about?", options).detail(), AMBIGUOUS @@ -141,17 +153,32 @@ def _return_status(guide: Guide, options: dict[str, str]) -> float: return dict(ranked.probabilities)[Key("return_status")] -def test_examples_move_the_distribution_towards_the_option_they_describe() -> None: +def _urgent(guide: Guide, yes: str, no: str) -> float: + verdict = guide.ask(noul("Is this ticket urgent?").criteria(yes, no).detail(), WORKAROUND) + return verdict.p + + +def test_examples_move_the_distribution_towards_the_alternative_they_describe() -> None: """The invariant the feature exists for, not a number the model is not stable to. - The rubric text is the same in both asks, so the examples are the only thing that - changed. Measured on 2026-09-21 against `jev-1.13.0`: 0.50 bare, 0.88 described, - over three runs each. + Both halves hold the rubric text constant across their two asks, so the examples + are the only thing that changed. Measured on 2026-09-21 against `jev-1.13.0`: a + choice over two vague options goes 0.50 bare to 0.88 described, and a noul over + vague criteria goes 0.75 plain to 0.17 described, three and four runs each. The + noul half is the one where the plain rubric is outright wrong: one of the + not-urgent examples is what this ticket describes. """ guide = Guide.from_env() try: bare = _return_status(guide, BARE) described = _return_status(guide, DESCRIBED) + plain = _urgent(guide, URGENT, NOT_URGENT) + told = _urgent( + guide, + option(URGENT, examples=URGENT_EXAMPLES, counterexamples=NOT_URGENT_EXAMPLES), + option(NOT_URGENT, examples=NOT_URGENT_EXAMPLES, counterexamples=URGENT_EXAMPLES), + ) finally: guide.close() assert described > bare + assert told < plain From 4dd456b3439485455f549b9b37f8ab4bda854dc3 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 22:08:05 -0300 Subject: [PATCH 07/18] Refuse a clause given as one string, and fallback off the runtime paths Two ways a rubric could be quietly wrong, both found in review. A `str` is a `Sequence[str]` of its own characters, so `examples="refund"` type-checked and rendered as six one-letter examples, and a multi-word string failed with the actively misleading "every entry in examples must say something, got ' '". This package already guards the identical trap two functions away, in `score_levels`, so the guard and its message now mirror that one. One check at the head of the clause validator covers all three constructors. `fallback(...)` was accepted and dropped by `choose_among`, `score_levels` and `noul(...).criteria(...)`, while the declarative `Levels` twin raises. Those three answer in a `Key`, a `Rank` and a `bool`, none of which has a member to fall back to, so a caller could believe an unsure answer was handled when it would raise. All three now refuse and point at `.otherwise(...)`, which is the rule `Levels` has always stated. Refusing rather than honouring it in `choose_among` keeps one story: `fallback(...)` marks a Choice member, and everywhere else the question carries the value. --- CHANGELOG.md | 8 ++++++++ src/guideme/enums.py | 28 +++++++++++++++++++++++++++- src/guideme/question.py | 10 +++++++++- tests/test_enums.py | 11 +++++++++++ tests/test_wire.py | 21 +++++++++++++++++++++ 5 files changed, 76 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9299d6f..19520ff 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -31,6 +31,14 @@ Additive. Nothing that worked in 0.1.0 sends different bytes. of the same question, or as both an example and a counterexample of the same alternative. One string as an example of one alternative and a counterexample of another stays legal — that is the confusable-options pattern the feature exists for. +- A clause given as one string is refused rather than shredded. A `str` is a `Sequence[str]` of + its own characters, so `examples="refund"` would have become six one-letter examples and no + type checker would have said so; it is a `ConfigError` naming the mistake, the same way + `score_levels` already refuses a scale given as one string. +- `fallback(…)` is refused by `choose_among`, `score_levels` and `noul(…).criteria(…)`. Those + answer in a `Key`, a `Rank` and a `bool`, none of which has a member to fall back to, so the + marking had nothing to act on and was being dropped in silence. Use `.otherwise(…)` on the + question, which is what the `Levels` rule has always said. ## 0.1.0 — 2026-09-21 diff --git a/src/guideme/enums.py b/src/guideme/enums.py index a4e787b..c1b1346 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -63,9 +63,19 @@ def _items(what: str, items: Sequence[str] | None, named: str) -> tuple[str, ... `None` is the clause left out. Anything else that is empty was written on purpose and says nothing, which is the caller's mistake rather than a default. + + A `str` is a `Sequence[str]` of its own characters, so `examples="refund"` would + quietly become six one-letter examples. It is a `ConfigError` instead, the same + way `score_levels` refuses a scale given as one string. """ if items is None: return () + if isinstance(items, str | bytes): + detail = ( + f"{what!r}: {named} must be a sequence of strings, got a single " + f"{type(items).__name__}; wrap it in a list" + ) + raise ConfigError(detail) if not items: detail = f"{what!r}: {named}= is empty; a clause written on purpose must say something" raise ConfigError(detail) @@ -93,7 +103,8 @@ def _rubric( raise ConfigError(detail) shown = _items(rubric, examples, "examples") excluded = _items(rubric, counterexamples, "counterexamples") - both = [item for item in shown if item in set(excluded)] + ruled_out = set(excluded) + both = [item for item in shown if item in ruled_out] if both: detail = ( f"{rubric!r}: {both[0]!r} is both an example and a counterexample of it, which " @@ -198,6 +209,21 @@ def require_unshared_examples(where: str, rubrics: Iterable[tuple[str, str | Non raise ConfigError(detail) +def require_no_fallback(where: str, rubric: str | None) -> None: + """Refuse a `fallback(...)` where there is no `Choice` member for it to mark. + + The runtime constructors answer in a `Key`, a `Rank` or a `bool`, none of which has + a member to fall back to, so the marking has nothing to act on. Dropping it would + leave a caller believing an unsure answer is handled when it raises instead. + """ + if isinstance(rubric, _Rubric) and rubric.is_fallback: + detail = ( + f"{where}: fallback(...) marks a member of a Choice and there is none here; " + f"use .otherwise(value) on the question" + ) + raise ConfigError(detail) + + def require_no_counterexamples(where: str, rubric: str) -> None: """Refuse counterexamples on a level, rather than rendering or dropping them. diff --git a/src/guideme/question.py b/src/guideme/question.py index db0d13d..dc42be0 100644 --- a/src/guideme/question.py +++ b/src/guideme/question.py @@ -15,6 +15,7 @@ Levels, render, require_no_counterexamples, + require_no_fallback, require_unshared_examples, ) from guideme.errors import ConfigError, ProtocolError, UnsureError @@ -337,6 +338,8 @@ def criteria(self, yes: str, no: str) -> Self: here: a yes and a no are as confusable as two options of a choice, and showing an input that belongs to each is what tells them apart. """ + require_no_fallback("criteria yes", yes) + require_no_fallback("criteria no", no) require_unshared_examples("criteria", (("yes", yes), ("no", no))) return replace(self, spec=NoulSpec(NoulCriteria(render(yes), render(no)))) @@ -446,8 +449,12 @@ def choose_among(instructions: Json, options: Mapping[str, str | None]) -> Choic them unique. A rubric is a string, `None`, or an `option(...)` carrying examples, which - are composed into it here: the same text a `Choice` member would send. + are composed into it here: the same text a `Choice` member would send. A + `fallback(...)` is a `ConfigError`: this answers in a `Key`, so there is no + member for the marking to name; use `.otherwise(...)` on the question. """ + for key, text in options.items(): + require_no_fallback(f"choose_among option {key!r}", text) require_unshared_examples("choose_among", options.items()) return _choice( instructions, @@ -497,6 +504,7 @@ def score_levels(instructions: Json, levels: Sequence[str]) -> ScoreQuestion[Ran raise ConfigError(detail) for index, text in enumerate(levels): require_no_counterexamples(f"level {index}", text) + require_no_fallback(f"level {index}", text) require_unshared_examples( "score_levels", ((f"level {index}", text) for index, text in enumerate(levels)) ) diff --git a/tests/test_enums.py b/tests/test_enums.py index 118f772..433799f 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -85,6 +85,15 @@ class Empty(Choice): return Empty +def _examples_given_as_one_string() -> type[Choice]: + # A `str` is a `Sequence[str]`, so the checker allows it and the clause would be + # the six letters of "refund". + class Shredded(Choice): + a = option("Payments", examples="refund") + + return Shredded + + def _an_empty_tuple_written_out() -> type[Choice]: # `None` is how a clause is left out, so an empty sequence is always a clause # written on purpose that says nothing -- whatever type the caller reached for. @@ -210,6 +219,7 @@ class Urgency(Levels): _fallback_on_a_level, _a_blank_rubric, _an_examples_clause_written_empty, + _examples_given_as_one_string, _an_empty_tuple_written_out, _a_blank_example, _a_repeated_example, @@ -230,6 +240,7 @@ class Urgency(Levels): "a_fallback_marker_on_a_levels", "a_blank_rubric", "an_examples_clause_written_empty", + "examples_given_as_one_string", "an_empty_tuple_written_out", "a_blank_example", "a_repeated_example", diff --git a/tests/test_wire.py b/tests/test_wire.py index 2856e58..f6201c5 100644 --- a/tests/test_wire.py +++ b/tests/test_wire.py @@ -567,6 +567,21 @@ def _a_runtime_level_with_counterexamples(_monkeypatch: pytest.MonkeyPatch) -> N _ = score_levels("How cross?", [option("Calm", counterexamples=["shouting"]), "Cross"]) +def _a_fallback_as_a_runtime_option(_monkeypatch: pytest.MonkeyPatch) -> None: + # `choose_among` answers in a `Key`, so there is no member for the marking to name. + # Accepting it and dropping it would leave the caller believing an unsure answer is + # handled when it raises instead. Same for the two below. + _ = choose_among("Which team?", {"sales": fallback("Pricing"), "billing": "Money"}) + + +def _a_fallback_as_a_runtime_level(_monkeypatch: pytest.MonkeyPatch) -> None: + _ = score_levels("How cross?", [fallback("Calm"), "Cross"]) + + +def _a_fallback_as_a_noul_criterion(_monkeypatch: pytest.MonkeyPatch) -> None: + _ = noul("Urgent?").criteria(fallback("Needs a person now"), "Can wait") + + def _one_example_of_two_runtime_options(_monkeypatch: pytest.MonkeyPatch) -> None: # One input cannot belong to two options. The reverse -- an example of one and a # counterexample of another -- is legal and is what tells confusable options apart. @@ -626,6 +641,9 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: _an_empty_model, _levels_given_as_one_string, _a_runtime_level_with_counterexamples, + _a_fallback_as_a_runtime_option, + _a_fallback_as_a_runtime_level, + _a_fallback_as_a_noul_criterion, _one_example_of_two_runtime_options, _one_example_of_both_noul_criteria, _events_log_without_the_logs_api, @@ -642,6 +660,9 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: "empty_model", "levels_given_as_one_string", "a_runtime_level_with_counterexamples", + "a_fallback_as_a_runtime_option", + "a_fallback_as_a_runtime_level", + "a_fallback_as_a_noul_criterion", "one_example_of_two_runtime_options", "one_example_of_both_noul_criteria", "events_log_without_the_logs_api", From 6bd96557473add9620a3a681a323e13c3b561ed8 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Mon, 21 Sep 2026 22:08:11 -0300 Subject: [PATCH 08/18] State the clause labels in the contract, and settle the capture's version docs/contract.md gave the separators and the verbatim rule but never the two literal labels or the order the clauses come in, and that file is what a third SDK implements from: as written, its author would pick their own labels and produce different bytes for the same declaration. The algorithm is now there in full, with `Examples: ` and `Not this option: ` quoted exactly and the reason the latter is not a formatting choice -- it was measured against `Not` and `Counterexamples` and reached the correct option most often. The observability capture gets its `0.1.0` back. Eliding it avoided inventing a version no run produced, which was the right instinct and the wrong fix: it also threw away the provenance. The version is the one the run emitted, the prose now says so, and the release procedure says to re-take the capture or change nothing rather than edit the number forward. Also recorded: that a blank bare rubric stays legal while `option(" ")` is refused, which is a version boundary rather than an oversight, and guideme-rust draws the same line from the other side. And that `render` and the `require_*` checks are internal despite living beside the three exported constructors. --- AGENTS.md | 11 +++++++---- README.md | 20 +++++++++++++++++--- docs/contract.md | 40 +++++++++++++++++++++++++++++++--------- docs/design.md | 10 ++++++++++ docs/observability.md | 17 ++++++++++------- tests/test_live.py | 15 ++++++++++----- tests/test_rubrics.py | 20 ++++++++++++++------ 7 files changed, 99 insertions(+), 34 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 50413ca..26aa66d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -40,7 +40,7 @@ two pages: `https://docs.typesafe.ai/api.md` covers `POST /v1/systemone` and | `src/guideme/scalars.py` | `Probability`, `Confidence`, `Key`, `Rank`, `ApiKey`, `Model` | validation happens once, here; `ApiKey` never prints | | `src/guideme/errors.py` | the `GuidemeError` tree and `kind` | `kind` is the cross-SDK name and the `error.type` value; imports nothing from `guideme` | | `src/guideme/policy.py` | `resolve`, `Policy`, `Thresholds`, `Verdict`, the answer and outcome dataclasses | pure: no I/O, no caller enums, keys and level indices only | -| `src/guideme/enums.py` | `Choice`, `Levels`, `option`, `level`, `fallback`, `render` | a member's name is its wire key and its value is its rubric; both validate at class definition, and a repeated rubric text is refused there. `render` is the one place a rubric's examples become wire text, and its output is a cross-SDK contract item | +| `src/guideme/enums.py` | `Choice`, `Levels`, `option`, `level`, `fallback`, and the internals `render` and the `require_*` checks | a member's name is its wire key and its value is its rubric; both validate at class definition, and a repeated rubric text is refused there. `render` is the one place a rubric's examples become wire text, and its output is a cross-SDK contract item. `render`, `require_unshared_examples`, `require_no_counterexamples` and `require_no_fallback` are internal despite their names: they are imported by `question.py` and are in no tier, like `question.validate` | | `src/guideme/question.py` | question kinds, constructors, `Ranked`, `Scored`, the unsure ladder | a question is inert until asked; the reader travels with it | | `src/guideme/ask.py` | shapes: `encode`, `decode`, `Plan` | ids are `q0..qN` in encounter order, insertion order for a dict | | `src/guideme/_ask_overloads.py` | the typed `ask` surfaces | GENERATED; edit `scripts/gen_ask_overloads.py` and run `mise run gen` | @@ -306,9 +306,12 @@ green before the tag, not after. 1. Bump `version` in `pyproject.toml`. Then `uv lock` at the root and `uv lock` inside `examples/otlp`: both lock files record the version, and both are resolved with `--locked`. 2. Move the `Unreleased` notes in `CHANGELOG.md` under the new version with today's date. -3. Refresh the **What arrives** capture in `docs/observability.md`, or elide the version in it. - Its `InstrumentationScope guideme X.Y.Z` lines carry the version the capture was taken at, - so they go stale on the first bump and a reader cannot tell a stale capture from a real one. +3. Leave the **What arrives** capture in `docs/observability.md` alone unless you re-take it. + Its `InstrumentationScope guideme X.Y.Z` lines carry the version the run actually emitted, + and the prose above it names that version, so a reader can tell the capture's age from a + claim about today. Do not edit the version forward to match a release no run produced, and + do not delete it either: that loses the provenance permanently. Re-take the capture and + update both together, or change nothing. 4. `mise run check`, then open a pull request and squash-merge it with CI green. 5. On `main`, at that commit: `git tag -a vX.Y.Z -m vX.Y.Z` and `git push origin vX.Y.Z`. The tag ruleset refuses a tag that is later moved or deleted, so tag the commit you mean. diff --git a/README.md b/README.md index d996fd8..4696607 100644 --- a/README.md +++ b/README.md @@ -142,9 +142,23 @@ Not this option: The dashboard is down ``` So a rubric with no examples sends exactly what it sent before, and the same strings work in -`choose_among("…", {"billing": option(…)})`, `score_levels("…", [level(…), …])` and -`noul("…").criteria(option(…), option(…))`. Examples and counterexamples render in the order -they are written, always: that order is part of the published contract. +`choose_among("…", {"billing": option(…)})` and `score_levels("…", [level(…), …])`. Examples +and counterexamples render in the order they are written, always: that order is part of the +published contract. + +A yes and a no are two alternatives of one question, so they take examples too, and this is +where they pay best — a vague pair is the easiest thing to get wrong: + +```python +urgent = noul("Is this ticket urgent?").criteria( + option("Urgent", examples=["customers cannot log in", "money is moving to the wrong place"]), + option("Not urgent", examples=["a broken job with a manual workaround", "a cosmetic bug"]), +) +``` + +Asked about a nightly export job that has been failing since Tuesday while the numbers are +pulled by hand, a plain `Urgent` / `Not urgent` answers yes at 0.75. The criteria above answer +no at 0.17, because one of the not-urgent examples is what the ticket describes. A string may be an example of one option and a counterexample of another. That is the point when two options are confusable, and it is the one overlap that stays legal. Offering the same diff --git a/docs/contract.md b/docs/contract.md index 2d9ce23..b34396b 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -18,15 +18,37 @@ changing the copy. The rendered rubric string is a contract item too. An alternative written with `option(…)`, `level(…)` or `fallback(…)` carries its examples beside its text, and the string those compose -into — clauses joined with a newline, items within one joined with `"; "`, the text verbatim, -and the text alone when there are no examples — is what goes on the wire. Every guideme SDK -composes the same string from the same parts, and `spec/vectors/rubric.json` is where the -renderers are held to each other. All three kinds of question carry them, a noul's yes and no -included. - -**Order is part of it.** Examples and counterexamples render in the order they were written, -never sorted and never collapsed into a set, because two SDKs ordering differently would send -different bytes for the same declaration. +into is what goes on the wire. Every guideme SDK composes the same string from the same parts, +and `spec/vectors/rubric.json` is where the renderers are held to each other. All three kinds +of question carry them, a noul's yes and no included. + +The algorithm, in full, because this is what a new SDK implements from: + +``` +render(what, examples, counterexamples) -> string: + if examples is empty and counterexamples is empty: + return what # unchanged, byte for byte + lines = [what] + if examples: + lines.append("Examples: " + join(examples, "; ")) + if counterexamples: + lines.append("Not this option: " + join(counterexamples, "; ")) + return join(lines, "\n") +``` + +The two labels are literal and exact: `"Examples: "` and `"Not this option: "`, each with its +trailing space. They are not formatting to be chosen locally — `Not this option` was measured +against `Not` and `Counterexamples` on the same confusable case and reached the correct option +with the highest mean probability of the three, so a different label is a different, and worse, +contract. The examples clause always precedes the counterexamples clause. + +`what` is used verbatim: never trimmed, never re-punctuated. Newline separation is what makes +that safe, since examples often end in `?` and a space-joined format would need a trailing `.` +that produces `Where is my refund?.`. + +**Order within a clause is part of it too.** Examples and counterexamples render in the order +they were written, never sorted and never collapsed into a set, because two SDKs ordering +differently would send different bytes for the same declaration. One asymmetry is deliberate. `choose_among` and `score_levels` here take an `option(…)` or a `level(…)` value; Rust's equivalents keep taking a plain string, because widening their diff --git a/docs/design.md b/docs/design.md index dfeb60b..eb3f51e 100644 --- a/docs/design.md +++ b/docs/design.md @@ -166,6 +166,16 @@ Each module survives the test. equals `"x"`, so two options whose text matches are still one member however their examples differ, and a member's `.value` still reads as it was written. The expansion happens only in the request. +- **A blank rubric is refused where examples can attach, and accepted where a bare string + always was.** `option(" ")` is a `ConfigError`, while a bare `""` reaching `choose_among`, + `score_levels` or a `Choice` member is not. The asymmetry is deliberate and it is a version + boundary, not an oversight: a bare blank rubric was legal in 0.1.0, and refusing it in a + patch release would break a caller who is doing nothing new. The new constructors have no + such history, so they are strict from the start. `guideme-rust` draws the same line from the + other side — its derive had made an empty `what` an unconditional compile error and narrowed + it to fire only when examples are attached, for this reason. A reader of both SDKs should + find the same rule: strict where examples are, unchanged where they are not. Revisit at a + major bump, together. - **A clause left out is `None`; an empty one was written on purpose.** `examples` and `counterexamples` default to `None`, which is how "there are none" is said, so any empty sequence that arrives was typed by the caller and says nothing — a `ConfigError`, whatever diff --git a/docs/observability.md b/docs/observability.md index 4136919..44ac878 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -328,9 +328,12 @@ a different backend. Captured from `otel/opentelemetry-collector-contrib` with the debug exporter and the `collector.yaml` above, running `examples/otlp` under `events("log")` against the live API. -Timestamps, `Flags`, the version on each `InstrumentationScope` line and the resource block are -trimmed throughout — the version because a capture is taken once and every release would -otherwise leave a number here that a reader cannot tell from a real one. The first two blocks are +Timestamps, `Flags` and the resource block are trimmed throughout. The capture was taken at +**0.1.0** and the `InstrumentationScope` lines carry that version, which is the one the run +actually emitted; it is left as it was rather than edited forward, so nothing here is a number +no run produced. Read a version older than the current release as the age of the capture, not +as a claim about today: the scope names are the contract, and their version is not. The first +two blocks are otherwise complete; the later ones are excerpts, cut to the lines each is making a point about, so a missing `Parent ID`, `Kind`, scope line or attribute there means it was cut, not that it was absent. Nothing is reworded, and no value is invented. @@ -338,7 +341,7 @@ was absent. Nothing is reworded, and no value is invented. The three-question batch is one ask span with one attempt under it: ``` -InstrumentationScope guideme +InstrumentationScope guideme 0.1.0 Span #1 Trace ID : 164107043647c42bc827fff022ec4308 Parent ID : @@ -358,7 +361,7 @@ Attributes: -> gen_ai.usage.input_tokens: Int(422) -> gen_ai.usage.output_tokens: Int(71) -InstrumentationScope guideme.api +InstrumentationScope guideme.api 0.1.0 Span #1 Trace ID : 164107043647c42bc827fff022ec4308 Parent ID : 53c4b9ada6417408 @@ -378,7 +381,7 @@ Attributes: Its three answers arrive on the logs pipeline, each carrying the ask span's ids: ``` -InstrumentationScope guideme +InstrumentationScope guideme 0.1.0 LogRecord #1 SeverityText: INFO SeverityNumber: Info(9) @@ -443,7 +446,7 @@ throttled attempt's rather than the ask's. The live API does not throttle on dem block is from a local server that answers `429` once: ``` -InstrumentationScope guideme.api +InstrumentationScope guideme.api 0.1.0 LogRecord #0 SeverityText: WARN SeverityNumber: Warn(13) diff --git a/tests/test_live.py b/tests/test_live.py index 56132c0..a6427a4 100644 --- a/tests/test_live.py +++ b/tests/test_live.py @@ -162,11 +162,16 @@ def test_examples_move_the_distribution_towards_the_alternative_they_describe() """The invariant the feature exists for, not a number the model is not stable to. Both halves hold the rubric text constant across their two asks, so the examples - are the only thing that changed. Measured on 2026-09-21 against `jev-1.13.0`: a - choice over two vague options goes 0.50 bare to 0.88 described, and a noul over - vague criteria goes 0.75 plain to 0.17 described, three and four runs each. The - noul half is the one where the plain rubric is outright wrong: one of the - not-urgent examples is what this ticket describes. + are the only thing that changed. Measured on 2026-09-21 against `jev-1.13.0`. + The choice half, three runs: 0.50 / 0.49 / 0.53 bare, 0.87 / 0.89 / 0.89 described. + The noul half, four runs: 0.75 / 0.75 / 0.74 / 0.76 plain, 0.17 every time with + examples. The noul half is the one where the plain rubric is outright wrong: one + of the not-urgent examples is what this ticket describes. + + The cross-SDK design measured the same noul case at 0.25 rather than 0.17. The + difference is that this ask gives each side counterexamples as well as examples, + which the design's probe did not; it is a stronger rubric, not a disagreement + between the two SDKs. """ guide = Guide.from_env() try: diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py index e2b75c3..21aa0d7 100644 --- a/tests/test_rubrics.py +++ b/tests/test_rubrics.py @@ -27,14 +27,16 @@ # The golden table of the cross-SDK design: the inputs, and the exact bytes both SDKs # render them to. `guideme-rust` reproduces this table from its derive macro, and # spec/vectors/rubric.json is where the two are held to each other. -GOLDEN: list[tuple[str, str]] = [ - (option(BILLING), BILLING), +GOLDEN: list[tuple[str, str, str]] = [ + (option(BILLING), BILLING, BILLING), ( option(TECHNICAL, examples=["502 on every request"]), + TECHNICAL, f"{TECHNICAL}\nExamples: 502 on every request", ), ( option(BILLING, examples=["My card was charged twice", "Where is my refund?"]), + BILLING, f"{BILLING}\nExamples: My card was charged twice; Where is my refund?", ), ( @@ -43,14 +45,17 @@ examples=["My card was charged twice"], counterexamples=["The dashboard is down"], ), + BILLING, f"{BILLING}\nExamples: My card was charged twice\nNot this option: The dashboard is down", ), ( option(BILLING, counterexamples=["The dashboard is down"]), + BILLING, f"{BILLING}\nNot this option: The dashboard is down", ), ( level(COSMETIC, examples=["typo in a label", "misaligned icon"]), + COSMETIC, f"{COSMETIC}\nExamples: typo in a label; misaligned icon", ), # Both clauses are written out of alphabetical order, so a renderer that sorted or @@ -62,6 +67,7 @@ examples=["Where is my refund?", "My card was charged twice"], counterexamples=["The dashboard is down", "A 502 on every request"], ), + BILLING, ( f"{BILLING}\nExamples: Where is my refund?; My card was charged twice" f"\nNot this option: The dashboard is down; A 502 on every request" @@ -80,12 +86,14 @@ ] -@pytest.mark.parametrize(("rubric", "expected"), GOLDEN, ids=GOLDEN_IDS) -def test_a_rubric_renders_the_bytes_the_contract_names(rubric: str, expected: str) -> None: +@pytest.mark.parametrize(("rubric", "bare", "expected"), GOLDEN, ids=GOLDEN_IDS) +def test_a_rubric_renders_the_bytes_the_contract_names( + rubric: str, bare: str, expected: str +) -> None: assert render(rubric) == expected # The value itself stays the bare text: a rubric only expands where it becomes - # wire text, so a member's value reads as it is written. - assert str(rubric) in {BILLING, TECHNICAL, COSMETIC} + # wire text, so a member's value reads exactly as it was written. + assert str(rubric) == bare @given(st.text(min_size=1).filter(lambda text: bool(text.strip()))) From 9997dcfc92a4ccea0d4130f1f2fbdd1b42d372c2 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:40:35 -0300 Subject: [PATCH 09/18] Refuse a blank rubric only when examples are attached to it Design 9.7 supersedes 4 here, and this build was on the wrong side of it: `option("")` and `option(" ")` raised whether or not anything was attached. That makes a declaration which was legal in 0.1.0 illegal in a patch release, on a degenerate input that was already meaningless, and it diverges from guideme-rust, which narrowed the same rule from the same starting point. The contract says one declaration is legal in both SDKs or in neither. So the check now fires only when a clause is present, and says what the mistake actually is: examples were attached to nothing, rather than the description being blank -- which is not an error this release gets to invent. The passthrough property drops its non-blank filter as a result. It now runs over unrestricted text, which is the honest statement of the invariant: a rubric carrying no examples renders to its own bytes, whatever those bytes are. --- AGENTS.md | 9 ++++++--- CHANGELOG.md | 7 ++++--- README.md | 12 +++++++++--- docs/design.md | 19 +++++++++---------- src/guideme/enums.py | 25 ++++++++++++++++++------- tests/test_enums.py | 10 ++++++---- tests/test_rubrics.py | 8 +++++++- 7 files changed, 59 insertions(+), 31 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 26aa66d..1f55a6b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -119,9 +119,12 @@ the rest are checked in review. so is the order: examples and counterexamples render in the order written, never sorted and never de-duplicated into a set. - **A rubric's examples must be consistent, and that is checked where it is written.** A blank - or whitespace-only rubric or entry, a repeat within one clause, an empty clause written out - (`examples` and `counterexamples` default to `None`, so any empty sequence was typed), - and a counterexample on a level are each a `ConfigError`. So are the two contradictions: one + or whitespace-only entry, a repeat within one clause, a clause given as one string rather + than a sequence of them, an empty clause written out (`examples` and `counterexamples` + default to `None`, so any empty sequence was typed), examples attached to a blank rubric, + a `fallback(…)` on a runtime path, and a counterexample on a level are each a `ConfigError`. + A blank rubric that carries no examples is **not** an error: it means what it meant in 0.1.0, + and this release does not redefine it. So are the two contradictions: one string as an example of two alternatives of the same question, and one string as both an example and a counterexample of the same alternative. One string as an example of one alternative and a counterexample of another is **legal and required** — it is the confusable diff --git a/CHANGELOG.md b/CHANGELOG.md index 19520ff..48e41d6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,9 +24,10 @@ Additive. Nothing that worked in 0.1.0 sends different bytes. dropped by reaching a runtime constructor. - `level(…)` takes no counterexamples: "not this option" means nothing on an ordered scale. An `option(…)` carrying counterexamples written where a level belongs is a `ConfigError`, as is - a blank or whitespace-only rubric or entry, a repeat within one clause, and an empty clause - written out: `examples` and `counterexamples` default to `None`, so any empty sequence that - arrives was typed on purpose and says nothing. + a blank or whitespace-only entry, a repeat within one clause, examples attached to a blank + rubric, and an empty clause written out: `examples` and `counterexamples` default to `None`, + so any empty sequence that arrives was typed on purpose and says nothing. A blank rubric + carrying no examples is untouched — it means what it meant in 0.1.0. - Contradictory examples are a `ConfigError` too: one string as an example of two alternatives of the same question, or as both an example and a counterexample of the same alternative. One string as an example of one alternative and a counterexample of another stays legal — that is diff --git a/README.md b/README.md index 4696607..5f1dd82 100644 --- a/README.md +++ b/README.md @@ -169,9 +169,15 @@ option, says an input belongs where it cannot, so each is refused. scale — an input that does not belong at one level scores at another. Leave a clause out to say there is none. An empty one written out — `examples=[]` or -`examples=()` — says nothing, so it is refused as the mistake it is, along with a blank or -whitespace-only rubric or entry, a repeat within one clause, a counterexample on a level, and -the two contradictions above. Each is a `ConfigError` where the rubric is written. +`examples=()` — says nothing, so it is refused as the mistake it is, along with a clause given +as one string rather than a list of them (`examples="refund"` would otherwise be six one-letter +examples), a blank entry, a repeat within one clause, a counterexample on a level, a +`fallback(…)` given to `choose_among`, `score_levels` or `.criteria(…)`, and the two +contradictions above. Each is a `ConfigError` where the rubric is written. + +Attaching examples to a blank rubric is refused too, because they describe something that is not +there. A blank rubric on its own is not: it means what it has always meant, and adding examples +to the language does not make an old declaration an error. The state is anything JSON-shaped: a text literal, a `dict`, a list of them. A dataclass goes through `dataclasses.asdict`, a pydantic model through `.model_dump()`. diff --git a/docs/design.md b/docs/design.md index eb3f51e..9fe13d9 100644 --- a/docs/design.md +++ b/docs/design.md @@ -166,16 +166,15 @@ Each module survives the test. equals `"x"`, so two options whose text matches are still one member however their examples differ, and a member's `.value` still reads as it was written. The expansion happens only in the request. -- **A blank rubric is refused where examples can attach, and accepted where a bare string - always was.** `option(" ")` is a `ConfigError`, while a bare `""` reaching `choose_among`, - `score_levels` or a `Choice` member is not. The asymmetry is deliberate and it is a version - boundary, not an oversight: a bare blank rubric was legal in 0.1.0, and refusing it in a - patch release would break a caller who is doing nothing new. The new constructors have no - such history, so they are strict from the start. `guideme-rust` draws the same line from the - other side — its derive had made an empty `what` an unconditional compile error and narrowed - it to fire only when examples are attached, for this reason. A reader of both SDKs should - find the same rule: strict where examples are, unchanged where they are not. Revisit at a - major bump, together. +- **A blank rubric is an error only when examples are attached to it.** `option(" ")` on its + own is accepted and means exactly what a bare `""` member has always meant; + `option(" ", examples=[…])` is a `ConfigError`. The first version of this refused any blank + rubric written through the new constructors, which read as tidy and was wrong: it made a + declaration that was legal in 0.1.0 illegal in a patch release, on a degenerate input that + was already meaningless. "You attached examples to nothing" is the real mistake; "your + description is blank" is not one this release gets to invent. `guideme-rust` narrowed the + same rule from the same starting point, so a reader of both finds one rule: strict where + examples are, untouched where they are not. Revisit at a major bump, together. - **A clause left out is `None`; an empty one was written on purpose.** `examples` and `counterexamples` default to `None`, which is how "there are none" is said, so any empty sequence that arrives was typed by the caller and says nothing — a `ConfigError`, whatever diff --git a/src/guideme/enums.py b/src/guideme/enums.py index c1b1346..8ff1abd 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -97,12 +97,21 @@ def _rubric( *, is_fallback: bool, ) -> str: - """Check the parts and hold them on the rubric, unrendered.""" - if not rubric.strip(): - detail = f"a rubric must say something, got {rubric!r}" - raise ConfigError(detail) + """Check the parts and hold them on the rubric, unrendered. + + A blank rubric is refused only when examples are attached to it. Attaching them + to nothing is the mistake; a blank rubric on its own was legal in 0.1.0, means + exactly what a bare `""` member means, and a patch release does not get to make + it an error. + """ shown = _items(rubric, examples, "examples") excluded = _items(rubric, counterexamples, "counterexamples") + if (shown or excluded) and not rubric.strip(): + detail = ( + f"examples were attached to a blank rubric ({rubric!r}); they say what the " + f"rubric covers, so there has to be something for them to say it about" + ) + raise ConfigError(detail) ruled_out = set(excluded) both = [item for item in shown if item in ruled_out] if both: @@ -133,9 +142,11 @@ def option( another: that is how two confusable ones are told apart. Leave a clause out to say there is none. An empty one written out says nothing and - is refused: a blank rubric or entry, an `examples=[]`, a repeat within one clause, - and a string given as both an example and a counterexample of this one alternative - are each a `ConfigError` where the option is written. + is refused: an `examples=[]`, a clause given as one string rather than a sequence of + them, a blank entry, a repeat within one clause, a string given as both an example + and a counterexample of this one alternative, and examples attached to a blank + rubric are each a `ConfigError` where the option is written. A blank rubric with no + examples is not: that is what it has always meant. """ return _rubric(rubric, examples, counterexamples, is_fallback=False) diff --git a/tests/test_enums.py b/tests/test_enums.py index 433799f..4635b5d 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -71,9 +71,11 @@ class OneLevel(Levels): return OneLevel -def _a_blank_rubric() -> type[Choice]: +def _examples_attached_to_a_blank_rubric() -> type[Choice]: + # The blank rubric alone is legal and means what it always meant. Attaching + # examples to it is the mistake: they describe something that is not there. class Blank(Choice): - a = option(" ") + a = option(" ", examples=["My card was charged twice"]) return Blank @@ -217,7 +219,7 @@ class Urgency(Levels): _duplicate_choice_rubric, _duplicate_level_rubric, _fallback_on_a_level, - _a_blank_rubric, + _examples_attached_to_a_blank_rubric, _an_examples_clause_written_empty, _examples_given_as_one_string, _an_empty_tuple_written_out, @@ -238,7 +240,7 @@ class Urgency(Levels): "two_options_with_the_same_rubric", "two_levels_with_the_same_rubric", "a_fallback_marker_on_a_levels", - "a_blank_rubric", + "examples_attached_to_a_blank_rubric", "an_examples_clause_written_empty", "examples_given_as_one_string", "an_empty_tuple_written_out", diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py index 21aa0d7..32c1822 100644 --- a/tests/test_rubrics.py +++ b/tests/test_rubrics.py @@ -96,10 +96,16 @@ def test_a_rubric_renders_the_bytes_the_contract_names( assert str(rubric) == bare -@given(st.text(min_size=1).filter(lambda text: bool(text.strip()))) +@given(st.text()) def test_a_rubric_with_no_parts_renders_byte_for_byte(what: str) -> None: # The load-bearing invariant: 0.1.0's bytes do not move. A bare string and an # `option(...)` with nothing attached both render to the text itself. + # + # The strategy is unrestricted on purpose, blank and whitespace-only text + # included. A rubric carrying no examples is refused for nothing at all: it + # meant whatever it meant in 0.1.0 and a patch release does not redefine it. + # Attaching examples to a blank rubric is the error, and that case is in + # `tests/test_enums.py`'s refusal list. assert render(what) == what assert render(option(what)) == what assert render(level(what)) == what From 6f52cc9732135a1fbf8cd71ccd2cba71a19cc8a0 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:40:37 -0300 Subject: [PATCH 10/18] Pin the vector's shape, blankness and duplicate equality in the contract Three things a third implementer would otherwise have to guess, and would guess differently, closed identically in both SDKs per design 9.8. The vector's `kind` field is named alongside `what`, `examples`, `counterexamples` and `rendered`, including that the `noul` cases come from the runtime renderer rather than a derive and are the only cover for that path. Blankness is Unicode `White_Space`, which is exactly Rust's `str::trim`. Python strips a superset, and the difference is stated rather than left to be discovered: measured against this runtime, 29 codepoints to 25, the four extra being `U+001C`-`U+001F`. They are C0 controls that are never valid rubric text, so no real declaration reaches the boundary -- but an implementer comparing the two SDKs would otherwise find it and have no way to tell deliberate from accidental. Duplicate detection is exact-string equality. `"a"` and `" a"` may coexist. An implementation that normalised or trimmed before comparing would refuse declarations these SDKs accept, which is the same divergence as a different clause label reached by another route. --- docs/contract.md | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/docs/contract.md b/docs/contract.md index b34396b..fd5954f 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -50,6 +50,35 @@ that produces `Where is my refund?.`. they were written, never sorted and never collapsed into a set, because two SDKs ordering differently would send different bytes for the same declaration. +`spec/vectors/rubric.json` carries one object per case with five fields: `kind`, one of +`choice`, `levels` or `noul`, which says what surface the case came from; `what`, the rubric +text; `examples` and `counterexamples`, the two clauses in declaration order; and `rendered`, +the exact bytes the algorithm above must produce. The `noul` cases come from the runtime +renderer rather than a derive, so they are the vector's only cover for that path. + +### What counts as blank, and what counts as a duplicate + +Two rules a third implementer would otherwise have to guess, and would guess differently. + +**A blank rubric is only an error when examples are attached to it.** A rubric that carries no +examples is never refused for its text, whatever that text is: it means what it meant before +this feature existed, and a patch release does not get to redefine it. Attaching examples to a +blank rubric is the error, because they describe something that is not there. Both SDKs draw +the line in the same place — Python inside `option()`, `level()` and `fallback()`, Rust in the +derive and in `Rubric::into_wire()` — so a declaration is legal in both or in neither. + +**"Blank" means Unicode `White_Space`.** Rust's `str::trim` is exactly that property. Python's +`str.strip()` is a superset: measured against the current runtime it strips 29 codepoints to +`White_Space`'s 25, the four extra being `U+001C`–`U+001F`, the file, group, record and unit +separators. So a rubric made only of those characters is blank to Python and not to Rust. That +boundary is stated rather than hidden, and it is deliberate: those four are C0 controls that are +never valid rubric text, so no real declaration reaches the difference. + +**A duplicate is an exact string match**, with no normalisation, no case folding and no +trimming. `"a"` and `" a"` are two different examples and may sit in the same clause. An +implementation that normalised before comparing would refuse declarations these SDKs accept, +which is the same divergence as a different label by another route. + One asymmetry is deliberate. `choose_among` and `score_levels` here take an `option(…)` or a `level(…)` value; Rust's equivalents keep taking a plain string, because widening their signatures risks inference breakage for existing callers on a path that can already pass a From 4840a8fa1585d4d64a1d6f373fb57b4a95dfe4e0 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:43:54 -0300 Subject: [PATCH 11/18] Name the C0 block exactly, and why the two checks disagree about " a" The whitespace paragraph already said C0, but only in passing at the end. It now gives both block ranges, because the whole reason this rule is written down is that two SDKs must not describe the boundary differently, and naming the wrong block would be precisely that failure. It also records that nothing goes the other way: every Unicode White_Space codepoint is one Python strips. The emptiness check trims and the duplicate check does not, so `" a"` is not blank yet is a different entry from `"a"`. That reads like a bug until the two questions are separated: emptiness asks whether the caller wrote anything, where a leading space changes nothing, and duplication asks whether two entries put the same bytes in front of the model, where it changes everything, because the text is rendered verbatim. --- docs/contract.md | 24 ++++++++++++++++++++---- 1 file changed, 20 insertions(+), 4 deletions(-) diff --git a/docs/contract.md b/docs/contract.md index fd5954f..28f6d02 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -69,16 +69,32 @@ derive and in `Rubric::into_wire()` — so a declaration is legal in both or in **"Blank" means Unicode `White_Space`.** Rust's `str::trim` is exactly that property. Python's `str.strip()` is a superset: measured against the current runtime it strips 29 codepoints to -`White_Space`'s 25, the four extra being `U+001C`–`U+001F`, the file, group, record and unit -separators. So a rubric made only of those characters is blank to Python and not to Rust. That -boundary is stated rather than hidden, and it is deliberate: those four are C0 controls that are -never valid rubric text, so no real declaration reaches the difference. +`White_Space`'s 25, and the four extra are `U+001C`, `U+001D`, `U+001E` and `U+001F` — file, +group, record and unit separator. Nothing goes the other way: every `White_Space` codepoint is +one Python strips. + +Those four are **C0** controls. The C0 block is `U+0000`–`U+001F`; C1 is `U+0080`–`U+009F` and +contains none of them. The distinction is worth stating because the point of writing this rule +down is that two SDKs must not describe the boundary differently, and naming the wrong block +would do exactly that. + +So a rubric made only of those four is blank to Python and not to Rust. The difference is +documented rather than hidden, and it is harmless: C0 controls are never valid rubric text, so +no real declaration reaches it. **A duplicate is an exact string match**, with no normalisation, no case folding and no trimming. `"a"` and `" a"` are two different examples and may sit in the same clause. An implementation that normalised before comparing would refuse declarations these SDKs accept, which is the same divergence as a different label by another route. +**The two checks disagree about `" a"`, and they are meant to.** The emptiness check trims +before deciding, so `" a"` is not blank and `" "` is; the duplicate check does not trim, so +`" a"` and `"a"` are two entries. That reads like a bug until you see what each one is asking. +Emptiness asks whether the caller wrote anything at all, and a leading space does not change +the answer. Duplication asks whether two entries would put the same bytes in front of the +model, and a leading space does change that, because the text is rendered verbatim. Trimming +for one and not the other is the only pairing that keeps both questions honest. + One asymmetry is deliberate. `choose_among` and `score_levels` here take an `option(…)` or a `level(…)` value; Rust's equivalents keep taking a plain string, because widening their signatures risks inference breakage for existing callers on a path that can already pass a From c7f6822200335c6347af4bda537e52202eeb95af Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:54:07 -0300 Subject: [PATCH 12/18] Make a rubric copyable and picklable again Carrying the parts on the string meant `__new__` took four arguments where `str.__getnewargs__` hands back one, so `copy.copy`, `copy.deepcopy` and `pickle` raised `TypeError: _Rubric.__new__() missing 2 required positional arguments`. A bare string did all three in 0.1.0, so this was a regression in a release whose whole premise is that nothing a 0.1.0 caller does changes. Enum members hid it: `Enum.__deepcopy__` returns self, so a declared `Choice` was fine and only bare values broke -- including `copy.deepcopy({"key": option(...)})`, which is exactly the dict a caller builds before handing it to `choose_among`. `__getnewargs_ex__` supplies all four, `is_fallback` included, so a copied fallback is still the fallback. Covered where the shapes already are: every golden row now round-trips through all three operations, and the choice test builds an enum from copied rubrics and checks both the rendering and `fallback_member()`. Error vocabulary aligned with the Rust SDK while here: "empty" and "duplicate" as the nouns, and the duplicate message now names the repeated string rather than only saying that one exists. --- src/guideme/enums.py | 26 +++++++++++++++++++++----- tests/test_enums.py | 14 ++++++++++++++ tests/test_rubrics.py | 8 ++++++++ 3 files changed, 43 insertions(+), 5 deletions(-) diff --git a/src/guideme/enums.py b/src/guideme/enums.py index 8ff1abd..2977341 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -57,6 +57,19 @@ def __new__( self.is_fallback = is_fallback return self + def __getnewargs_ex__( + self, + ) -> tuple[tuple[str, tuple[str, ...], tuple[str, ...]], dict[str, bool]]: + """Hand `copy` and `pickle` every argument `__new__` needs, marking included. + + `str`'s own `__getnewargs__` supplies the text alone, which is one argument + where this takes four, so without this a copied or pickled rubric raised a + `TypeError`. A bare `str` rubric copied fine in 0.1.0 and has to keep doing so: + `copy.deepcopy({"key": option(...)})` is exactly the shape a caller builds + before handing it to `choose_among`. + """ + return (str(self), self.examples, self.counterexamples), {"is_fallback": self.is_fallback} + def _items(what: str, items: Sequence[str] | None, named: str) -> tuple[str, ...]: """Check one clause's items: written means non-empty, each says something, no repeats. @@ -77,16 +90,19 @@ def _items(what: str, items: Sequence[str] | None, named: str) -> tuple[str, ... ) raise ConfigError(detail) if not items: - detail = f"{what!r}: {named}= is empty; a clause written on purpose must say something" + detail = f"{what!r}: empty {named}=; a clause written on purpose must say something" raise ConfigError(detail) values = tuple(items) blank = [item for item in values if not item.strip()] if blank: - detail = f"{what!r}: every entry in {named} must say something, got {blank[0]!r}" - raise ConfigError(detail) - if len(set(values)) != len(values): - detail = f"{what!r}: {named} repeats an entry; each must be distinct" + detail = f"{what!r}: empty {named[:-1]} {blank[0]!r}; every entry must say something" raise ConfigError(detail) + seen: set[str] = set() + for item in values: + if item in seen: + detail = f"{what!r}: duplicate {named[:-1]} {item!r}; each entry must be distinct" + raise ConfigError(detail) + seen.add(item) return values diff --git a/tests/test_enums.py b/tests/test_enums.py index 4635b5d..f738851 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -1,3 +1,4 @@ +import copy from collections.abc import Callable from typing import cast @@ -187,6 +188,19 @@ class Department(Choice): assert Department.from_key("technical") is Department.technical assert Department.from_key("marketing") is None + # A rubric survives being copied, parts and fallback marking alike. `copy.deepcopy` + # of a dict of options is what a caller builds before `choose_among`, and a bare + # string copied fine before this feature existed. + class Copied(Choice): + billing = copy.deepcopy(option("Payments", examples=["My card was charged twice"])) + sales = copy.deepcopy(fallback("Pricing")) + + assert Copied.rubric() == ( + ("billing", "Payments\nExamples: My card was charged twice"), + ("sales", "Pricing"), + ) + assert Copied.fallback_member() is Copied.sales + def test_levels_are_totally_ordered_by_declaration() -> None: class Frustration(Levels): diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py index 32c1822..2d58590 100644 --- a/tests/test_rubrics.py +++ b/tests/test_rubrics.py @@ -1,3 +1,6 @@ +import copy +import pickle + import pytest from hypothesis import given from hypothesis import strategies as st @@ -94,6 +97,11 @@ def test_a_rubric_renders_the_bytes_the_contract_names( # The value itself stays the bare text: a rubric only expands where it becomes # wire text, so a member's value reads exactly as it was written. assert str(rubric) == bare + # And it survives the three ways a value gets duplicated. `__new__` takes four + # arguments where `str` hands back one, so each of these raised a `TypeError` + # until the rubric said how to rebuild itself. + for clone in (copy.copy(rubric), copy.deepcopy(rubric), pickle.loads(pickle.dumps(rubric))): # noqa: S301 -- the payload is this test's own value + assert render(clone) == expected @given(st.text()) From 30d08287fc2485475d02495b6c019c8834c5ee62 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:54:08 -0300 Subject: [PATCH 13/18] State every legality rule in the contract, and undate the changelog docs/contract.md gave the blankness and duplicate-equality definitions but not the rules they serve, so a reader of only this file got the hard cases and none of the ordinary ones. All eight are now listed, including the one that must be allowed -- a string as an example of one alternative and a counterexample of another -- because an implementation that refused it would break the case the feature exists for, and a list of refusals with no permission in it invites exactly that. Also stated: items go in verbatim, nothing is escaped, and the rendering is not reversible. No SDK parses a rendered rubric back into its parts and none should be written to. The changelog had its notes under a dated 0.1.1 while nothing is published. They move back under Unreleased, where this repo's own release procedure says the release commit picks them up and sets the date. --- CHANGELOG.md | 9 +++++---- docs/contract.md | 24 ++++++++++++++++++++++++ 2 files changed, 29 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 48e41d6..ccccf68 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,10 +2,6 @@ ## Unreleased -Nothing yet. - -## 0.1.1 — 2026-09-21 - Additive. Nothing that worked in 0.1.0 sends different bytes. - `option(rubric, examples=…, counterexamples=…)` and `level(rubric, examples=…)` join @@ -40,6 +36,11 @@ Additive. Nothing that worked in 0.1.0 sends different bytes. answer in a `Key`, a `Rank` and a `bool`, none of which has a member to fall back to, so the marking had nothing to act on and was being dropped in silence. Use `.otherwise(…)` on the question, which is what the `Levels` rule has always said. +- A rubric built by `option(…)`, `level(…)` or `fallback(…)` can be copied and pickled again. + Carrying the parts meant `__new__` took four arguments where `str` hands back one, so + `copy.copy`, `copy.deepcopy` and `pickle` raised a `TypeError` — including on the + `copy.deepcopy({"key": option(…)})` a caller writes before `choose_among`. A bare string did + this in 0.1.0 and does it again. ## 0.1.0 — 2026-09-21 diff --git a/docs/contract.md b/docs/contract.md index 28f6d02..acce76a 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -56,6 +56,30 @@ text; `examples` and `counterexamples`, the two clauses in declaration order; an the exact bytes the algorithm above must produce. The `noul` cases come from the runtime renderer rather than a derive, so they are the vector's only cover for that path. +Items are inserted **verbatim**. Nothing is escaped: an item containing `"; "` or a newline is +placed in the clause exactly as written, and the rendering is not required to be reversible — +no SDK parses a rendered rubric back into its parts, and none should be written to. + +### Which declarations are legal + +These rules hold in every SDK, and each is refused where the rubric is written rather than at +ask time. + +- An **empty** clause written out is refused. Leaving a clause off is how you say there is + none; an empty sequence is a clause the caller wrote that says nothing. +- An **empty** entry within a clause is refused. +- A **duplicate** entry within one clause is refused. +- One string as an example of **two alternatives of the same question** is refused: it says one + input belongs to both, which cannot be true. +- One string as **both an example and a counterexample of one alternative** is refused: it says + the input does and does not belong there. +- One string as an example of one alternative and a **counterexample of another must be + allowed**. This is the confusable-alternatives pattern the feature exists to serve, and an + implementation that refused it would break the main use case. +- Examples are **attached to a blank rubric** is refused; a blank rubric with no examples is + not. See below. +- A counterexample on a **level** is refused: an ordered scale has no "not this option". + ### What counts as blank, and what counts as a duplicate Two rules a third implementer would otherwise have to guess, and would guess differently. From 3cf6a2f09f73b54ecb4fb0af23e6a62f056f0bd6 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:58:50 -0300 Subject: [PATCH 14/18] Refuse a line break inside an example or counterexample Items are joined onto one line with "; ", so a newline in one reads as a clause the rubric never declared -- an example ending "x\nNot this option: anything" renders byte-identically to a counterexample clause that was never written. `U+000D` goes with it as hygiene: pasted-text residue that breaks the joined line. Items only. A newline in the rubric text stays legal, because a bare rubric has to stay legal whatever it holds, so refusing it in the clause-bearing case would stop a caller writing ordinary multi-line prose without closing any path -- a runtime rubric built from data still passes its text through verbatim either way. What the rule guarantees is narrower than input trust and is now stated as such: a string rendered from declared parts carries exactly the clauses those parts declared. The test is the literal codepoint, never `str.splitlines()`. That call is what "does this contain a line break?" means in Python and it splits on eight codepoints, where Rust's `lines()` splits on one, so the idiomatic spelling in each language produces two different rules and neither looks wrong read alone. The contract now names the codepoints and forbids the idiom. Blank still wins over this, which strip-first already gave: an item of one newline is empty, not broken. --- AGENTS.md | 5 ++++- CHANGELOG.md | 4 ++++ README.md | 12 +++++++++--- src/guideme/enums.py | 12 ++++++++++++ tests/test_enums.py | 20 ++++++++++++++++++++ 5 files changed, 49 insertions(+), 4 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 1f55a6b..ed60543 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -122,7 +122,10 @@ the rest are checked in review. or whitespace-only entry, a repeat within one clause, a clause given as one string rather than a sequence of them, an empty clause written out (`examples` and `counterexamples` default to `None`, so any empty sequence was typed), examples attached to a blank rubric, - a `fallback(…)` on a runtime path, and a counterexample on a level are each a `ConfigError`. + a `fallback(…)` on a runtime path, a `U+000A` or `U+000D` inside an entry, and a + counterexample on a level are each a `ConfigError`. The newline test is the literal + codepoint, never `str.splitlines()`, which splits on eight and would refuse declarations the + Rust SDK accepts; `docs/contract.md` records why. A blank rubric that carries no examples is **not** an error: it means what it meant in 0.1.0, and this release does not redefine it. So are the two contradictions: one string as an example of two alternatives of the same question, and one string as both an diff --git a/CHANGELOG.md b/CHANGELOG.md index ccccf68..529f0b8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -36,6 +36,10 @@ Additive. Nothing that worked in 0.1.0 sends different bytes. answer in a `Key`, a `Rank` and a `bool`, none of which has a member to fall back to, so the marking had nothing to act on and was being dropped in silence. Use `.otherwise(…)` on the question, which is what the `Levels` rule has always said. +- An example or counterexample containing `U+000A` or `U+000D` is refused. Items are joined + onto one line, so a newline inside one would read as a clause the rubric never declared. The + rubric text itself is unrestricted; only the entries are. `"; "` inside an entry stays legal, + because it changes how many examples a reader sees rather than which clause they are in. - A rubric built by `option(…)`, `level(…)` or `fallback(…)` can be copied and pickled again. Carrying the parts meant `__new__` took four arguments where `str` hands back one, so `copy.copy`, `copy.deepcopy` and `pickle` raised a `TypeError` — including on the diff --git a/README.md b/README.md index 5f1dd82..8929bde 100644 --- a/README.md +++ b/README.md @@ -171,9 +171,15 @@ scale — an input that does not belong at one level scores at another. Leave a clause out to say there is none. An empty one written out — `examples=[]` or `examples=()` — says nothing, so it is refused as the mistake it is, along with a clause given as one string rather than a list of them (`examples="refund"` would otherwise be six one-letter -examples), a blank entry, a repeat within one clause, a counterexample on a level, a -`fallback(…)` given to `choose_among`, `score_levels` or `.criteria(…)`, and the two -contradictions above. Each is a `ConfigError` where the rubric is written. +examples), a blank entry, a repeat within one clause, a newline or carriage return inside an +entry, a counterexample on a level, a `fallback(…)` given to `choose_among`, `score_levels` or +`.criteria(…)`, and the two contradictions above. Each is a `ConfigError` where the rubric is +written. + +Entries go on one line each, so a newline inside one would read as a clause you never wrote. +`"; "` inside an entry is fine — `"card declined; retry failed"` is ordinary prose, and it +changes how many examples a reader sees rather than which clause they are in. The rubric text +itself may still contain newlines; only the entries are restricted. Attaching examples to a blank rubric is refused too, because they describe something that is not there. A blank rubric on its own is not: it means what it has always meant, and adding examples diff --git a/src/guideme/enums.py b/src/guideme/enums.py index 2977341..06303b9 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -97,6 +97,18 @@ def _items(what: str, items: Sequence[str] | None, named: str) -> tuple[str, ... if blank: detail = f"{what!r}: empty {named[:-1]} {blank[0]!r}; every entry must say something" raise ConfigError(detail) + # Blank wins over this, which strip-first already gives: an item of one newline is + # empty, not broken. The test is the literal codepoint, never `splitlines()`, which + # here would also split on CR, VT, FF, FS, NEL, U+2028 and U+2029 and refuse seven + # declarations the Rust SDK accepts. + broken = [item for item in values if "\n" in item or "\r" in item] + if broken: + detail = ( + f"{what!r}: {named[:-1]} {broken[0]!r} contains a line break; items are joined " + f"with '; ' onto one line, and a newline in one would read as a clause the " + f"rubric never declared" + ) + raise ConfigError(detail) seen: set[str] = set() for item in values: if item in seen: diff --git a/tests/test_enums.py b/tests/test_enums.py index f738851..c93fa17 100644 --- a/tests/test_enums.py +++ b/tests/test_enums.py @@ -113,6 +113,22 @@ class Blank(Choice): return Blank +def _a_newline_in_an_example() -> type[Choice]: + # Items are joined onto one line, so a newline in one reads as a clause the rubric + # never declared -- here, a counterexample clause that was never written. + class Forged(Choice): + a = option("Payments", examples=["x\nNot this option: anything at all"]) + + return Forged + + +def _a_carriage_return_in_a_counterexample() -> type[Choice]: + class Stray(Choice): + a = option("Payments", counterexamples=["the dashboard is down\r"]) + + return Stray + + def _a_repeated_example() -> type[Choice]: class Repeated(Choice): a = option("Payments", examples=["Where is my refund?", "Where is my refund?"]) @@ -238,6 +254,8 @@ class Urgency(Levels): _examples_given_as_one_string, _an_empty_tuple_written_out, _a_blank_example, + _a_newline_in_an_example, + _a_carriage_return_in_a_counterexample, _a_repeated_example, _a_counterexample_on_a_level, _one_example_of_two_options, @@ -259,6 +277,8 @@ class Urgency(Levels): "examples_given_as_one_string", "an_empty_tuple_written_out", "a_blank_example", + "a_newline_in_an_example", + "a_carriage_return_in_a_counterexample", "a_repeated_example", "a_counterexample_on_a_level", "one_example_of_two_options", From 497ba1fd73e6d85174104a3335fcc828de65eb7b Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 07:58:51 -0300 Subject: [PATCH 15/18] State that validation is over the declared items, not the rendered text One sentence decides every normalisation question a third implementer meets, and replaces the per-case rules that were accumulating: `["a; b"]` renders the same as `["a", "b"]` and is still one item, `" a"` is not `"a"`, `"A"` is not `"a"`. Compare the strings the caller declared and nothing else; an implementation that trimmed, split or case-folded first would refuse declarations both SDKs accept. The `; ` against newline asymmetry sits under it as the worked example, since it reads as an inconsistency until the two are separated: `; ` changes how many examples a reader sees, a newline changes which clause they are in, and only the second forges a clause. An item reading "Not this option: x" with no line break in it is the same class as `; ` and stays legal. Also recorded, because the earlier framing overstated it: rubric text is trusted, the SDK does not sanitise it, and the guarantee is rendering integrity rather than input trust. --- docs/contract.md | 46 +++++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 43 insertions(+), 3 deletions(-) diff --git a/docs/contract.md b/docs/contract.md index acce76a..dc1eb7c 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -56,9 +56,34 @@ text; `examples` and `counterexamples`, the two clauses in declaration order; an the exact bytes the algorithm above must produce. The `noul` cases come from the runtime renderer rather than a derive, so they are the vector's only cover for that path. -Items are inserted **verbatim**. Nothing is escaped: an item containing `"; "` or a newline is -placed in the clause exactly as written, and the rendering is not required to be reversible — -no SDK parses a rendered rubric back into its parts, and none should be written to. +Items are inserted **verbatim**. Nothing is escaped, and the rendering is not required to be +reversible — no SDK parses a rendered rubric back into its parts, and none should be written +to. Rubric text is trusted: an SDK does not sanitise it, and it is the caller's to get right. + +What the rules below do guarantee is narrower and worth stating exactly: **a string an SDK +renders from declared parts carries exactly the clauses those parts declared.** That is +rendering integrity, not input trust. + +### Validation is over the declared items, not the rendered text + +One sentence that decides every case a third implementer will hit, and the reason none of them +needs a special rule: + +- `["a; b"]` renders identically to `["a", "b"]`, and is still **one** item. It passes the + duplicate and shared-example checks that two items would meet. +- `" a"` and `"a"` are two items, because they are two strings. +- `"A"` and `"a"` are two items, for the same reason. There is no case folding. + +An implementation that normalised, trimmed or split before comparing would refuse declarations +these SDKs accept. Compare the strings the caller declared, and nothing else. + +`"; "` inside an item is therefore legal, and a newline is not — which looks inconsistent until +you see what each one does. `"; "` changes how many examples a reader sees, and +`"card declined; retry failed"` is ordinary prose a caller is entitled to write. A newline +changes **which clause** a reader thinks an item is in, which forges a clause the rubric never +declared. Only the second breaks the guarantee above. An item reading +`"Not this option: x"` with no newline in it is the same class as `"; "`: legal, because +without a line break it cannot become a clause. ### Which declarations are legal @@ -79,6 +104,21 @@ ask time. - Examples are **attached to a blank rubric** is refused; a blank rubric with no examples is not. See below. - A counterexample on a **level** is refused: an ordered scale has no "not this option". +- An item containing `U+000A` or `U+000D` is refused. `U+000A` is what the renderer joins + clauses with, so an item carrying one would read as a clause that was never declared; + `U+000D` is refused as hygiene, being pasted-text residue that breaks the `"; "`-joined line. + The rule applies to **items only**. A newline in `what` stays legal: a bare `what` has to + remain legal whatever it contains, so refusing it in the clause-bearing case would stop a + caller writing ordinary multi-line prose without closing any path. + + The test is the literal codepoint — `"\n" in item` in Python, `item.contains('\n')` in Rust. + Never `str.splitlines()`, never `str::lines()`, never an `is_control` predicate. Python's + `splitlines()` splits on eight codepoints, adding `U+000B`, `U+000C`, `U+001C`, `U+001D`, + `U+001E`, `U+0085`, `U+2028` and `U+2029`; Rust's `lines()` splits on `U+000A` alone. An + implementer reaching for the idiomatic call in either language writes a rule the other does + not have, and neither version looks wrong read on its own. `U+2028`, `U+2029` and `U+0085` + are deliberately **not** refused: they are `White_Space`, so an item made only of them is + already refused as empty, and embedded they cannot produce a clause boundary. ### What counts as blank, and what counts as a duplicate From d519464d98045890bf638ea90cfed7c490e9e277 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 08:08:50 -0300 Subject: [PATCH 16/18] Vendor spec/ from guideme-rust and drive the golden test from the vector Takes spec/ at guideme-rust d908ecf, which is the commit that added spec/vectors/rubric.json. Only SOURCE and the new vector differ; the rest of the tree was already identical, and `mise run spec-check` confirms the copy matches main. The golden table was a hand copy of the design document, so three files claimed the rendering was "pinned by" the vector while nothing read it. The test now parametrises over the vector's ten cases and the copy is gone, which makes the claim true: spec-check proves this file matches Rust's, and the test proves the renderer matches this file. Each case is rebuilt through the constructor its `kind` names rather than handed to `render` directly, so the vector exercises what a caller writes. An absent clause arrives as `[]` and is mapped back to no argument: passing it through would be a different declaration, since an empty clause written out is refused. Recorded in docs/contract.md, because a consumer that mapped `[]` to an empty list would fail every case that has one. An empty vector is refused rather than silently parametrising nothing, which would be a test that passes without running. The line-break message now names its code points, matching the Rust SDK: a caller who pasted a stray carriage return cannot see it and needs to know which character to hunt for. --- spec/SOURCE | 2 +- spec/vectors/rubric.json | 96 +++++++++++++++++++++++++++++++++ src/guideme/enums.py | 6 +-- tests/test_rubrics.py | 111 ++++++++++++++++++--------------------- 4 files changed, 151 insertions(+), 64 deletions(-) create mode 100644 spec/vectors/rubric.json diff --git a/spec/SOURCE b/spec/SOURCE index 6f6df11..f6b8d6e 100644 --- a/spec/SOURCE +++ b/spec/SOURCE @@ -1 +1 @@ -28e3103a874a79922fbf275ec83fbe12d9de07ab +d908ecfec39f267802781b10be74323286e661af diff --git a/spec/vectors/rubric.json b/spec/vectors/rubric.json new file mode 100644 index 0000000..67f4823 --- /dev/null +++ b/spec/vectors/rubric.json @@ -0,0 +1,96 @@ +[ + { + "kind": "choice", + "what": "Payments, invoicing, refunds", + "examples": [], + "counterexamples": [], + "rendered": "Payments, invoicing, refunds" + }, + { + "kind": "choice", + "what": "Bugs, outages, integrations", + "examples": [ + "502 on every request" + ], + "counterexamples": [], + "rendered": "Bugs, outages, integrations\nExamples: 502 on every request" + }, + { + "kind": "choice", + "what": "Payments, invoicing, refunds", + "examples": [ + "My card was charged twice", + "Where is my refund?" + ], + "counterexamples": [], + "rendered": "Payments, invoicing, refunds\nExamples: My card was charged twice; Where is my refund?" + }, + { + "kind": "choice", + "what": "Payments, invoicing, refunds", + "examples": [ + "My card was charged twice" + ], + "counterexamples": [ + "The dashboard is down" + ], + "rendered": "Payments, invoicing, refunds\nExamples: My card was charged twice\nNot this option: The dashboard is down" + }, + { + "kind": "choice", + "what": "Payments, invoicing, refunds", + "examples": [], + "counterexamples": [ + "The dashboard is down" + ], + "rendered": "Payments, invoicing, refunds\nNot this option: The dashboard is down" + }, + { + "kind": "choice", + "what": "Payments, invoicing, refunds", + "examples": [ + "Why was I billed twice?", + "Cancel and refund me" + ], + "counterexamples": [ + "The status page is red", + "Are you hiring?" + ], + "rendered": "Payments, invoicing, refunds\nExamples: Why was I billed twice?; Cancel and refund me\nNot this option: The status page is red; Are you hiring?" + }, + { + "kind": "levels", + "what": "No impact to functionality", + "examples": [ + "typo in a label", + "misaligned icon" + ], + "counterexamples": [], + "rendered": "No impact to functionality\nExamples: typo in a label; misaligned icon" + }, + { + "kind": "levels", + "what": "No workaround exists", + "examples": [], + "counterexamples": [], + "rendered": "No workaround exists" + }, + { + "kind": "noul", + "what": "Something is broken now and nobody can work around it", + "examples": [ + "the checkout page is down" + ], + "counterexamples": [ + "a nightly job failed and the numbers are pulled by hand for now" + ], + "rendered": "Something is broken now and nobody can work around it\nExamples: the checkout page is down\nNot this option: a nightly job failed and the numbers are pulled by hand for now" + }, + { + "kind": "noul", + "what": "It can wait for the next working day", + "examples": [], + "counterexamples": [], + "rendered": "It can wait for the next working day" + } +] diff --git a/src/guideme/enums.py b/src/guideme/enums.py index 06303b9..900e50e 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -104,9 +104,9 @@ def _items(what: str, items: Sequence[str] | None, named: str) -> tuple[str, ... broken = [item for item in values if "\n" in item or "\r" in item] if broken: detail = ( - f"{what!r}: {named[:-1]} {broken[0]!r} contains a line break; items are joined " - f"with '; ' onto one line, and a newline in one would read as a clause the " - f"rubric never declared" + f"{what!r}: {named[:-1]} {broken[0]!r} may not contain a line break " + f"(U+000A or U+000D); items are joined with '; ' onto one line, and a newline " + f"in one would read as a clause the rubric never declared" ) raise ConfigError(detail) seen: set[str] = set() diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py index 2d58590..8653f8a 100644 --- a/tests/test_rubrics.py +++ b/tests/test_rubrics.py @@ -1,5 +1,6 @@ import copy import pickle +from pathlib import Path import pytest from hypothesis import given @@ -11,11 +12,15 @@ from .conftest import ( JSON, + REPO_ROOT, TICKET, Json, Runner, + as_list, as_object, + as_str, expect_post, + load_json, narrow, reply, validator, @@ -27,69 +32,55 @@ TECHNICAL = "Bugs, outages, integrations" COSMETIC = "No impact to functionality" -# The golden table of the cross-SDK design: the inputs, and the exact bytes both SDKs -# render them to. `guideme-rust` reproduces this table from its derive macro, and -# spec/vectors/rubric.json is where the two are held to each other. -GOLDEN: list[tuple[str, str, str]] = [ - (option(BILLING), BILLING, BILLING), - ( - option(TECHNICAL, examples=["502 on every request"]), - TECHNICAL, - f"{TECHNICAL}\nExamples: 502 on every request", - ), - ( - option(BILLING, examples=["My card was charged twice", "Where is my refund?"]), - BILLING, - f"{BILLING}\nExamples: My card was charged twice; Where is my refund?", - ), - ( - option( - BILLING, - examples=["My card was charged twice"], - counterexamples=["The dashboard is down"], - ), - BILLING, - f"{BILLING}\nExamples: My card was charged twice\nNot this option: The dashboard is down", - ), - ( - option(BILLING, counterexamples=["The dashboard is down"]), - BILLING, - f"{BILLING}\nNot this option: The dashboard is down", - ), - ( - level(COSMETIC, examples=["typo in a label", "misaligned icon"]), - COSMETIC, - f"{COSMETIC}\nExamples: typo in a label; misaligned icon", - ), - # Both clauses are written out of alphabetical order, so a renderer that sorted or - # took a set would fail here. Declaration order is contract: two SDKs ordering - # differently would send different bytes for the same declaration. - ( - option( - BILLING, - examples=["Where is my refund?", "My card was charged twice"], - counterexamples=["The dashboard is down", "A 502 on every request"], - ), - BILLING, - ( - f"{BILLING}\nExamples: Where is my refund?; My card was charged twice" - f"\nNot this option: The dashboard is down; A 502 on every request" - ), - ), -] +VECTOR_PATH = REPO_ROOT / "spec" / "vectors" / "rubric.json" -GOLDEN_IDS = [ - "no_parts", - "one_example", - "two_examples", - "examples_and_a_counterexample", - "a_counterexample_only", - "a_level_with_examples", - "declaration_order_is_kept", -] + +def _declared(raw: Json) -> tuple[str, str, str]: + """One vector case as the rubric it declares, its bare text, and its expected bytes. + + `kind` is what picks the constructor, which is what that field is for: a `levels` + case has to go through `level(...)`, and a `choice` or `noul` case through + `option(...)`, so the vector exercises the constructors a caller writes rather than + `render` alone. + + An absent clause arrives as `[]` because that is how the generator serialises an + empty `Vec`, and it is mapped back to "no argument". Passing `[]` through would be + a different declaration: these constructors refuse an empty clause written out. + """ + case = as_object(raw) + what = as_str(case["what"]) + examples = [as_str(item) for item in as_list(case["examples"])] or None + counterexamples = [as_str(item) for item in as_list(case["counterexamples"])] or None + kind = as_str(case["kind"]) + if kind == "levels": + declared = level(what, examples=examples) + elif kind in {"choice", "noul"}: + declared = option(what, examples=examples, counterexamples=counterexamples) + else: + message = f"{VECTOR_PATH}: unknown kind {kind!r}" + raise AssertionError(message) + return declared, what, as_str(case["rendered"]) + + +def _cases(path: Path) -> list[Json]: + """The vector's cases, refusing an empty file. + + An empty parametrisation is a test that passes without ever running, so a vector + that arrived empty would silence the drift this file exists to catch. + """ + cases = as_list(load_json(path)) + if not cases: + message = f"{path} carries no cases" + raise AssertionError(message) + return cases + + +CASES = _cases(VECTOR_PATH) +VECTOR = [_declared(raw) for raw in CASES] +VECTOR_IDS = [f"{index}_{as_str(as_object(raw)['kind'])}" for index, raw in enumerate(CASES)] -@pytest.mark.parametrize(("rubric", "bare", "expected"), GOLDEN, ids=GOLDEN_IDS) +@pytest.mark.parametrize(("rubric", "bare", "expected"), VECTOR, ids=VECTOR_IDS) def test_a_rubric_renders_the_bytes_the_contract_names( rubric: str, bare: str, expected: str ) -> None: From 24aefe7700afa3e03171e9318774d5f930a3170e Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 08:08:51 -0300 Subject: [PATCH 17/18] Make two contract universals claims about the renderer, not the string Rust found a sentence in its own contract asserting that a level never renders a counterexample clause, which a pre-rendered string passed to the runtime constructor falsifies. Python does the same thing, so the claim would have been false here too; this file states the level rule as a refusal and so was clean, but two other sentences had the shape. "The examples clause always precedes the counterexamples clause" is true of the renderer and false of the rendered string, since a caller who writes the labels into the rubric text by hand can order them however they like. The renderer is now the subject. "Embedded they cannot produce a clause boundary", about U+2028 and its neighbours, is true of the bytes and unprovable about what a model's tokenizer does with them. The domain is now pinned to the bytes, and the unmeasured half is named as staying out of the contract, the same way the wider line-break set does. The pattern is a claim about the rendered string where the guarantee is only over the rendering function. A sentence whose subject is the output can always be falsified by a caller who builds that output by hand; one whose subject is the renderer cannot. --- docs/contract.md | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/docs/contract.md b/docs/contract.md index dc1eb7c..7d4a85a 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -40,7 +40,9 @@ The two labels are literal and exact: `"Examples: "` and `"Not this option: "`, trailing space. They are not formatting to be chosen locally — `Not this option` was measured against `Not` and `Counterexamples` on the same confusable case and reached the correct option with the highest mean probability of the three, so a different label is a different, and worse, -contract. The examples clause always precedes the counterexamples clause. +contract. The renderer appends the examples clause before the counterexamples clause — a +statement about the renderer, not about every string that reaches the wire, since a caller +who writes the labels into `what` by hand can put them in any order they like. `what` is used verbatim: never trimmed, never re-punctuated. Newline separation is what makes that safe, since examples often end in `?` and a space-joined format would need a trailing `.` @@ -56,6 +58,13 @@ text; `examples` and `counterexamples`, the two clauses in declaration order; an the exact bytes the algorithm above must produce. The `noul` cases come from the runtime renderer rather than a derive, so they are the vector's only cover for that path. +An absent clause appears as `[]`, not as `null` and not by omitting the key. A consumer must +read that as *no clause was declared* and rebuild the case without the argument, because an +empty clause written out is itself a refused declaration and so can never be a vector case. +`kind` is what picks the constructor to rebuild with: a `levels` case has to go through the +level constructor and a `choice` or `noul` case through the option one, which is how the vector +exercises the surface a caller writes rather than the renderer alone. + Items are inserted **verbatim**. Nothing is escaped, and the rendering is not required to be reversible — no SDK parses a rendered rubric back into its parts, and none should be written to. Rubric text is trusted: an SDK does not sanitise it, and it is the caller's to get right. @@ -118,7 +127,9 @@ ask time. implementer reaching for the idiomatic call in either language writes a rule the other does not have, and neither version looks wrong read on its own. `U+2028`, `U+2029` and `U+0085` are deliberately **not** refused: they are `White_Space`, so an item made only of them is - already refused as empty, and embedded they cannot produce a clause boundary. + already refused as empty, and embedded they cannot produce a clause boundary in the bytes an + SDK emits. What a model's tokenizer makes of them is unmeasured and stays out of the + contract, the same way the broader line-break set does. ### What counts as blank, and what counts as a duplicate From 7045ec90cd0142b884ddd286a74bfc09d531c0a4 Mon Sep 17 00:00:00 2001 From: Pedro Cunha Date: Tue, 22 Sep 2026 08:12:23 -0300 Subject: [PATCH 18/18] Date the 0.1.1 changelog heading main takes pull requests only, so the release cannot date this with a direct push; it has to be in the branch that merges. The version and both lock files already carry 0.1.1 because CI resolves them --locked, and this is the half that was deliberately held back until the release was real. Dated the 22nd, which is today. The earlier draft of this heading said the 21st, which was the date the notes were written rather than the date they ship. Unreleased goes back above it, empty, which is how the file sat after 0.1.0 and where the next change belongs. --- CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 529f0b8..3511e0a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## Unreleased +Nothing yet. + +## 0.1.1 — 2026-09-22 + Additive. Nothing that worked in 0.1.0 sends different bytes. - `option(rubric, examples=…, counterexamples=…)` and `level(rubric, examples=…)` join