diff --git a/AGENTS.md b/AGENTS.md index ed60543..0931dcd 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -8,16 +8,24 @@ A Python package that makes a TypeSafe Jev judgment usable as control flow: a ye `if`, a choice is an exhaustive `match`, a score is a comparison. One distribution, `guideme`, published to PyPI under `MIT OR Apache-2.0`. The public surface has two tiers: -- the 35 names in `__all__` in `src/guideme/__init__.py`, imported from `guideme` itself; +- the 44 names in `__all__` in `src/guideme/__init__.py`, imported from `guideme` itself; - `guideme.api` and `guideme.policy` as whole modules, imported by their own path and not re-exported at the top level: `guideme.api` is the wire mirror and `guideme.api.client` holds `Client` and `AsyncClient`, and `guideme.policy` holds `resolve`. -- one name from a third module, `guideme.question.Question`: what every constructor in the - first tier returns, and the only way to write the type of a stored question down. The - promise covers that name and nothing else in `guideme.question` — `validate`, `Spec` and - the concrete question classes stay private, because a tier is a promise and this is the - narrowest one that lets a caller annotate. The README's **Lower layers** section documents - all of it. + +Six modules declare an `__all__`: `guideme` itself, `guideme.api`, and the four it draws +names from — `question`, `policy`, `enums` and `errors`. Where a module has one, the list is +what it owns rather than what it happens to have imported. `guideme/__init__.py` is the one +exception to that reading, because it is a façade and every name in its list arrived by +import; `tests/test_surface.py` therefore holds only `guideme.api` to the stricter rule, +proving by AST that nothing it imported is re-exported. The rest — `receipt`, `scalars`, +`guide`, `api.client`, `telemetry`, `ask`, `_json` — declare none, and nothing here asks +them to: an `__all__` earns its place by being checked, and the check is the surface test. `guideme.question` no longer has a tier of its +own: `Question`, the three question classes and the three detail classes are all in the top +list now, because a caller annotating a stored question needed them and importing from a +module the README called private to do it was the wrong answer. `validate`, `Spec` and the +criteria shapes stay private, as do `render` and the `require_*` checks in `guideme.enums` +despite their public-looking names. The README's **Lower layers** section documents all of it. Everything else in the package is private, whatever its name looks like. @@ -43,19 +51,31 @@ two pages: `https://docs.typesafe.ai/api.md` covers `POST /v1/systemone` and | `src/guideme/enums.py` | `Choice`, `Levels`, `option`, `level`, `fallback`, and the internals `render` and the `require_*` checks | a member's name is its wire key and its value is its rubric; both validate at class definition, and a repeated rubric text is refused there. `render` is the one place a rubric's examples become wire text, and its output is a cross-SDK contract item. `render`, `require_unshared_examples`, `require_no_counterexamples` and `require_no_fallback` are internal despite their names: they are imported by `question.py` and are in no tier, like `question.validate` | | `src/guideme/question.py` | question kinds, constructors, `Ranked`, `Scored`, the unsure ladder | a question is inert until asked; the reader travels with it | | `src/guideme/ask.py` | shapes: `encode`, `decode`, `Plan` | ids are `q0..qN` in encounter order, insertion order for a dict | -| `src/guideme/_ask_overloads.py` | the typed `ask` surfaces | GENERATED; edit `scripts/gen_ask_overloads.py` and run `mise run gen` | +| `src/guideme/receipt.py` | `Receipt` and the value-object `Usage` | its own module because the generated `ask` surfaces name `Receipt` in a return type, so it has to sit below them; it imports nothing from the package | +| `src/guideme/_ask_overloads.py` | the typed `ask` and `ask_with_receipt` surfaces | GENERATED; edit `scripts/gen_ask_overloads.py` and run `mise run gen` | | `src/guideme/telemetry.py` | every span, event, log record and attribute | the names are the contract, documented in `docs/observability.md`; installs no provider | | `src/guideme/api/__init__.py` | the wire mirror and the adapters to the core | mirrors `spec/schema/*.json` field for field; no policy here | | `src/guideme/api/client.py` | HTTP, retries, statuses to errors, one span per attempt | the only importer of `httpx`; every decision it makes is made by the pure `step` | | `src/guideme/guide.py` | `Guide`, `AsyncGuide`, `GuideBuilder`, `ModelInfo` | the two executors share `_prepare` and `_finish`; what is written twice is the two `await`s | | `spec/` | the vendored schemas and golden vectors | read-only here; it is `guideme-rust`'s output, and `mise run spec-check` proves this copy matches | +**No pydantic type reaches the top-level surface.** `ModelInfo` copies the wire's +`ModelEntry` and `receipt.Usage` copies the wire's `api.Usage`, so what `models()` and +`ask_with_receipt` hand back is this package's own value object in both cases. The wire +models keep their names inside `guideme.api`, where naming pydantic is the point. A type on +the published surface must not carry a dependency's methods or change shape when that +dependency has a major release, and "it is only two integers" is not an exception to that — +it is how the first one would get in. + Modules keep a one-way import graph, which `pyright`'s `reportImportCycles` enforces: `errors` imports nothing from the package; `_json` and `scalars` import `errors`; `policy` imports `errors` and `scalars`; `enums` imports `errors` and `policy`; `question` imports the -above; `ask` imports `question`; `_ask_overloads` imports `_json` and `question`; `telemetry` -imports `errors` and `policy`; `api` imports `question` and below; `api.client` imports `api` -and `telemetry`. Nothing inside the package writes `from guideme import ...`: that would +above; `ask` imports `question`; `telemetry` imports `errors` and `policy`; `api` imports +`question` and below; `receipt` imports nothing from the package; `_ask_overloads` imports +`_json`, `question` and `receipt`; `api.client` imports `api` and `telemetry`. `receipt` sits where it does +because `_ask_overloads` names `Receipt` in a return type and `guide` imports +`_ask_overloads`, so the type cannot live in `guide`. +Nothing inside the package writes `from guideme import ...`: that would import the package's own `__init__`, which imports the executors, which import `api`. `docs/design.md` records the decisions and the sharp edges. Update it when a decision changes. @@ -185,7 +205,15 @@ is made in `guideme-rust` first, not here. ## Tests -Few tests, high grade. The ceiling is 47 test functions; a parametrised function counts once. +Few tests, high grade. The ceiling is 51 test functions; a parametrised function counts once. +It was 47 before 0.2.0, which added four. Each one reaches something no existing function +could: the counted connection pool needs two guides over one pool, which nothing else builds; +the receipt needs the response's `model` and `usage` read back by a caller, where every other +test reads them off a span; an injected transport answers with no server at all, and every +other wire assertion is written against one; and the connection-failure retry needs a +transport that fails on demand, which no local server can be made to do. Everything else 0.2.0 +added — the `529` carrying its `retry-after`, `GET /v1/models` being retried, and the five +transport-and-timeout refusals — went into a parameter of a test that was already there. It was 40 before the logs signal, which is user-requested scope that the span assertions could not cover: correlation, severity and routing each need a record to look at. The forty-third is the pre-publish proof that a log sink which raises reaches neither the caller nor the ask span: @@ -209,6 +237,16 @@ A new test must be one of: - a structural import-boundary proof (AST) over `src/guideme`; - a redaction proof. +The server-backed checks live in two modules, split at 0.2.0 when the one they were in +reached a thousand lines. `tests/test_wire.py` is what guideme puts on the wire and reads +back: the schemas, the docs examples, the request a shape builds, the errors a malformed +response raises, the unsure ladder, and what a configuration mistake refuses. +`tests/test_retries.py` is the request that does not simply succeed: the status-to-error +table, the retry policy on both endpoints, what is and is not resent before a response +arrives, an injected transport, and the pool two guides share. They are one subject each +because they are one code path each, and a helper both need — `answering_offline` and +`MODELS_BODY` — belongs in `conftest.py` rather than being imported across. + Two of the illegal states this package refuses cannot be checker errors: an enum body is opaque to pyright, so a `Choice` with two `fallback` members and a `Levels` with one member are a `ConfigError` raised at class definition, proven in `tests/test_enums.py`. Every other negative diff --git a/CHANGELOG.md b/CHANGELOG.md index 3511e0a..bc1972d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,96 @@ Nothing yet. +## 0.2.0 — 2026-09-22 + +One breaking change, and it is one nobody outside this repository can have depended on yet. +Everything else is additive. + +### Breaking + +- `OverloadedError` now takes the `retry-after` the API sent: `OverloadedError(retry_after)` + with a `retry_after` attribute, mirroring `RateLimitedError`. A `529` carries that header as + often as a `429` does, and both SDKs were throwing it away on the one path where it is the + only thing that says when to come back. Constructing the error by hand is the only code this + moves; catching it is unchanged. + +### Added + +- `ask_with_receipt` on `Guide` and `AsyncGuide`, with the same overload family as `ask`. It + returns `Receipt[T]` — `answer`, `model`, `usage` — so cost attribution and pinning a policy + to the model version that produced its numbers no longer need an OpenTelemetry pipeline. + `ask` is that call followed by `.answer`. `Receipt` and `Usage` are exported from `guideme`, + and `Usage` is a frozen dataclass of two `int`s copied out of the wire model, so no + pydantic type reaches the top-level surface — the trade `ModelInfo` already makes. +- `with` and `async with` on the two guides, each closing the guide on the way out. The pool a + guide holds is now counted: `with_policy(…)` takes a second hold on it, and closing either + guide leaves the other able to ask. Before this, closing a derived guide closed its parent's + pool, which was a documented sharp edge and would have been a trap under `with`. +- `GuideBuilder.transport(…)` and `.async_transport(…)`, taking an `httpx.BaseTransport` and an + `httpx.AsyncBaseTransport`. A proxy, a client certificate, or an `httpx.MockTransport` that + answers a test with no server, no port and no key — the README's new **Testing your code** + section is that test written out. A transport and `timeout(…)` refuse each other in either + order, because a custom transport is free to ignore the budget `httpx` hands it and a + silent no-op is worse than a `ConfigError`; so does building the wrong kind of guide from + one, and so does setting both transports on one builder, which could build neither. +- `GET /v1/models` is retried on `429` and `529`, through the same loop and the same spans an + ask uses. The API's docs say an SDK handles a `429` for you, and a `429` during startup used + to fail the start. +- A failed connection is retried inside the same `max_retries` budget and backoff: + `httpx.ConnectError`, which means the request never reached a server, so nothing was + judged and nothing is repeated. A disconnect part-way through a response and a body that + will not decode are still not retried — the request arrived, and a resend would buy the + same judgment twice. **No timeout is retried, of any phase**, `httpx.ConnectTimeout` + included: Rust sets one deadline over the whole attempt and cannot tell a connect timeout + from a read one, so retrying it here would make the two SDKs disagree about the same + failure, and a retried timeout multiplies the wall time `timeout(…)` exists to bound. +- `guideme.__all__` gains `Question`, `NoulQuestion`, `ChoiceQuestion`, `ScoreQuestion`, + `DetailedNoul`, `DetailedChoice`, `DetailedScore`, `Receipt` and `Usage`, reaching 44 names. + Annotating a stored question no longer means importing from a module the README calls + private. `guideme.api` gains an `__all__` of its own, so `import *` from it stops handing + back `BaseModel`, `Field` and `Mapping`; `guideme.question`, `guideme.policy`, + `guideme.enums` and `guideme.errors` each gained one too. + +### Fixed + +- `with_policy(…)` no longer leaks a hold on the connection pool when the patch it is given + cannot settle. Python evaluates arguments left to right, so the hold was taken before the + patch was validated and nothing released it: the guide that would have was never built. + The patch settles first now. A pool that never closes is invisible until a process runs + out of sockets, so the regression asserts the count rather than the symptom. +- The pool's count and every holder's spent-flag are taken under one lock. `Guide`'s + docstring says to share a guide across threads, so two threads closing two guides over one + pool is a documented thing to do, and a flag read, a flag flip and a decrement are three + steps that must not interleave. `ask` is untouched and takes no lock. +- Closing one guide twice no longer closes the connection pool under a guide derived from it + with `with_policy(…)`. `share()` now hands back a distinct client over the shared pool, each + carrying its own release-once flag, so a guide releases exactly once however many times it + is closed; a count alone cannot tell which holder a release came from. Sharing from an + already-closed guide is a `ConfigError` rather than a guide holding nothing. + +### Changed + +- `guideme.ask` walks a shape through four `TypeGuard` predicates instead of four `cast()` + calls. There are now no casts anywhere in `src/guideme`: a `TypeGuard` replaces the narrowed + type outright where an annotated assignment only intersects with it, so the element types + are `object` rather than unknown and a checker verifies what was being asserted before. + +- `guideme.retry` carries `error.type = "transport"` and **no** `http.response.status_code` + when the attempt it is resending never got a response. Exactly one of the two is on every + such event. A dashboard grouping retries by cause has to tell a throttled API from an + unreachable one, so the cause is which field is present. `docs/observability.md` has the + table and `docs/contract.md` the retry policy in full. + +### Documented + +- The timeout's scope, which is per phase in `httpx` and per attempt in Rust's `reqwest`, is + now on `GuideBuilder.timeout`, in the README's configuration table and in `docs/contract.md` + as a stated divergence rather than something a reader has to find. +- Concurrency: build one guide, share it across threads or tasks, close it once. It was true + before and written down nowhere. +- `docs/contract.md` drops the rubric asymmetry between the SDKs' runtime constructors, which + guideme-rust closed at 0.2.0, and corrects a stale `Rubric::into_wire()` to `Rubric::render`. + ## 0.1.1 — 2026-09-22 Additive. Nothing that worked in 0.1.0 sends different bytes. diff --git a/README.md b/README.md index 8929bde..0c6df9e 100644 --- a/README.md +++ b/README.md @@ -271,19 +271,109 @@ list or dict, which is the shape above. `await` and the `httpx` client underneath. ```python -guide = AsyncGuide.from_env() -verdict: Verdict = await guide.ask(noul("Is this about billing?").detail(), ticket) -await guide.close() +async with AsyncGuide.from_env() as guide: + verdict: Verdict = await guide.ask(noul("Is this about billing?").detail(), ticket) ``` -The synchronous version is the same three lines with `Guide` and without the `await`s. +The synchronous version is the same two lines with `Guide`, `with` and no `await`. `Guide.builder()` and `AsyncGuide.builder()` return the same `GuideBuilder`; `.build()` gives -the synchronous guide and `.build_async()` the asynchronous one. +the synchronous guide and `.build_async()` the asynchronous one. Leaving the block closes the +guide, and `guide.close()` does the same thing by hand for a guide that outlives any block. + +A guide holds a connection pool, so build one and share it: both kinds are safe to use from +several threads or several tasks at once, and one guide asking concurrently is what the pool +is for. Building one per request works but opens a pool per request, which is the cost the +pool exists to avoid. `guide.with_policy(…)` returns a second guide over the *same* pool, and +the two are counted, so closing either leaves the other able to ask and the pool closes when +the last of them does. Close each guide once. Both guides also answer `models()`, which returns a `tuple[ModelInfo, ...]`: the models the account may use, each with its `name`, `description` and `release_date`. It is one call to -`GET /v1/models` and gets no ask span of its own. Unlike `ask`, it is not retried: a `429` or a -`529` raises on the first attempt, so the two rows below that mention retries do not apply to it. +`GET /v1/models` and gets no ask span of its own. It is retried on exactly the terms an ask +is, so a `429` while your process is starting up does not fail the start. + +## What a request cost + +`ask_with_receipt` is `ask` with the response's own numbers kept. It takes the same shapes and +infers the same types; `ask` is this call followed by `.answer`. + +```python +receipt: Receipt[bool] = guide.ask_with_receipt(noul("Is this urgent?"), ticket) + +if receipt.answer: + prioritise() +meter(model=receipt.model, tokens=receipt.usage.input_tokens) +``` + +`Receipt` is frozen and carries three things: `answer`, whatever `ask` would have returned; +`model`, the versioned id that actually answered, which is `jev-1.13.0` and not `jev-latest` +even when an alias was asked for; and `usage`, a `Usage` with `input_tokens` and +`output_tokens`. Input tokens are what is billed. Log the model: thresholds are tuned against +one model's numbers, and the alias moves under you. + +## Configuration + +Every setter on `GuideBuilder` returns the builder, and `Guide.builder()` starts one. + +| Setter | Default | What it does | +|---|---|---| +| `api_key(ApiKey(…))` | none; required | The key. `from_env()` reads it from `TYPESAFE_API_KEY`. | +| `base_url(…)` | `https://api.typesafe.ai` | The API origin. It may not carry credentials. | +| `model(Model(…))` | `jev-latest` | The model or alias to ask. | +| `policy(Policy(…))` | the defaults | The guide-wide policy patch; a question's own wins over it. | +| `max_retries(n)` | `3` | Resends per call, `0` to never resend. | +| `backoff(…)` | 500 ms | Base of the exponential backoff. | +| `timeout(…)` | 30 s | Per phase of one attempt. Read the paragraph below. | +| `transport(…)` / `async_transport(…)` | `httpx`'s own | Send through your `httpx` transport. | +| `record_state(True)` | off | Put the state JSON on the ask span. It is your users' data. | +| `events(…)` | `"both"` | Whether an answer and a retry go to the span, a log record, or both. | + +**`timeout` is not a deadline for the attempt.** `httpx` gives the whole budget to each phase +separately — connecting, writing, reading, and waiting for a pooled connection — so one attempt +that is slow in more than one phase takes longer than the timeout without breaching anything. +Worst case for a call is `max_retries + 1` attempts of several phases each, plus the backoff +between them. The Rust SDK's `reqwest` deadline covers the attempt as a whole instead; the two +differ because their HTTP clients do, and `docs/contract.md` records it as a divergence rather +than leaving you to find it. + +`transport(…)` and `timeout(…)` refuse each other, in whichever order you write them: a +timeout belongs to the transport that honours it, and `httpx` hands yours this budget as a +request extension it is free to ignore. A silent no-op would be worse than a `ConfigError`. +`transport(…)` is for `build()` and `async_transport(…)` for `build_async()`; using one with +the other build is a `ConfigError` too. + +## Testing your code + +Pass a transport and your control flow is testable with no server, no port and no key: + +```python +import httpx +from guideme import ApiKey, Guide, noul + + +def answer(_request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + json={ + "model": "jev-1.13.0", + "answers": {"q0": {"type": "noul", "noul": 0.95}}, + "usage": {"input_tokens": 296, "output_tokens": 20}, + }, + ) + + +def test_an_urgent_ticket_is_prioritised() -> None: + builder = Guide.builder().api_key(ApiKey("not-a-real-key")) + with builder.transport(httpx.MockTransport(answer)).build() as guide: + assert guide.ask(noul("Is this urgent?"), "payouts failing") is True +``` + +`q0` is the first question in encounter order; a batch of three is answered with `q0`, `q1` +and `q2`. Raise from the handler instead of returning and you get the failure paths: an +`httpx.ConnectError` is resent inside the retry budget, and an `httpx.ReadTimeout`, an +`httpx.ConnectTimeout` or an `httpx.RemoteProtocolError` is not. For +`AsyncGuide`, hand the same `httpx.MockTransport` to `async_transport(…)` and `build_async()`; +it is both kinds of transport at once. ## Observability @@ -313,7 +403,7 @@ One span named `guideme.ask` per request, shaped by the OpenTelemetry GenAI conv `gen_ai.request.model`, `gen_ai.response.model`, `gen_ai.usage.*`, and on failure `error.type` with an error status. Under it, one HTTP client span per attempt with `http.response.status_code`, so a retry is visible as sibling spans, plus a `guideme.retry` -event when an attempt is throttled. One `guideme.answer` event per question with the outcome, +event when an attempt is resent. One `guideme.answer` event per question with the outcome, the probability or confidence, the unsure verdict and the settled thresholds that produced it. The state is never recorded unless you opt in with `record_state(True)`. The API key never appears anywhere. @@ -341,41 +431,50 @@ and the value of `error.type` on the failed span. |---|---|---| | `AuthError` | `auth` | 401 | | `InvalidError` | `invalid` | 422; `.detail` is the body | -| `RateLimitedError` | `rate_limited` | 429 after retries, or a `retry-after` too long to wait for | -| `OverloadedError` | `overloaded` | 529 after retries | +| `RateLimitedError` | `rate_limited` | 429 after retries, or a `retry-after` too long to wait for; `.retry_after` carries it | +| `OverloadedError` | `overloaded` | 529 on the same terms; `.retry_after` carries it too | | `TransportError` | `transport` | connection, TLS, timeout | | `UnexpectedStatusError` | `unexpected_status` | anything the contract does not define | | `ProtocolError` | `protocol` | the response violates the contract: undecodable body, wrong answer kind, option or level not in the rubric, probability outside 0..1 | | `UnsureError` | `unsure` | the policy said unsure and nothing caught it | | `ConfigError` | `config` | raised where the mistake is written: bad thresholds, missing key, empty batch, unserialisable state, a duplicate rubric, a rubric outside 1..255 options or 2..10 levels, a bad `events(...)`, a non-positive timeout, negative retries or backoff, a `base_url` carrying credentials, and so on | -Retries on 429 and 529 use exponential backoff with jitter, capped at 30 s, and honour -`retry-after`. +Retries on 429 and 529 use exponential backoff with jitter, capped at 30 s, and honour an +integer `retry-after`. They apply to `models()` as much as to `ask`. + +A failed connection is resent in the same budget: refused, reset, or a TLS handshake that did +not complete. The request never reached a server, so nothing was judged and nothing is +repeated. A disconnect part-way through a response is **not** resent — the request arrived, +the API may have answered it, and asking again would buy the same judgment twice. + +**No timeout is resent, of any phase.** A connect timeout included, although `httpx` names it +separately: Rust's SDK sets one deadline over the whole attempt and cannot tell a connect +timeout from a read timeout, so retrying one here would make the two SDKs disagree about the +same failure, and a retried timeout multiplies the wall time `timeout(…)` is there to bound. +Everything not resent raises `TransportError` on the first failure. ## Lower layers Everything above is re-exported from the `guideme` package, and `guideme.__all__` is that list. -The three modules below are a second supported tier: you import them by their own path, they -are not re-exported at the top level, and they are under the same rule as the first tier — -nothing in them is removed or renamed without a major version and a `CHANGELOG.md` entry. -Anything else in the package is private, whatever its name looks like. - -- `guideme.api` is the exact wire mirror of `POST /v1/systemone` and `GET /v1/models`. - `guideme.api.client` holds `Client` and `AsyncClient` for callers who want to build requests - themselves. They live one level down rather than on `guideme.api` because re-exporting them - would make `api` and `api.client` import each other, and the gate fails an import cycle. -- `guideme.question.Question` is the type `noul`, `choose`, `choose_among`, `score`, - `score_levels` and every `.detail()` return. Inference covers most uses, so import it when - you need to annotate a question you are storing or passing on: a `dict` is invariant, so a - `dict[str, NoulQuestion]` is not a `dict[str, Question[bool]]` and the annotation has to be - written. It is out of `__all__` because the top-level surface is a fixed list, not because - the type is private. The promise covers that one name: everything else in `guideme.question` - is private. -- The scalars are validated once and never re-checked: `Probability` and `Confidence` hold the - unit-interval numbers on `Verdict`, `Ranked` and `Scored`, `Key` and `Rank` are what a runtime - rubric answers with, `Model` names the model to ask, and `ApiKey` carries the key without ever - printing it. The first four are `NewType` brands, so the guarantee is that only the wire mints - them, not that `Probability(2.0)` is rejected; it is not. + +Three modules are a second supported tier: `guideme.api`, `guideme.api.client` and +`guideme.policy`. You import those by their own path, they are not re-exported at the top +level, and they are under the same rule as the first tier — nothing in them is removed or +renamed without a major version and a `CHANGELOG.md` entry. Anything else in the package is +private, whatever its name looks like. + +The two bullets after them are not a tier. They say where some of the names above are +declared, which is worth knowing when two of them share a spelling. + +- `guideme.api` is the exact wire mirror of `POST /v1/systemone` and `GET /v1/models`, and + `guideme.api.__all__` is what it offers: the request and response models, its own `Usage`, + and the four adapters between them and the core. That `Usage` is the pydantic model a + response is parsed into, not the `Usage` a receipt carries — a receipt gets the frozen + dataclass of the same name from the top level, copied out of this one, so that nothing + pydantic sits on the surface you import from `guideme`. `guideme.api.client` holds `Client` and + `AsyncClient` for callers who want to build requests themselves. They live one level down + rather than on `guideme.api` because re-exporting them would make `api` and `api.client` + import each other, and the gate fails an import cycle. - `guideme.policy.resolve(answer, thresholds)` is the pure decision function. `spec/` holds its JSON Schemas and 42 golden vectors, vendored from [guideme-rust](https://github.com/pedro-pscunha/guideme-rust), which publishes the contract. @@ -383,6 +482,22 @@ Anything else in the package is private, whatever its name looks like. says what every guideme SDK must satisfy and [`docs/design.md`](https://github.com/pedro-pscunha/guideme-python/blob/main/docs/design.md) records the design and its sharp edges. +- `guideme.question` is where the question types are declared, and all of them are re-exported + above: `Question` is what `noul`, `choose`, `choose_among`, `score` and `score_levels` + return, and `NoulQuestion`, `ChoiceQuestion`, `ScoreQuestion`, `DetailedNoul`, + `DetailedChoice` and `DetailedScore` are the concrete ones. Inference covers most uses, so + reach for them when you need to annotate a question you are storing or passing on: a `dict` + is invariant, so a `dict[str, NoulQuestion]` is not a `dict[str, Question[bool]]` and the + annotation has to be written. Everything else in `guideme.question` is private. + `guideme.api` declares its own `Question`, `NoulQuestion`, `ChoiceQuestion` and + `ScoreQuestion`: same names, different classes. Those are the wire shapes the ones above + become on the way out, and you only meet them if you build requests by hand. The import + line says which you have. +- The scalars are validated once and never re-checked: `Probability` and `Confidence` hold the + unit-interval numbers on `Verdict`, `Ranked` and `Scored`, `Key` and `Rank` are what a runtime + rubric answers with, `Model` names the model to ask, and `ApiKey` carries the key without ever + printing it. The first four are `NewType` brands, so the guarantee is that only the wire mints + them, not that `Probability(2.0)` is rejected; it is not. ## Other SDKs diff --git a/docs/contract.md b/docs/contract.md index 7d4a85a..da12807 100644 --- a/docs/contract.md +++ b/docs/contract.md @@ -140,7 +140,7 @@ examples is never refused for its text, whatever that text is: it means what it this feature existed, and a patch release does not get to redefine it. Attaching examples to a blank rubric is the error, because they describe something that is not there. Both SDKs draw the line in the same place — Python inside `option()`, `level()` and `fallback()`, Rust in the -derive and in `Rubric::into_wire()` — so a declaration is legal in both or in neither. +derive and in `Rubric::render` — so a declaration is legal in both or in neither. **"Blank" means Unicode `White_Space`.** Rust's `str::trim` is exactly that property. Python's `str.strip()` is a superset: measured against the current runtime it strips 29 codepoints to @@ -170,11 +170,53 @@ the answer. Duplication asks whether two entries would put the same bytes in fro model, and a leading space does change that, because the text is rendered verbatim. Trimming for one and not the other is the only pairing that keeps both questions honest. -One asymmetry is deliberate. `choose_among` and `score_levels` here take an `option(…)` or a -`level(…)` value; Rust's equivalents keep taking a plain string, because widening their -signatures risks inference breakage for existing callers on a path that can already pass a -string its own renderer composed. It is revisited at 0.2.0. Equivalent inputs put identical -bytes on the wire either way, which is what the contract actually promises. +Both SDKs' runtime constructors take a rubric that carries examples — `option(…)` and +`level(…)` here, `Rubric` in Rust — and every rule above holds on **every** path, declaration +and runtime alike, in both: the rules a single rubric can see, the cross-option shared-example +rule, and the no-counterexample-on-a-level rule. Where they fire differs and nothing else +does. Python refuses the declaration where it is written, as a `ConfigError`; Rust refuses it +when the question is asked, as an `Error::Config`, because its constructors are infallible +values by design. A declaration is legal in both or in neither. + +### The interface shape, beyond the wire + +Two items of §4 of the published statement that this package satisfies, written here because +what they promise is behaviour a caller can see rather than bytes `spec/` can pin. + +**A receipt.** Alongside the answer, a caller can read the response's `model` — the versioned +id that answered, never the alias that was asked for — and its `usage`, the `input_tokens` and +`output_tokens` of the one request. Here that is `ask_with_receipt` returning `Receipt[T]` with +`answer`, `model` and `usage`; in Rust, `Guide::ask_with_receipt` returning +`Receipt { answer, model, usage }`. + +**Retry policy.** `429` and `529` are retried with exponential backoff honouring an integer +`retry-after`, on `POST /v1/systemone` and on `GET /v1/models` alike. A connection failure — +the request never reached a server: connect refused or reset, TLS handshake failure — is +retried inside the same budget. **A timeout of any phase** (connect, read, write) and a body +failure are not. + +The timeout rule is the one place where the wider language wins and the narrower one is held +to it. Rust sets a single deadline over the whole attempt, under which a connect-phase +timeout is indistinguishable from a read timeout: `reqwest` reports it as `is_timeout()`, not +`is_connect()`. Python can tell them apart — `httpx.ConnectTimeout` is its own class — and +declines to, because an SDK that resent one failure the other could not see would be the two +disagreeing about the same incident. A retried timeout also multiplies the wall time the +builder's `timeout` promises, which is the one number a caller sets to bound a call. So the +contract excludes every timeout and both SDKs implement that exclusion. + +After the last retry a `429` is a rate-limited error carrying the `retry-after` and a `529` +is an overloaded error carrying it too. + +### One divergence, and it is the language's + +`timeout` means different things in the two SDKs, and no amount of care makes it mean the +same. Rust's `reqwest` applies a deadline to the whole attempt. `httpx` has no per-request +deadline and instead spends the budget per phase — connecting, writing, reading, and waiting +for a pooled connection each get the whole of it — so an attempt that is slow in more than one +phase outlasts the number written in the builder. Wrapping it to match would need a different +wrapper for the synchronous and the asyncio surfaces and would change what cancellation means, +which is a worse trade than saying so. It is documented on `GuideBuilder.timeout`, in the +README's configuration section, and here. Nothing on the wire depends on it. Drift is caught rather than trusted. `mise run spec-check` clones guideme-rust, diffs its `spec/` against this one and fails on any difference except `spec/SOURCE`, which is provenance and has diff --git a/docs/design.md b/docs/design.md index 9fe13d9..572cdcf 100644 --- a/docs/design.md +++ b/docs/design.md @@ -92,6 +92,31 @@ Each module survives the test. `asyncio.run` rather than adding a pytest plugin. - **Jitter comes from the standard library.** `random.SystemRandom`, so backoff needs no extra dependency and does not disturb a caller who seeded the global `random`. +- **Only a failed connection is resent, and no timeout ever is.** `httpx.ConnectError` means + the request did not arrive, so nothing was judged and a resend repeats nothing. A + `RemoteProtocolError` and a body that will not decode mean it did arrive: the API may have + answered and billed it, and asking again would buy the same judgment twice. Idempotency is + the line, not whether the failure looks transient. + `httpx.ConnectTimeout` is the interesting exclusion, because idempotency alone would let it + through. It is excluded because the contract is shared and Rust cannot draw that line: + `reqwest` sets one deadline over the attempt, so a connect-phase timeout is `is_timeout()` + there and not `is_connect()`. An SDK that resent a failure the other could not even see + would be the two disagreeing about one incident. The second reason stands on its own: a + retried timeout multiplies the wall time `timeout(…)` is set to bound, and `httpx` already + spends that budget per phase, so the worst case is long enough without resending it. +- **A transport is injectable and refuses a timeout beside it.** `transport(…)` gives a caller + a proxy, a client certificate or an `httpx.MockTransport`, which is what makes their own + control flow testable without a server. `httpx` hands a transport the client's timeout as a + request extension it may ignore — `MockTransport` does — so a timeout set beside one is a + promise nothing keeps. It is a `ConfigError` in either order rather than a silent override. +- **`Receipt` has its own module, and declares its own `Usage`.** `_ask_overloads` names + `Receipt` in a return type and `guide` imports `_ask_overloads`, so it cannot live beside + `ModelInfo` in `guide` without a cycle. `guideme.receipt` imports nothing from the package, + which is what lets it sit that low: its `Usage` is a frozen dataclass of two `int`s, copied + out of `guideme.api.Usage` in `_receipt` the same way `_described` copies `ModelEntry` into + `ModelInfo`. The wire model keeps its name inside `guideme.api`. Exporting the pydantic one + would have saved a copy of two integers and put a dependency's whole surface — 28 attributes + that are not guideme's — on a published type, which is the thing `ModelInfo` exists to stop. - **Telemetry speaks OpenTelemetry.** The ask span uses the GenAI conventions, each HTTP attempt is its own client span with the HTTP conventions, and a failure is `error.type` plus an error span status rather than an error-level record. Anything without a convention is @@ -192,17 +217,47 @@ Each module survives the test. name the internal `Client` and the private `_Config`, because Python has no private constructor, not because either is supported. Build one through `Guide.builder()` or `Guide.from_env()`; those are what validate the policy and the origin before a socket opens. -- **`with_policy` shares the pool.** The copy holds the same client, so `close()` on either the - original or the copy closes the connection pool for both. -- **`Question` is supported, and not in `__all__`.** The top-level surface is a fixed list and - a question's type is whatever its constructor returns, so callers annotate by inference. - `guideme.question.Question` is in the second tier, beside `guideme.api` and - `guideme.policy`, and carries the same promise: import it from there when you need to write - the type of a stored question down. The promise is that one name. The rest of the module, - `validate`, `Spec` and the concrete question classes, is private, because naming the whole - module would publish all of it to let a caller annotate one thing. -- **Two `Question`s.** `guideme.question.Question` is the user-facing value; - `guideme.api`'s question model is the wire shape it becomes. +- **`with_policy` shares the pool, and the pool is counted.** The copy holds the same client + over the same `httpx` client, and a private `_Pool` counts its holders. Each guide releases + once and the transport closes when the last one does, so closing a derived guide leaves its + parent able to ask. Rust needs none of this: `Arc` drops when the last clone does. + Until 0.2.0 the count did not exist and `close()` on either guide closed both, which was + survivable as a documented sharp edge and would have been a trap the moment `with` existed. + The count alone is not enough, and the second half is that `share()` hands back a *new* + client over the same pool, each carrying its own release-once flag. Returning `self` would + give both guides one flag between them, so closing the first guide twice would spend the + second guide's hold and shut the pool under it — the count cannot tell which holder a + release came from. With a flag per client, closing one guide twice releases once. Sharing + from an already-closed client is refused rather than copied: it would hand back a client + holding nothing over a pool that may already be shut, and that only surfaces on the first + ask. The clamp at zero inside `drop` is the last line of defence, not the mechanism. + All of it — the count and every holder's flag — sits under one `threading.Lock` on the + pool, because `Guide` tells you to share a guide across threads and a flag read, a flag + flip and a decrement are three steps that must not interleave. The lock is taken when a + guide is derived and when one is closed, never per request: `ask` reads the `httpx` client + and nothing else. `httpx`'s own close happens after the lock is released, and no `await` + is ever reached while it is held, so the asyncio pool uses the same plain lock. + A patch that cannot settle is settled **before** the hold is taken. Building the derived + guide in one expression took the hold first, because Python evaluates arguments left to + right, and a bad patch then raised with nothing left to release it. +- **Some names mean two things, and the pairs are not renamed.** `Question`, + `NoulQuestion`, `ChoiceQuestion` and `ScoreQuestion` each name a user-facing value in + `guideme.question`, re-exported from `guideme`, and a pydantic wire model in `guideme.api`. + The first is what a caller builds and annotates; the second is the shape it becomes on the + way out. They never meet — nothing takes one where the other belongs, the wire models are + built only inside `question_to_wire`, and the import graph is one-way — so the collision + costs a reader one moment of "which one is this" that the import line answers, and renaming + either side would cost more. The wire names have to mirror the API's `type` tags, and the + user-facing ones are what a caller writes; a third spelling of either would be the one that + had to be explained. `guideme.api.NoulCriteria` and `guideme.question.NoulCriteria` are a + fifth such pair, for the same reason, and `api`'s module docstring already says so. + `Usage` is the pair 0.2.0 added and the only one where **both** halves sit in a published + `__all__`: `guideme.Usage` is the frozen dataclass a receipt carries and `guideme.api.Usage` + is the pydantic model the response is parsed into, and `_receipt` copies one into the other. + That is the cost of the decision above — not exporting the wire model is what creates a + second `Usage` — and it is the right way round: a caller reaching the top-level surface gets + the value object, and the name they would otherwise collide with is in a module they only + import when they are building requests by hand. - **State is JSON-shaped.** Anything `json.dumps` accepts without a default hook. A dataclass goes through `dataclasses.asdict`, a pydantic model through `.model_dump()`. This is the one untyped value in the package, and it is serialised at the boundary. diff --git a/docs/observability.md b/docs/observability.md index 44ac878..2c42ded 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -21,7 +21,7 @@ is announced to every other SDK. ``` guideme.ask span, kind CLIENT, one per ask ├── POST /v1/systemone span, kind CLIENT, one per HTTP attempt -│ └── guideme.retry event and WARN log record, when that attempt was throttled +│ └── guideme.retry event and WARN log record, when that attempt is being resent └── guideme.answer event and INFO log record, one per question ``` @@ -103,16 +103,37 @@ so it says what the model answered rather than what the caller ended up with. ### Event `guideme.retry` -Emitted inside the throttled attempt's span, just before the wait. It is the +Emitted inside the failed attempt's span, just before the wait. It is the warning-equivalent: OpenTelemetry span events carry no severity, so where the Rust SDK logs this at `WARN`, here the event's presence is the signal. | Field | Type | Meaning | |---|---|---| -| `http.response.status_code` | int | `429` or `529` | +| `http.response.status_code` | int | `429` or `529`; present only when a response arrived | +| `error.type` | str | `transport`; present only when none did | | `guideme.retry.attempt` | int | ordinal of the resend about to be made; `1` for the first retry | | `guideme.retry.delay_ms` | int | how long guideme is about to wait | +**Exactly one of the first two is on every event, and neither is ever a placeholder.** An +attempt is resent either because the API answered `429` or `529`, or because the connection +failed: refused, reset, or a TLS handshake that did not complete. Those two are different +incidents — a throttled API and an unreachable one — and a dashboard grouping retries by +cause must be able to tell them apart, so the cause is the field that is present rather than +a value inside one field. The attempt's own span is marked `error.type = transport` for the +second case, as it already is for a transport failure that is not resent. + +A timeout of any phase, a disconnect part-way through a response, and a body that will not +decode produce no retry event, because none of them is resent. The attempt span still +carries `error.type = transport` for them; only the retry event is absent, and its absence +is what says the call ended there. `docs/contract.md` has the reasoning, including why a +connect-phase timeout is excluded although `httpx` can name it. + +One caveat, and the Rust SDK carries the same one. A transport handed in through +`GuideBuilder.transport(…)` decides which exception a failure is raised as, and therefore +which side of that line it falls on: a custom transport that reports a connect timeout as +`httpx.ConnectError` will see it resent. guideme's own rule does not change — it classifies +what it is given — so what these events report stays exactly what the transport reported. + ### Log records Every answer and every retry is also an OTLP log record, so a backend with a logs pipeline @@ -125,15 +146,15 @@ cannot have: a severity, a message, and an identity of its own. | instrumentation scope | `guideme` | `guideme.api` | | event name | `guideme.answer` | `guideme.retry` | | severity text, number | `INFO`, 9 | `WARN`, 13 | -| body | `q0 noul: yes`, `q1 choice: billing`, `q2 score: level 1` | `429 from TypeSafe, retrying in 1000 ms` | +| body | `q0 noul: yes`, `q1 choice: billing`, `q2 score: level 1` | `429 from TypeSafe, retrying in 1000 ms`, or `could not reach TypeSafe, retrying in 500 ms` | | attributes | the `guideme.answer` table above, unchanged | the `guideme.retry` table above, unchanged | -| trace id, span id | the `guideme.ask` span's | the throttled attempt span's | +| trace id, span id | the `guideme.ask` span's | the resent attempt span's | The bodies are the messages the Rust SDK writes, so one saved query reads both SDKs. Correlation needs no configuration. A record resolves the active OpenTelemetry context when it is built, and guideme builds it inside the span the event belongs to, so an answer -points at its ask span and a retry at the attempt that was throttled. Nothing has to be +points at its ask span and a retry at the attempt being resent. Nothing has to be passed through, and there is nothing to get wrong. #### Choosing a signal diff --git a/examples/otlp/uv.lock b/examples/otlp/uv.lock index 88d00b9..1eeb339 100644 --- a/examples/otlp/uv.lock +++ b/examples/otlp/uv.lock @@ -103,7 +103,7 @@ wheels = [ [[package]] name = "guideme" -version = "0.1.1" +version = "0.2.0" source = { editable = "../../" } dependencies = [ { name = "httpx" }, diff --git a/pyproject.toml b/pyproject.toml index 0fc7092..b2373a7 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "guideme" -version = "0.1.1" +version = "0.2.0" description = "Type-safe inline judgments from TypeSafe Jev: a yes/no is an if, a choice is an exhaustive match, a score is a comparison." readme = "README.md" requires-python = ">=3.12" diff --git a/scripts/gen_ask_overloads.py b/scripts/gen_ask_overloads.py index eb256c0..b8b2426 100644 --- a/scripts/gen_ask_overloads.py +++ b/scripts/gen_ask_overloads.py @@ -44,24 +44,52 @@ TransportError: the request never completed: connection, TLS or timeout. UnexpectedStatusError: a status the contract does not define. """''' -"""The docstring on every generated `ask`. It is the package's only verb, and with +"""The docstring on every generated `ask`. It is the package's main verb, and with `py.typed` shipped this is what a caller's editor shows.""" +RECEIPT_DOC = '''"""Answer `shape` about `state`, with what the request cost and what answered it. + +Exactly `ask`, returning a `Receipt` instead of the answer alone: `ask` is this +call followed by `.answer`. Reach for it to attribute cost, or to pin a policy to +the model version whose numbers it was tuned against, and note that `model` is +the versioned id even where an alias such as `jev-latest` was asked for. + +Args: + shape: A question, or a tuple, list or dict of questions, nested freely. + Ids are `q0..qN` in encounter order, and `Receipt.answer` has the + shape of the request. + state: What to judge, as anything JSON-shaped. + +Returns: + A `Receipt` carrying the answers, the model that produced them, and the + token usage of the one request they came from. + +Raises: + The same errors as `ask`, on the same terms. +"""''' +"""The docstring on every generated `ask_with_receipt`. It defers to `ask`'s rather +than repeating nine `Raises:` lines that would then have to be kept in step.""" + HEADER = '''"""Typed `ask` surfaces. GENERATED by `scripts/gen_ask_overloads.py`; do not edit. Every shape whose types `ask` carries through is written out as an overload, because Python cannot map a type over a tuple. Any nesting still works at -runtime; the forms below are the ones a type checker can follow. +runtime; the forms below are the ones a type checker can follow. Each verb +carries the same family twice over, once returning the answer and once +returning a `Receipt` around it. Regenerate with `mise run gen`. """ +# pylint: disable=too-many-lines # 25 shapes x 2 verbs x 2 surfaces, all of it generated + from abc import ABC, abstractmethod from collections.abc import Awaitable from typing import overload from guideme._json import Json from guideme.question import Question +from guideme.receipt import Receipt ''' @@ -133,16 +161,23 @@ def shape_lines(shape: str) -> list[str]: ] -def overload_lines(parameters: str, shape: str, answer: str, *, is_async: bool) -> list[str]: +def returned(answer: str, *, is_async: bool, wrapped: bool) -> str: + """What one signature returns: the answer, maybe in a `Receipt`, maybe awaitable.""" + inner = f"Receipt[{answer}]" if wrapped else answer + return f"Awaitable[{inner}]" if is_async else inner + + +def overload_lines( + verb: str, parameters: str, shape: str, answer: str, *, is_async: bool, wrapped: bool +) -> list[str]: """One `@overload` declaration, in the shape `ruff format` leaves alone.""" - returns = f"Awaitable[{answer}]" if is_async else answer return [ " @overload", - f" def ask[{parameters}](", + f" def {verb}[{parameters}](", " self,", *shape_lines(shape), " state: Json,", - f" ) -> {returns}: ...", + f" ) -> {returned(answer, is_async=is_async, wrapped=wrapped)}: ...", ] @@ -151,25 +186,38 @@ def doc_lines(doc: str, indent: str) -> list[str]: return [f"{indent}{line}" if line else "" for line in doc.splitlines()] +def verb_lines(verb: str, doc: str, *, is_async: bool, wrapped: bool) -> list[str]: + """One public verb: its whole overload family, then the implementation they cover.""" + lines: list[str] = [] + for parameters, shape, answer in shapes(): + lines.append("") + lines += overload_lines(verb, parameters, shape, answer, is_async=is_async, wrapped=wrapped) + returns = returned("object", is_async=is_async, wrapped=wrapped) + lines += [ + "", + f" def {verb}(self, shape: object, state: Json) -> {returns}:", + *doc_lines(doc, " "), + f" return self._{verb}(shape, state)", + ] + return lines + + def surface(name: str, summary: str, *, is_async: bool) -> list[str]: """One overload-carrying base class, sync or async.""" - returns = "Awaitable[object]" if is_async else "object" + plain = returned("object", is_async=is_async, wrapped=False) + receipt = returned("object", is_async=is_async, wrapped=True) lines = [ f"class {name}(ABC):", f' """{summary}"""', "", " @abstractmethod", - f" def _ask(self, shape: object, state: Json) -> {returns}: ...", - ] - for parameters, shape, answer in shapes(): - lines.append("") - lines += overload_lines(parameters, shape, answer, is_async=is_async) - lines += [ + f" def _ask(self, shape: object, state: Json) -> {plain}: ...", "", - f" def ask(self, shape: object, state: Json) -> {returns}:", - *doc_lines(ASK_DOC, " "), - " return self._ask(shape, state)", + " @abstractmethod", + f" def _ask_with_receipt(self, shape: object, state: Json) -> {receipt}: ...", ] + lines += verb_lines("ask", ASK_DOC, is_async=is_async, wrapped=False) + lines += verb_lines("ask_with_receipt", RECEIPT_DOC, is_async=is_async, wrapped=True) return lines @@ -179,13 +227,13 @@ def main() -> None: lines += ["", ""] lines += surface( "SyncAskOverloads", - "The typed `ask` surface of `Guide`; `_ask` does the work.", + "The typed `ask` surfaces of `Guide`; the two `_ask` methods do the work.", is_async=False, ) lines += ["", ""] lines += surface( "AsyncAskOverloads", - "The typed `ask` surface of `AsyncGuide`; `_ask` does the work.", + "The typed `ask` surfaces of `AsyncGuide`; the two `_ask` methods do the work.", is_async=True, ) print("\n".join(lines)) diff --git a/spec/SOURCE b/spec/SOURCE index f6b8d6e..b63258e 100644 --- a/spec/SOURCE +++ b/spec/SOURCE @@ -1 +1 @@ -d908ecfec39f267802781b10be74323286e661af +1565ce95015196e8f49bbc18dcb4b4b0cf429aed diff --git a/src/guideme/__init__.py b/src/guideme/__init__.py index 7c34762..fac9a47 100644 --- a/src/guideme/__init__.py +++ b/src/guideme/__init__.py @@ -16,14 +16,22 @@ from guideme.guide import AsyncGuide, Guide, GuideBuilder, ModelInfo from guideme.policy import Policy, Thresholds, Verdict from guideme.question import ( + ChoiceQuestion, + DetailedChoice, + DetailedNoul, + DetailedScore, + NoulQuestion, + Question, Ranked, Scored, + ScoreQuestion, choose, choose_among, noul, score, score_levels, ) +from guideme.receipt import Receipt, Usage from guideme.scalars import ApiKey, Confidence, Key, Model, Probability, Rank __all__ = [ @@ -31,8 +39,12 @@ "AsyncGuide", "AuthError", "Choice", + "ChoiceQuestion", "Confidence", "ConfigError", + "DetailedChoice", + "DetailedNoul", + "DetailedScore", "Guide", "GuideBuilder", "GuidemeError", @@ -41,18 +53,23 @@ "Levels", "Model", "ModelInfo", + "NoulQuestion", "OverloadedError", "Policy", "Probability", "ProtocolError", + "Question", "Rank", "Ranked", "RateLimitedError", + "Receipt", + "ScoreQuestion", "Scored", "Thresholds", "TransportError", "UnexpectedStatusError", "UnsureError", + "Usage", "Verdict", "choose", "choose_among", diff --git a/src/guideme/_ask_overloads.py b/src/guideme/_ask_overloads.py index 4c28a87..26599ea 100644 --- a/src/guideme/_ask_overloads.py +++ b/src/guideme/_ask_overloads.py @@ -2,25 +2,33 @@ Every shape whose types `ask` carries through is written out as an overload, because Python cannot map a type over a tuple. Any nesting still works at -runtime; the forms below are the ones a type checker can follow. +runtime; the forms below are the ones a type checker can follow. Each verb +carries the same family twice over, once returning the answer and once +returning a `Receipt` around it. Regenerate with `mise run gen`. """ +# pylint: disable=too-many-lines # 25 shapes x 2 verbs x 2 surfaces, all of it generated + from abc import ABC, abstractmethod from collections.abc import Awaitable from typing import overload from guideme._json import Json from guideme.question import Question +from guideme.receipt import Receipt class SyncAskOverloads(ABC): - """The typed `ask` surface of `Guide`; `_ask` does the work.""" + """The typed `ask` surfaces of `Guide`; the two `_ask` methods do the work.""" @abstractmethod def _ask(self, shape: object, state: Json) -> object: ... + @abstractmethod + def _ask_with_receipt(self, shape: object, state: Json) -> Receipt[object]: ... + @overload def ask[T]( self, @@ -357,53 +365,46 @@ def ask(self, shape: object, state: Json) -> object: """ return self._ask(shape, state) - -class AsyncAskOverloads(ABC): - """The typed `ask` surface of `AsyncGuide`; `_ask` does the work.""" - - @abstractmethod - def _ask(self, shape: object, state: Json) -> Awaitable[object]: ... - @overload - def ask[T]( + def ask_with_receipt[T]( self, shape: Question[T], state: Json, - ) -> Awaitable[T]: ... + ) -> Receipt[T]: ... @overload - def ask[T]( + def ask_with_receipt[T]( self, shape: list[Question[T]], state: Json, - ) -> Awaitable[list[T]]: ... + ) -> Receipt[list[T]]: ... @overload - def ask[K, T]( + def ask_with_receipt[K, T]( self, shape: dict[K, Question[T]], state: Json, - ) -> Awaitable[dict[K, T]]: ... + ) -> Receipt[dict[K, T]]: ... @overload - def ask[T0]( + def ask_with_receipt[T0]( self, shape: tuple[Question[T0]], state: Json, - ) -> Awaitable[tuple[T0]]: ... + ) -> Receipt[tuple[T0]]: ... @overload - def ask[T0, T1]( + def ask_with_receipt[T0, T1]( self, shape: tuple[ Question[T0], Question[T1], ], state: Json, - ) -> Awaitable[tuple[T0, T1]]: ... + ) -> Receipt[tuple[T0, T1]]: ... @overload - def ask[T0, T1, T2]( + def ask_with_receipt[T0, T1, T2]( self, shape: tuple[ Question[T0], @@ -411,10 +412,10 @@ def ask[T0, T1, T2]( Question[T2], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2]]: ... + ) -> Receipt[tuple[T0, T1, T2]]: ... @overload - def ask[T0, T1, T2, T3]( + def ask_with_receipt[T0, T1, T2, T3]( self, shape: tuple[ Question[T0], @@ -423,10 +424,10 @@ def ask[T0, T1, T2, T3]( Question[T3], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3]]: ... @overload - def ask[T0, T1, T2, T3, T4]( + def ask_with_receipt[T0, T1, T2, T3, T4]( self, shape: tuple[ Question[T0], @@ -436,10 +437,10 @@ def ask[T0, T1, T2, T3, T4]( Question[T4], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5]( self, shape: tuple[ Question[T0], @@ -450,10 +451,10 @@ def ask[T0, T1, T2, T3, T4, T5]( Question[T5], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5, T6]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6]( self, shape: tuple[ Question[T0], @@ -465,10 +466,10 @@ def ask[T0, T1, T2, T3, T4, T5, T6]( Question[T6], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5, T6]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5, T6, T7]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6, T7]( self, shape: tuple[ Question[T0], @@ -481,20 +482,20 @@ def ask[T0, T1, T2, T3, T4, T5, T6, T7]( Question[T7], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6, T7]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5, T6, T7]]: ... @overload - def ask[T0, K, T]( + def ask_with_receipt[T0, K, T]( self, shape: tuple[ Question[T0], dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, dict[K, T]]]: ... @overload - def ask[T0, T1, K, T]( + def ask_with_receipt[T0, T1, K, T]( self, shape: tuple[ Question[T0], @@ -502,10 +503,10 @@ def ask[T0, T1, K, T]( dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, T1, dict[K, T]]]: ... @overload - def ask[T0, T1, T2, K, T]( + def ask_with_receipt[T0, T1, T2, K, T]( self, shape: tuple[ Question[T0], @@ -514,10 +515,10 @@ def ask[T0, T1, T2, K, T]( dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, dict[K, T]]]: ... @overload - def ask[T0, T1, T2, T3, K, T]( + def ask_with_receipt[T0, T1, T2, T3, K, T]( self, shape: tuple[ Question[T0], @@ -527,10 +528,10 @@ def ask[T0, T1, T2, T3, K, T]( dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, dict[K, T]]]: ... @overload - def ask[T0, T1, T2, T3, T4, K, T]( + def ask_with_receipt[T0, T1, T2, T3, T4, K, T]( self, shape: tuple[ Question[T0], @@ -541,10 +542,10 @@ def ask[T0, T1, T2, T3, T4, K, T]( dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, dict[K, T]]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5, K, T]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5, K, T]( self, shape: tuple[ Question[T0], @@ -556,10 +557,10 @@ def ask[T0, T1, T2, T3, T4, T5, K, T]( dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5, dict[K, T]]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5, T6, K, T]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6, K, T]( self, shape: tuple[ Question[T0], @@ -572,20 +573,20 @@ def ask[T0, T1, T2, T3, T4, T5, T6, K, T]( dict[K, Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6, dict[K, T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5, T6, dict[K, T]]]: ... @overload - def ask[T0, T]( + def ask_with_receipt[T0, T]( self, shape: tuple[ Question[T0], list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, list[T]]]: ... + ) -> Receipt[tuple[T0, list[T]]]: ... @overload - def ask[T0, T1, T]( + def ask_with_receipt[T0, T1, T]( self, shape: tuple[ Question[T0], @@ -593,10 +594,10 @@ def ask[T0, T1, T]( list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, list[T]]]: ... + ) -> Receipt[tuple[T0, T1, list[T]]]: ... @overload - def ask[T0, T1, T2, T]( + def ask_with_receipt[T0, T1, T2, T]( self, shape: tuple[ Question[T0], @@ -605,10 +606,10 @@ def ask[T0, T1, T2, T]( list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, list[T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, list[T]]]: ... @overload - def ask[T0, T1, T2, T3, T]( + def ask_with_receipt[T0, T1, T2, T3, T]( self, shape: tuple[ Question[T0], @@ -618,10 +619,10 @@ def ask[T0, T1, T2, T3, T]( list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, list[T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, list[T]]]: ... @overload - def ask[T0, T1, T2, T3, T4, T]( + def ask_with_receipt[T0, T1, T2, T3, T4, T]( self, shape: tuple[ Question[T0], @@ -632,10 +633,10 @@ def ask[T0, T1, T2, T3, T4, T]( list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, list[T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, list[T]]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5, T]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T]( self, shape: tuple[ Question[T0], @@ -647,10 +648,10 @@ def ask[T0, T1, T2, T3, T4, T5, T]( list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, list[T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5, list[T]]]: ... @overload - def ask[T0, T1, T2, T3, T4, T5, T6, T]( + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6, T]( self, shape: tuple[ Question[T0], @@ -663,39 +664,697 @@ def ask[T0, T1, T2, T3, T4, T5, T6, T]( list[Question[T]], ], state: Json, - ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6, list[T]]]: ... + ) -> Receipt[tuple[T0, T1, T2, T3, T4, T5, T6, list[T]]]: ... - def ask(self, shape: object, state: Json) -> Awaitable[object]: - """Answer `shape` about `state` in one request and one `guideme.ask` span. + def ask_with_receipt(self, shape: object, state: Json) -> Receipt[object]: + """Answer `shape` about `state`, with what the request cost and what answered it. - A batch is atomic: one answer the policy cannot resolve fails the whole call, so - put `.otherwise(...)` or `.detail()` on the questions that may come back unsure. + Exactly `ask`, returning a `Receipt` instead of the answer alone: `ask` is this + call followed by `.answer`. Reach for it to attribute cost, or to pin a policy to + the model version whose numbers it was tuned against, and note that `model` is + the versioned id even where an alias such as `jev-latest` was asked for. Args: shape: A question, or a tuple, list or dict of questions, nested freely. - Ids are `q0..qN` in encounter order, and the answer has the shape of - the request. + Ids are `q0..qN` in encounter order, and `Receipt.answer` has the + shape of the request. state: What to judge, as anything JSON-shaped. Returns: - The answers, in the shape of `shape`. + A `Receipt` carrying the answers, the model that produced them, and the + token usage of the one request they came from. Raises: - ConfigError: `shape` holds something that is not a question, a question's - own thresholds are outside `0..=1` or leave `no_below` above - `yes_above`, the batch is empty, or `state` or the instructions cannot - be serialised to JSON. A rubric this call could not ask is refused - where the question is built, not here. - UnsureError: the policy read an answer as unsure and neither - `.otherwise(...)` nor a `fallback(...)` member caught it. - ProtocolError: the response breaks the contract: an undecodable body, the - wrong answer kind, an option or level outside the rubric, or a - probability outside `0..=1`. - AuthError: the key was refused (`401`). - InvalidError: the API rejected the request (`422`). - RateLimitedError: still throttled after the retries (`429`). - OverloadedError: still overloaded after the retries (`529`). - TransportError: the request never completed: connection, TLS or timeout. - UnexpectedStatusError: a status the contract does not define. + The same errors as `ask`, on the same terms. """ - return self._ask(shape, state) + return self._ask_with_receipt(shape, state) + + +class AsyncAskOverloads(ABC): + """The typed `ask` surfaces of `AsyncGuide`; the two `_ask` methods do the work.""" + + @abstractmethod + def _ask(self, shape: object, state: Json) -> Awaitable[object]: ... + + @abstractmethod + def _ask_with_receipt(self, shape: object, state: Json) -> Awaitable[Receipt[object]]: ... + + @overload + def ask[T]( + self, + shape: Question[T], + state: Json, + ) -> Awaitable[T]: ... + + @overload + def ask[T]( + self, + shape: list[Question[T]], + state: Json, + ) -> Awaitable[list[T]]: ... + + @overload + def ask[K, T]( + self, + shape: dict[K, Question[T]], + state: Json, + ) -> Awaitable[dict[K, T]]: ... + + @overload + def ask[T0]( + self, + shape: tuple[Question[T0]], + state: Json, + ) -> Awaitable[tuple[T0]]: ... + + @overload + def ask[T0, T1]( + self, + shape: tuple[ + Question[T0], + Question[T1], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1]]: ... + + @overload + def ask[T0, T1, T2]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2]]: ... + + @overload + def ask[T0, T1, T2, T3]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3]]: ... + + @overload + def ask[T0, T1, T2, T3, T4]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5, T6]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5, T6, T7]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + Question[T7], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6, T7]]: ... + + @overload + def ask[T0, K, T]( + self, + shape: tuple[ + Question[T0], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, dict[K, T]]]: ... + + @overload + def ask[T0, T1, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, dict[K, T]]]: ... + + @overload + def ask[T0, T1, T2, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, dict[K, T]]]: ... + + @overload + def ask[T0, T1, T2, T3, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, dict[K, T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, dict[K, T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, dict[K, T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5, T6, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6, dict[K, T]]]: ... + + @overload + def ask[T0, T]( + self, + shape: tuple[ + Question[T0], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, list[T]]]: ... + + @overload + def ask[T0, T1, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, list[T]]]: ... + + @overload + def ask[T0, T1, T2, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, list[T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, list[T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, list[T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, list[T]]]: ... + + @overload + def ask[T0, T1, T2, T3, T4, T5, T6, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[tuple[T0, T1, T2, T3, T4, T5, T6, list[T]]]: ... + + def ask(self, shape: object, state: Json) -> Awaitable[object]: + """Answer `shape` about `state` in one request and one `guideme.ask` span. + + A batch is atomic: one answer the policy cannot resolve fails the whole call, so + put `.otherwise(...)` or `.detail()` on the questions that may come back unsure. + + Args: + shape: A question, or a tuple, list or dict of questions, nested freely. + Ids are `q0..qN` in encounter order, and the answer has the shape of + the request. + state: What to judge, as anything JSON-shaped. + + Returns: + The answers, in the shape of `shape`. + + Raises: + ConfigError: `shape` holds something that is not a question, a question's + own thresholds are outside `0..=1` or leave `no_below` above + `yes_above`, the batch is empty, or `state` or the instructions cannot + be serialised to JSON. A rubric this call could not ask is refused + where the question is built, not here. + UnsureError: the policy read an answer as unsure and neither + `.otherwise(...)` nor a `fallback(...)` member caught it. + ProtocolError: the response breaks the contract: an undecodable body, the + wrong answer kind, an option or level outside the rubric, or a + probability outside `0..=1`. + AuthError: the key was refused (`401`). + InvalidError: the API rejected the request (`422`). + RateLimitedError: still throttled after the retries (`429`). + OverloadedError: still overloaded after the retries (`529`). + TransportError: the request never completed: connection, TLS or timeout. + UnexpectedStatusError: a status the contract does not define. + """ + return self._ask(shape, state) + + @overload + def ask_with_receipt[T]( + self, + shape: Question[T], + state: Json, + ) -> Awaitable[Receipt[T]]: ... + + @overload + def ask_with_receipt[T]( + self, + shape: list[Question[T]], + state: Json, + ) -> Awaitable[Receipt[list[T]]]: ... + + @overload + def ask_with_receipt[K, T]( + self, + shape: dict[K, Question[T]], + state: Json, + ) -> Awaitable[Receipt[dict[K, T]]]: ... + + @overload + def ask_with_receipt[T0]( + self, + shape: tuple[Question[T0]], + state: Json, + ) -> Awaitable[Receipt[tuple[T0]]]: ... + + @overload + def ask_with_receipt[T0, T1]( + self, + shape: tuple[ + Question[T0], + Question[T1], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5, T6]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6, T7]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + Question[T7], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5, T6, T7]]]: ... + + @overload + def ask_with_receipt[T0, K, T]( + self, + shape: tuple[ + Question[T0], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6, K, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + dict[K, Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5, T6, dict[K, T]]]]: ... + + @overload + def ask_with_receipt[T0, T]( + self, + shape: tuple[ + Question[T0], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, list[T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, list[T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, list[T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, list[T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, list[T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5, list[T]]]]: ... + + @overload + def ask_with_receipt[T0, T1, T2, T3, T4, T5, T6, T]( + self, + shape: tuple[ + Question[T0], + Question[T1], + Question[T2], + Question[T3], + Question[T4], + Question[T5], + Question[T6], + list[Question[T]], + ], + state: Json, + ) -> Awaitable[Receipt[tuple[T0, T1, T2, T3, T4, T5, T6, list[T]]]]: ... + + def ask_with_receipt(self, shape: object, state: Json) -> Awaitable[Receipt[object]]: + """Answer `shape` about `state`, with what the request cost and what answered it. + + Exactly `ask`, returning a `Receipt` instead of the answer alone: `ask` is this + call followed by `.answer`. Reach for it to attribute cost, or to pin a policy to + the model version whose numbers it was tuned against, and note that `model` is + the versioned id even where an alias such as `jev-latest` was asked for. + + Args: + shape: A question, or a tuple, list or dict of questions, nested freely. + Ids are `q0..qN` in encounter order, and `Receipt.answer` has the + shape of the request. + state: What to judge, as anything JSON-shaped. + + Returns: + A `Receipt` carrying the answers, the model that produced them, and the + token usage of the one request they came from. + + Raises: + The same errors as `ask`, on the same terms. + """ + return self._ask_with_receipt(shape, state) diff --git a/src/guideme/api/__init__.py b/src/guideme/api/__init__.py index 5f6cca3..9382856 100644 --- a/src/guideme/api/__init__.py +++ b/src/guideme/api/__init__.py @@ -33,6 +33,36 @@ from guideme.question import ChoiceSpec, NoulSpec, ScoreSpec, Spec from guideme.scalars import confidence, probability +__all__ = [ + "Answer", + "ChoiceAnswer", + "ChoiceQuestion", + "Count", + "ModelEntry", + "ModelsResponse", + "NoulAnswer", + "NoulCriteria", + "NoulQuestion", + "Question", + "Request", + "Response", + "ScoreAnswer", + "ScoreQuestion", + "Unit", + "Usage", + "answer_from_wire", + "question_to_wire", + "request_to_wire", + "validation_detail", +] +"""What this module owns. + +Everything here is declared below; nothing imported into it is re-exported. Without +this list a `from guideme.api import *` would hand back `BaseModel`, `Field` and +`Mapping` as though they were part of guideme's surface, and a reader looking for the +wire would have to tell the mirror from what the mirror is built out of. +""" + class _Sent(BaseModel): """Base of everything guideme puts on the wire. diff --git a/src/guideme/api/client.py b/src/guideme/api/client.py index 79cca0c..3618412 100644 --- a/src/guideme/api/client.py +++ b/src/guideme/api/client.py @@ -3,7 +3,8 @@ Retries `429` and `529` with exponential backoff, honouring `retry-after`, and maps every status to a typed error. Each attempt is one span shaped by the OpenTelemetry HTTP client conventions, so a retried request is sibling spans under the ask span, -each with its own status code, and a throttled one also carries a retry event. +each with its own status code, and one that is about to be resent also carries a retry +event. `Client` and `AsyncClient` are the same flow twice. Everything either of them decides is decided by `step`, which is pure; what is written out twice is the `await` and the @@ -12,9 +13,12 @@ import asyncio import time -from dataclasses import dataclass +from collections.abc import Callable +from copy import copy +from dataclasses import dataclass, field from datetime import timedelta from random import SystemRandom +from threading import Lock from typing import Self, final import httpx @@ -66,6 +70,17 @@ RETRYABLE = frozenset({TOO_MANY_REQUESTS, OVERLOADED}) """The two statuses guideme resends after. Everything else fails on the first attempt.""" +type Transport = httpx.BaseTransport +"""A caller-supplied synchronous transport, as `GuideBuilder.transport` takes one. + +Named here so that `guideme.guide` can offer the setter without importing `httpx`: this +module stays the package's only importer of it, and a caller still writes +`httpx.MockTransport` or `httpx.HTTPTransport` in their own code. +""" + +type AsyncTransport = httpx.AsyncBaseTransport +"""A caller-supplied asyncio transport, named here for the same reason as `Transport`.""" + _KNOWN_PORTS = {"http": 80, "https": 443} _CAP_EXPONENT = 20 _random = SystemRandom() @@ -110,7 +125,7 @@ def url(self, path: str) -> str: @final @dataclass(frozen=True, slots=True) class RetryPolicy: - """How often to resend a throttled request, and how long to wait first.""" + """How often to resend a request that failed, and how long to wait first.""" max_retries: int = 3 backoff: timedelta = timedelta(milliseconds=500) @@ -134,6 +149,60 @@ def delay(self, attempt: int) -> timedelta: return exponential + JITTER * _random.random() +@final +@dataclass(slots=True) +class _Pool[T]: + """One `httpx` client, the number of clients holding it, and the lock over both. + + A guide derived with `with_policy` shares the pool its parent opened, and Python has + no `Arc` to count that for us. So the count is explicit: every holder releases once, + and the transport closes when the last one does. Closing a derived guide therefore + leaves its parent able to ask, which is what makes `with` safe on both. + + The count alone is not enough, and `Client.close` is where the rest of it lives: a + holder that is closed twice must release once. Each `share()` hands back a distinct + client carrying its own flag for that, so no count kept here can be spent by the + wrong holder. The clamp in `drop` is the last line of defence rather than the + mechanism. + + **`guard` covers the count and every holder's flag together.** `Guide`'s docstring + says to share a guide across threads, so two threads closing two guides over one pool + is a documented thing to do; a flag read, a flag flip and a decrement are three steps, + and interleaving them either closes a live transport or leaks it. Hold `guard` across + all three. `take` and `drop` therefore do not lock: their caller is already inside it, + and a lock taken twice would deadlock. Nothing slow happens under it — `httpx`'s own + close is called after it is released, and no `await` is ever reached while it is held, + so the asyncio client can use the same plain lock as the synchronous one. + + `ask` never touches any of this. It reads `http` and nothing else, so the pool's lock + is taken once when a guide is derived and once when one is closed, never per request. + """ + + http: T + holders: int = 1 + guard: Lock = field(default_factory=Lock) + + def take(self) -> None: + """Take one more hold, for a client derived from one already holding it. + + Call with `guard` held. + """ + self.holders += 1 + + def drop(self) -> bool: + """Drop one hold. True when this was the last, so the transport must be closed. + + Call with `guard` held. Every caller is a `Client.close` that has just flipped its + own flag to spent in the same critical section, so in practice the count never + reaches here already at zero. It is clamped anyway: returning `True` off a + negative count would close a live transport. + """ + if not self.holders: + return False + self.holders -= 1 + return not self.holders + + @final @dataclass(frozen=True, slots=True) class Retry: @@ -142,32 +211,86 @@ class Retry: delay: timedelta -type Step = Response | Retry | GuidemeError -"""What one attempt concluded: a decoded response, a resend, or the error to raise.""" +@final +@dataclass(frozen=True, slots=True) +class Ready: + """The attempt succeeded. The body is the caller's to read, inside the attempt's span.""" + + body: str + + +type Step = Ready | Retry | GuidemeError +"""What one attempt concluded: a body to read, a resend, or the error to raise.""" def step( status: int, body: str, retry_after: timedelta | None, attempt: int, retry: RetryPolicy ) -> Step: - """Classify one attempt. + """Classify one attempt that got a response. Pure: no I/O, no sleeping, no telemetry. A `retry-after` longer than `MAX_BACKOFF` is not waited for; the call fails carrying that duration so the caller decides. + + Both endpoints go through this, so a `429` on `GET /v1/models` is resent on exactly + the terms one on `POST /v1/systemone` is. The API's own docs say an SDK handles a + `429` for you, and saying it of one endpoint only would be a promise with a hole in it. """ if status == OK: - return _decode(body) + return Ready(body) if status not in RETRYABLE: return _classify(status, body) if attempt >= retry.max_retries or (retry_after is not None and retry_after > MAX_BACKOFF): - return RateLimitedError(retry_after) if status == TOO_MANY_REQUESTS else OverloadedError() + if status == TOO_MANY_REQUESTS: + return RateLimitedError(retry_after) + return OverloadedError(retry_after) return Retry(retry.delay(attempt) if retry_after is None else retry_after) -def _decode(body: str) -> Response | ProtocolError: +def resend_after(error: httpx.HTTPError, attempt: int, retry: RetryPolicy) -> timedelta | None: + """How long to wait before resending a request that never got a response. + + `None` means do not resend, for either of two reasons: the failure was not a + connection failure, or the budget for them is spent. The budget is the one `step` + spends on a `429`, so a call cannot exceed `max_retries + 1` attempts by mixing them. + + Only `httpx.ConnectError` is resent: a refused or reset connection, or a TLS handshake + that failed. The request never reached a server, so nothing was judged and nothing is + repeated by trying again. A read timeout and a disconnect part-way through a response + mean the opposite: the request arrived, the API may have answered it, and a resend + would ask for the same judgment a second time. A body that fails to decode is not here + at all; it arrived, and it is a `ProtocolError`. + + **A timeout is never resent, whatever phase it names.** `httpx.ConnectTimeout` looks + like a connection failure and is excluded anyway, for two reasons. Rust has no such + case to exclude: `reqwest` sets one deadline over the whole attempt, so a connect-phase + timeout there is `is_timeout()` and not `is_connect()`, and retrying one here would be + the two SDKs disagreeing about the same failure. And a retried timeout multiplies the + wall time `GuideBuilder.timeout` promises, which is the one number a caller sets to + bound how long a call may take. + """ + if not isinstance(error, httpx.ConnectError): + return None + if attempt >= retry.max_retries: + return None + return retry.delay(attempt) + + +def _evaluated(body: str) -> Response: + """Read one `POST /v1/systemone` body. Raises `ProtocolError` when it is not one.""" try: return Response.model_validate_json(body) except ValidationError as error: - return ProtocolError(f"response body: {validation_detail(error)}") + detail = f"response body: {validation_detail(error)}" + raise ProtocolError(detail) from error + + +def _listed(body: str) -> list[ModelEntry]: + """Read one `GET /v1/models` body. Raises `ProtocolError` when it is not one.""" + try: + return ModelsResponse.model_validate_json(body).models + except ValidationError as error: + detail = f"models body: {validation_detail(error)}" + raise ProtocolError(detail) from error def _classify(status: int, body: str) -> GuidemeError: @@ -178,11 +301,6 @@ def _classify(status: int, body: str) -> GuidemeError: return UnexpectedStatusError(status, body) -def _attempt_error(status: int, error: GuidemeError) -> str: - """`error.type` for an attempt span: the status when one arrived, else the error's kind.""" - return error.kind if status == OK else str(status) - - def _retry_after(headers: httpx.Headers) -> timedelta | None: """Read `retry-after` as whole seconds. Anything else is treated as absent, like Rust's.""" raw = headers.get("retry-after") @@ -232,21 +350,13 @@ def _scrub(error: httpx.HTTPError) -> None: _ = request.headers.pop("authorization", None) -def _models(span: Span, status: int, body: str, retry_after: timedelta | None) -> list[ModelEntry]: - """Read one `GET /v1/models` response, marking the span on the way out.""" - if status != OK: - fail_attempt(span, str(status)) - if status == TOO_MANY_REQUESTS: - raise RateLimitedError(retry_after) - if status == OVERLOADED: - raise OverloadedError - raise _classify(status, body) +def _read[T](span: Span, body: str, decode: Callable[[str], T]) -> T: + """Read one successful body, marking the attempt's span when it is not the shape promised.""" try: - return ModelsResponse.model_validate_json(body).models - except ValidationError as error: - failure = ProtocolError(f"models body: {validation_detail(error)}") + return decode(body) + except ProtocolError as failure: fail_attempt(span, failure.kind) - raise failure from error + raise def _transport(span: Span, error: httpx.HTTPError) -> TransportError: @@ -268,31 +378,85 @@ def __init__( retry: RetryPolicy, timeout: timedelta, events: Events, + transport: Transport | None = None, ) -> None: """Open the connection pool. The key is held as an `ApiKey`, never as a header. `events` is read back by the guide that owns this client, so the routing knob has one home: this client emits the retries under it, the guide emits the answers. + + `transport` is the caller's, when they supplied one; `None` leaves `httpx` to + build its own. The builder refuses a transport beside a timeout, so the two never + arrive together. """ self.endpoint = endpoint # Declared, because pyright widens a `Literal` inferred from an assignment. self.events: Events = events self._retry = retry self._api_key = api_key - self._http = httpx.Client(timeout=timeout.total_seconds(), follow_redirects=True) + self._pool = _Pool( + httpx.Client( + timeout=timeout.total_seconds(), follow_redirects=True, transport=transport + ) + ) + self._holding = True + + def share(self) -> "Client": + """A second client over this one's pool, for a guide derived from this one's. + + A distinct object rather than `self`, because the hold is what has to be released + exactly once and `self` cannot carry two of them. Two guides over one returned + `self` would share one flag, so closing the first guide twice would spend the + second guide's hold and shut the pool under it. + + Everything else is shared: the same pool, key, endpoint, retry policy and event + routing, so the copy costs nothing and no setting can drift between the two. The + copy carries this client's own flag, which is why a closed one is refused rather + than copied: sharing from it would hand back a client holding nothing, over a + pool that may already be shut, and the mistake would only surface on the first ask. + + Raises: + ConfigError: this client has already been closed. + """ + with self._pool.guard: + if not self._holding: + detail = "cannot share a closed client" + raise ConfigError(detail) + self._pool.take() + return copy(self) def evaluate(self, request: Request) -> Response: - """`POST /v1/systemone`, resending a `429` or a `529` up to the retry policy's limit.""" - url = self.endpoint.url(EVALUATE) - body = request.model_dump_json(by_alias=True) + """`POST /v1/systemone`, resent while the API throttles or a connection fails.""" + return self._fetch(EVALUATE, request.model_dump_json(by_alias=True), _evaluated) + + def models(self) -> list[ModelEntry]: + """`GET /v1/models`, resent on exactly the terms an ask is.""" + return self._fetch(MODELS, None, _listed) + + def _fetch[T](self, path: str, body: str | None, decode: Callable[[str], T]) -> T: + """The retry loop, which is the whole of what this client decides how to do. + + One span per attempt, one `guideme.retry` before each wait, and the body read + inside the span it arrived on so a malformed one marks the attempt that carried + it. Both endpoints run through here: what tells them apart is a body to send. + """ + url = self.endpoint.url(path) + method = "GET" if body is None else "POST" for attempt in range(self._retry.max_retries + 1): with attempt_span( - "POST", url, EVALUATE, self.endpoint.host, self.endpoint.port, attempt + method, url, path, self.endpoint.host, self.endpoint.port, attempt ) as span: try: - response = self._http.post(url, content=body, headers=_sending(self._api_key)) + response = self._send(url, body) except httpx.HTTPError as error: - raise _transport(span, error) from error + delay = resend_after(error, attempt, self._retry) + if delay is None: + raise _transport(span, error) from error + _scrub(error) + fail_attempt(span, TransportError.kind) + retry_event(span, None, attempt + 1, delay, self.events) + time.sleep(delay.total_seconds()) + continue record_status(span, response.status_code) outcome = step( response.status_code, @@ -302,34 +466,38 @@ def evaluate(self, request: Request) -> Response: self._retry, ) match outcome: - case Response(): - return outcome + case Ready(body=text): + return _read(span, text, decode) case Retry(delay=delay): fail_attempt(span, str(response.status_code)) retry_event(span, response.status_code, attempt + 1, delay, self.events) time.sleep(delay.total_seconds()) case GuidemeError(): - fail_attempt(span, _attempt_error(response.status_code, outcome)) + fail_attempt(span, str(response.status_code)) raise outcome # Unreachable: the loop runs at least once and every arm returns or raises. raise RateLimitedError(None) - def models(self) -> list[ModelEntry]: - """`GET /v1/models`. Not retried.""" - url = self.endpoint.url(MODELS) - with attempt_span("GET", url, MODELS, self.endpoint.host, self.endpoint.port, 0) as span: - try: - response = self._http.get(url, headers=_auth(self._api_key)) - except httpx.HTTPError as error: - raise _transport(span, error) from error - record_status(span, response.status_code) - return _models( - span, response.status_code, response.text, _retry_after(response.headers) - ) + def _send(self, url: str, body: str | None) -> httpx.Response: + """One attempt on the wire. The bearer is built here and lives only as long as it.""" + if body is None: + return self._pool.http.get(url, headers=_auth(self._api_key)) + return self._pool.http.post(url, content=body, headers=_sending(self._api_key)) def close(self) -> None: - """Close the connection pool.""" - self._http.close() + """Release this client's hold on the pool, closing it when this was the last hold. + + Idempotent: a client closed twice releases once. The flag is per client, so a + second close here can never spend a hold that belongs to a client `share()` + handed out. + """ + with self._pool.guard: + if not self._holding: + return + self._holding = False + last = self._pool.drop() + if last: + self._pool.http.close() @final @@ -343,33 +511,69 @@ def __init__( retry: RetryPolicy, timeout: timedelta, events: Events, + transport: AsyncTransport | None = None, ) -> None: """Open the connection pool. The key is held as an `ApiKey`, never as a header. `events` is read back by the guide that owns this client, so the routing knob has one home: this client emits the retries under it, the guide emits the answers. + + `transport` is the caller's, when they supplied one; `None` leaves `httpx` to + build its own. The builder refuses a transport beside a timeout, so the two never + arrive together. """ self.endpoint = endpoint # Declared, because pyright widens a `Literal` inferred from an assignment. self.events: Events = events self._retry = retry self._api_key = api_key - self._http = httpx.AsyncClient(timeout=timeout.total_seconds(), follow_redirects=True) + self._pool = _Pool( + httpx.AsyncClient( + timeout=timeout.total_seconds(), follow_redirects=True, transport=transport + ) + ) + self._holding = True + + def share(self) -> "AsyncClient": + """A second client over this one's pool. `Client.share` says why it is a new object. + + Raises: + ConfigError: this client has already been closed. + """ + with self._pool.guard: + if not self._holding: + detail = "cannot share a closed client" + raise ConfigError(detail) + self._pool.take() + return copy(self) async def evaluate(self, request: Request) -> Response: - """`POST /v1/systemone`, resending a `429` or a `529` up to the retry policy's limit.""" - url = self.endpoint.url(EVALUATE) - body = request.model_dump_json(by_alias=True) + """`POST /v1/systemone`, resent while the API throttles or a connection fails.""" + return await self._fetch(EVALUATE, request.model_dump_json(by_alias=True), _evaluated) + + async def models(self) -> list[ModelEntry]: + """`GET /v1/models`, resent on exactly the terms an ask is.""" + return await self._fetch(MODELS, None, _listed) + + async def _fetch[T](self, path: str, body: str | None, decode: Callable[[str], T]) -> T: + """`Client._fetch` with the two `await`s. Every decision in it is made the same way.""" + url = self.endpoint.url(path) + method = "GET" if body is None else "POST" for attempt in range(self._retry.max_retries + 1): with attempt_span( - "POST", url, EVALUATE, self.endpoint.host, self.endpoint.port, attempt + method, url, path, self.endpoint.host, self.endpoint.port, attempt ) as span: try: - response = await self._http.post( - url, content=body, headers=_sending(self._api_key) - ) + response = await self._send(url, body) except httpx.HTTPError as error: - raise _transport(span, error) from error + delay = resend_after(error, attempt, self._retry) + if delay is None: + raise _transport(span, error) from error + _scrub(error) + fail_attempt(span, TransportError.kind) + retry_event(span, None, attempt + 1, delay, self.events) + await asyncio.sleep(delay.total_seconds()) + continue record_status(span, response.status_code) outcome = step( response.status_code, @@ -379,31 +583,30 @@ async def evaluate(self, request: Request) -> Response: self._retry, ) match outcome: - case Response(): - return outcome + case Ready(body=text): + return _read(span, text, decode) case Retry(delay=delay): fail_attempt(span, str(response.status_code)) retry_event(span, response.status_code, attempt + 1, delay, self.events) await asyncio.sleep(delay.total_seconds()) case GuidemeError(): - fail_attempt(span, _attempt_error(response.status_code, outcome)) + fail_attempt(span, str(response.status_code)) raise outcome # Unreachable: the loop runs at least once and every arm returns or raises. raise RateLimitedError(None) - async def models(self) -> list[ModelEntry]: - """`GET /v1/models`. Not retried.""" - url = self.endpoint.url(MODELS) - with attempt_span("GET", url, MODELS, self.endpoint.host, self.endpoint.port, 0) as span: - try: - response = await self._http.get(url, headers=_auth(self._api_key)) - except httpx.HTTPError as error: - raise _transport(span, error) from error - record_status(span, response.status_code) - return _models( - span, response.status_code, response.text, _retry_after(response.headers) - ) + async def _send(self, url: str, body: str | None) -> httpx.Response: + """One attempt on the wire. The bearer is built here and lives only as long as it.""" + if body is None: + return await self._pool.http.get(url, headers=_auth(self._api_key)) + return await self._pool.http.post(url, content=body, headers=_sending(self._api_key)) async def close(self) -> None: - """Close the connection pool.""" - await self._http.aclose() + """Release this client's hold on the pool. `Client.close` says why it is idempotent.""" + with self._pool.guard: + if not self._holding: + return + self._holding = False + last = self._pool.drop() + if last: + await self._pool.http.aclose() diff --git a/src/guideme/ask.py b/src/guideme/ask.py index 83efe7b..2e092f6 100644 --- a/src/guideme/ask.py +++ b/src/guideme/ask.py @@ -7,7 +7,7 @@ from collections.abc import Mapping from dataclasses import dataclass, field -from typing import cast, final +from typing import TypeGuard, final from guideme._json import Json from guideme.errors import ConfigError, ProtocolError @@ -48,30 +48,58 @@ def push(self, question: Question[object]) -> str: return qid +# The four predicates below are what lets `encode` take `object` and still type-check +# without a `cast`. A bare `isinstance` narrows `object` to `tuple[Unknown, ...]`, and +# strict's `reportUnknown*` family then fires on every use of the elements; annotating the +# result does not help, because narrowing intersects with the declaration rather than +# replacing it. A `TypeGuard` replaces it outright, so the element type is `object`, which +# is exactly what `encode` accepts and therefore claims nothing the isinstance has not +# already proved. `TypeIs` would be the tighter spelling and needs 3.12 to mean this; +# `TypeGuard` has meant it since 3.10, which is under the floor this package supports. + + +def _is_question(value: object) -> TypeGuard[Question[object]]: + """A question of anything. + + Only `.policy`, `.instructions`, `.spec` and `.read()` are used from here on, and + `read()` widens to `object`, so no `T` is ever written back in. + """ + return isinstance(value, Question) + + +def _is_tuple(value: object) -> TypeGuard[tuple[object, ...]]: + """A tuple of anything. The element type is `object`, which is what `encode` takes.""" + return isinstance(value, tuple) + + +def _is_list(value: object) -> TypeGuard[list[object]]: + """A list of anything, for the same reason as `_is_tuple`.""" + return isinstance(value, list) + + +def _is_dict(value: object) -> TypeGuard[dict[object, object]]: + """A dict of anything, for the same reason as `_is_tuple`.""" + return isinstance(value, dict) + + def encode(shape: object, plan: Plan) -> Claim: - """Walk the shape in encounter order, minting ids. Anything else is a `ConfigError`.""" - match shape: - case Question(): - # Only .policy, .instructions, .spec and .read() are used from here - # on, and read() widens to object; no T is ever written back in. - question = cast("Question[object]", shape) - return Answered(plan.push(question), question) - # The three container arms narrow `shape` from `object` to a bare `tuple`, - # `list` or `dict`, whose element and key types pyright cannot know. Each - # cast below names them `object`, which is exactly what `encode` accepts, so - # none of them claims anything the arm has not already proved. - case tuple(): - return tuple(encode(item, plan) for item in cast("tuple[object, ...]", shape)) - case list(): - return [encode(item, plan) for item in cast("list[object]", shape)] - case dict(): - items = cast("dict[object, object]", shape).items() - return {key: encode(item, plan) for key, item in items} - case _: - # Over `object`, not over one of this package's own enums: this arm - # is how anything that is not a shape gets rejected. - detail = f"not a question shape: {type(shape).__name__}" - raise ConfigError(detail) + """Walk the shape in encounter order, minting ids. Anything else is a `ConfigError`. + + An if-chain rather than a `match`, because the narrowing is what the predicates above + are for and a class pattern cannot call one. The fall-through is how anything that is + not a shape gets rejected, and it is over `object` rather than one of this package's + own unions, so it is not the catch-all arm the invariants forbid. + """ + if _is_question(shape): + return Answered(plan.push(shape), shape) + if _is_tuple(shape): + return tuple(encode(item, plan) for item in shape) + if _is_list(shape): + return [encode(item, plan) for item in shape] + if _is_dict(shape): + return {key: encode(item, plan) for key, item in shape.items()} + detail = f"not a question shape: {type(shape).__name__}" + raise ConfigError(detail) def decode(claim: Claim, reply: Reply) -> object: diff --git a/src/guideme/enums.py b/src/guideme/enums.py index 900e50e..9d95ead 100644 --- a/src/guideme/enums.py +++ b/src/guideme/enums.py @@ -16,6 +16,15 @@ from guideme.errors import ConfigError from guideme.policy import MAX_LEVELS, MAX_OPTIONS, MIN_LEVELS +__all__ = ["Choice", "Levels", "fallback", "level", "option"] +"""What this module offers a caller, all of it re-exported from `guideme` itself. + +`render`, `require_unshared_examples`, `require_no_counterexamples` and +`require_no_fallback` are internal despite their names: `guideme.question` imports them +and no caller ever does. They are in no tier, and leaving them out of this list is what +says so. +""" + MIN_OPTIONS = 1 """Fewest options a choice may carry.""" diff --git a/src/guideme/errors.py b/src/guideme/errors.py index e0889b0..de6c803 100644 --- a/src/guideme/errors.py +++ b/src/guideme/errors.py @@ -8,6 +8,22 @@ from datetime import timedelta from typing import ClassVar, Literal, final, override +__all__ = [ + "AuthError", + "ConfigError", + "ErrorKind", + "GuidemeError", + "InvalidError", + "OverloadedError", + "ProtocolError", + "RateLimitedError", + "TransportError", + "UnexpectedStatusError", + "UnsureError", +] +"""The whole tree, all of it re-exported from `guideme` itself except `ErrorKind`, +which is the type of `kind` rather than something a caller catches.""" + type ErrorKind = Literal[ "auth", "invalid", @@ -78,13 +94,18 @@ def __init__(self, retry_after: timedelta | None) -> None: @final class OverloadedError(GuidemeError): - """`529` after every retry.""" + """`529` after every retry, or a `retry-after` longer than the cap.""" kind = "overloaded" - def __init__(self) -> None: - """Build the error; the API sends no detail with a 529.""" - super().__init__("TypeSafe is overloaded") + def __init__(self, retry_after: timedelta | None) -> None: + """Build the error around the last `retry-after` the API sent, if any. + + A `529` carries the header as often as a `429` does, so it is kept here for the + same reason: it is the only thing that says when a caller may come back. + """ + super().__init__(f"TypeSafe is overloaded (retry-after: {retry_after})") + self.retry_after = retry_after @final diff --git a/src/guideme/guide.py b/src/guideme/guide.py index a638852..6d9881c 100644 --- a/src/guideme/guide.py +++ b/src/guideme/guide.py @@ -26,10 +26,19 @@ question_to_wire, request_to_wire, ) -from guideme.api.client import DEFAULT_BASE_URL, AsyncClient, Client, Endpoint, RetryPolicy +from guideme.api.client import ( + DEFAULT_BASE_URL, + AsyncClient, + AsyncTransport, + Client, + Endpoint, + RetryPolicy, + Transport, +) from guideme.ask import Claim, Plan, decode, encode from guideme.errors import ConfigError, GuidemeError, ProtocolError from guideme.policy import Outcome, Policy, Thresholds, resolve +from guideme.receipt import Receipt, Usage from guideme.scalars import ApiKey, Model from guideme.telemetry import ( EVENT_MODES, @@ -51,6 +60,15 @@ MODEL_VAR = "GUIDEME_MODEL" """Optional model override for `from_env`.""" +_BOTH_TRANSPORTS = ( + "transport and async_transport cannot both be set; a builder carrying both can build " + "neither kind of guide, so the second one is refused where it is written" +) +"""Why only one transport may be set. One builder produces one guide, and `build()` refuses +an async transport while `build_async()` refuses a sync one, so a builder holding both is +already unbuildable; saying so at the setter beats two build errors that each name the +other setter.""" + DEFAULT_TIMEOUT = timedelta(seconds=30) """How long one phase of one attempt may take. @@ -166,6 +184,19 @@ def _finish(span: Span, prepared: _Prepared, response: Response, events: Events) return decode(prepared.claim, reply) +def _receipt(answer: object, response: Response) -> Receipt[object]: + """Put one answered shape beside what the response said it cost and what produced it. + + The two counts are copied out of the wire model rather than handed over inside it, + so what a caller holds is this package's own value object. `_described` does the same + for `models()`, and for the same reason. + """ + usage = Usage( + input_tokens=response.usage.input_tokens, output_tokens=response.usage.output_tokens + ) + return Receipt(answer=answer, model=response.model, usage=usage) + + def _merged(config: _Config, policy: Policy) -> _Config: """Patch a policy over a guide's, settling it now so a bad patch fails where it is written.""" merged = policy.over(config.policy) @@ -190,7 +221,15 @@ def _describe(kind: str, config: _Config, endpoint: Endpoint) -> str: @final class Guide(SyncAskOverloads): - """A configured entry point to Jev. Build one and share it; it owns a connection pool.""" + """A configured entry point to Jev. Build one and share it; it owns a connection pool. + + Share it across threads: a guide is frozen configuration over one `httpx.Client`, and + both are safe to use from several threads at once. One guide asking concurrently is + what the pool is for. Building one per request works and opens a pool per request, + which is the cost the pool exists to avoid. Close it once, by hand or by leaving a + `with` block; a guide derived with `with_policy` is a second holder of the same pool + and closing it leaves this one able to ask. + """ def __init__(self, client: Client, config: _Config) -> None: """Wrap a built client. Use `Guide.from_env` or `Guide.builder` instead.""" @@ -210,21 +249,42 @@ def builder() -> "GuideBuilder": def with_policy(self, policy: Policy) -> "Guide": """A guide sharing this client, with `policy` patched over this one's. - The connection pool is shared, so `close()` on either guide closes it for both. + The connection pool is shared and counted: the guide returned here is a second + holder of it, so closing either one leaves the other able to ask and the pool + closes when the last of them does. Each guide holds and releases on its own, so + closing one of them twice releases once and never touches the other's hold. + + The patch is settled before the hold is taken. Python evaluates arguments left to + right, so building the guide in one expression took the hold first and leaked it + when a bad patch raised: the pool was left holding a client that never existed. + + Raises: + ConfigError: the patched policy's thresholds are out of range, or this guide + has already been closed, so there is no hold to share. """ - return Guide(self._client, _merged(self._config, policy)) + config = _merged(self._config, policy) + return Guide(self._client.share(), config) + + def __enter__(self) -> Self: + """Enter a `with` block. The guide is ready to ask before this; nothing is opened here.""" + return self + + def __exit__(self, kind: object, error: object, traceback: object) -> None: + """Leave a `with` block by closing this guide. Nothing is suppressed.""" + self.close() def models(self) -> tuple[ModelInfo, ...]: """The models this account may use. No ask span of its own. - One `GET /v1/models`, never retried: unlike `ask`, a `429` or a `529` raises on the - first attempt rather than after a backoff. + One `GET /v1/models`, resent on exactly the terms an ask is: a `429` or a `529` + waits out the backoff and goes again, and a connection failure does too. A `429` + while a process is starting up therefore does not fail the start. Raises: AuthError: the key was missing or rejected. InvalidError: the API rejected the request (`422`). - RateLimitedError: the API answered `429`. - OverloadedError: the API answered `529`. + RateLimitedError: still throttled after the retries (`429`). + OverloadedError: still overloaded after the retries (`529`). TransportError: the request never completed. UnexpectedStatusError: any other status. ProtocolError: the body did not match the contract. @@ -241,6 +301,10 @@ def __repr__(self) -> str: @override def _ask(self, shape: object, state: Json) -> object: + return self._ask_with_receipt(shape, state).answer + + @override + def _ask_with_receipt(self, shape: object, state: Json) -> Receipt[object]: prepared = _prepare(shape, state, self._config) with _open(prepared, self._config, self._client.endpoint) as span: try: @@ -249,12 +313,20 @@ def _ask(self, shape: object, state: Json) -> object: except GuidemeError as error: fail_ask(span, error) raise - return decoded + return _receipt(decoded, response) @final class AsyncGuide(AsyncAskOverloads): - """A configured entry point to Jev for `asyncio`. The same surface as `Guide`.""" + """A configured entry point to Jev for `asyncio`. The same surface as `Guide`. + + Share it across tasks: a guide is frozen configuration over one `httpx.AsyncClient`, + and concurrent asks from one guide are what the pool is for — `asyncio.gather` over + a batch of them is the intended shape. It belongs to the event loop it was built on. + Close it once, by hand or by leaving an `async with` block; a guide derived with + `with_policy` is a second holder of the same pool and closing it leaves this one able + to ask. + """ def __init__(self, client: AsyncClient, config: _Config) -> None: """Wrap a built client. Use `AsyncGuide.from_env` or `AsyncGuide.builder` instead.""" @@ -274,21 +346,42 @@ def builder() -> "GuideBuilder": def with_policy(self, policy: Policy) -> "AsyncGuide": """A guide sharing this client, with `policy` patched over this one's. - The connection pool is shared, so `close()` on either guide closes it for both. + The connection pool is shared and counted: the guide returned here is a second + holder of it, so closing either one leaves the other able to ask and the pool + closes when the last of them does. Each guide holds and releases on its own, so + closing one of them twice releases once and never touches the other's hold. + + The patch is settled before the hold is taken. Python evaluates arguments left to + right, so building the guide in one expression took the hold first and leaked it + when a bad patch raised: the pool was left holding a client that never existed. + + Raises: + ConfigError: the patched policy's thresholds are out of range, or this guide + has already been closed, so there is no hold to share. """ - return AsyncGuide(self._client, _merged(self._config, policy)) + config = _merged(self._config, policy) + return AsyncGuide(self._client.share(), config) + + async def __aenter__(self) -> Self: + """Enter an `async with` block. Nothing is opened here; the guide is already ready.""" + return self + + async def __aexit__(self, kind: object, error: object, traceback: object) -> None: + """Leave an `async with` block by closing this guide. Nothing is suppressed.""" + await self.close() async def models(self) -> tuple[ModelInfo, ...]: """The models this account may use. No ask span of its own. - One `GET /v1/models`, never retried: unlike `ask`, a `429` or a `529` raises on the - first attempt rather than after a backoff. + One `GET /v1/models`, resent on exactly the terms an ask is: a `429` or a `529` + waits out the backoff and goes again, and a connection failure does too. A `429` + while a process is starting up therefore does not fail the start. Raises: AuthError: the key was missing or rejected. InvalidError: the API rejected the request (`422`). - RateLimitedError: the API answered `429`. - OverloadedError: the API answered `529`. + RateLimitedError: still throttled after the retries (`429`). + OverloadedError: still overloaded after the retries (`529`). TransportError: the request never completed. UnexpectedStatusError: any other status. ProtocolError: the body did not match the contract. @@ -308,7 +401,15 @@ def _ask(self, shape: object, state: Json) -> Awaitable[object]: """Match the base exactly: a plain call returning something awaitable.""" return self._answer(shape, state) + @override + def _ask_with_receipt(self, shape: object, state: Json) -> Awaitable[Receipt[object]]: + """Match the base exactly: a plain call returning something awaitable.""" + return self._answer_with_receipt(shape, state) + async def _answer(self, shape: object, state: Json) -> object: + return (await self._answer_with_receipt(shape, state)).answer + + async def _answer_with_receipt(self, shape: object, state: Json) -> Receipt[object]: prepared = _prepare(shape, state, self._config) with _open(prepared, self._config, self._client.endpoint) as span: try: @@ -317,7 +418,7 @@ async def _answer(self, shape: object, state: Json) -> object: except GuidemeError as error: fail_ask(span, error) raise - return decoded + return _receipt(decoded, response) @final @@ -335,7 +436,11 @@ def __init__(self) -> None: self._model: Model = Model.latest() self._policy: Policy = Policy() self._retry: RetryPolicy = RetryPolicy() - self._timeout: timedelta = DEFAULT_TIMEOUT + # `None` is "the caller never said", which is not the same as "the default", and + # only the first of the two may sit beside an injected transport. + self._timeout: timedelta | None = None + self._transport: Transport | None = None + self._async_transport: AsyncTransport | None = None self._record_state: bool = False self._events: Events = "both" @@ -414,19 +519,71 @@ def backoff(self, base: timedelta) -> Self: def timeout(self, per_attempt: timedelta) -> Self: """Timeout for one phase of one attempt; 30 s by default. - `httpx` applies it to connecting, writing, reading and pool acquisition - separately rather than as one deadline for the attempt, so an attempt that is - slow in more than one phase can outlast it. See `DEFAULT_TIMEOUT`. + **This is not a deadline for the attempt.** `httpx` gives the whole budget to + each phase separately — connecting, writing, reading, and waiting for a pooled + connection — so one attempt that is slow in more than one phase takes longer + than `per_attempt` and is not in breach of anything. Worst case for a call is + `(max_retries + 1)` attempts of several phases each, plus the backoff between + them. Rust's `reqwest` deadline covers the attempt as a whole instead; the two + SDKs differ here because their HTTP clients do, and `docs/contract.md` records + it as a divergence rather than leaving a reader to find it. + + A timeout belongs to the transport that honours it, so this and + `transport(...)`/`async_transport(...)` refuse each other in whichever order + they are written. An injected transport decides its own deadlines, and `httpx` + hands it this budget as a request extension it is free to ignore — as + `httpx.MockTransport` does — so accepting both would promise a timeout that + nothing applies. Raises: - ConfigError: `per_attempt` is zero or negative. + ConfigError: `per_attempt` is zero or negative, or a transport is already set. """ if per_attempt <= timedelta(): detail = f"timeout {per_attempt} is not positive" raise ConfigError(detail) + if self._transport is not None or self._async_transport is not None: + detail = "timeout and an injected transport conflict; the transport owns its deadlines" + raise ConfigError(detail) self._timeout = per_attempt return self + def transport(self, transport: Transport) -> Self: + """Send through this `httpx.BaseTransport` rather than one `httpx` opens. + + A proxy, a client certificate, or an `httpx.MockTransport` that answers a test + without a socket — the README's "Testing your code" section is the last of those + written out. It is used by `build()`; `build_async()` needs `async_transport`. + + Raises: + ConfigError: `timeout(...)` or `async_transport(...)` is already set. + """ + self._refuse_a_timeout("transport") + if self._async_transport is not None: + raise ConfigError(_BOTH_TRANSPORTS) + self._transport = transport + return self + + def async_transport(self, transport: AsyncTransport) -> Self: + """Send through this `httpx.AsyncBaseTransport` rather than one `httpx` opens. + + The asyncio half of `transport`, used by `build_async()`. `httpx.MockTransport` + is both kinds at once, so one of those can be given to either setter. + + Raises: + ConfigError: `timeout(...)` or `transport(...)` is already set. + """ + self._refuse_a_timeout("async_transport") + if self._transport is not None: + raise ConfigError(_BOTH_TRANSPORTS) + self._async_transport = transport + return self + + def _refuse_a_timeout(self, setter: str) -> None: + """Refuse a transport written after a timeout, as `timeout` refuses the other order.""" + if self._timeout is not None: + detail = f"{setter} and timeout conflict; the transport owns its deadlines" + raise ConfigError(detail) + def events(self, where: Events) -> Self: """Where an answer and a retry are written: `"span"`, `"log"` or `"both"`. @@ -463,22 +620,39 @@ def build(self) -> Guide: Raises: ConfigError: no API key was set, the `base_url` is malformed or - carries credentials, or the policy's thresholds are out of range. + carries credentials, the policy's thresholds are out of range, or + `async_transport(...)` was set, which only `build_async` can use. """ + if self._async_transport is not None: + detail = "build() cannot use an async_transport; use build_async() or transport()" + raise ConfigError(detail) key, endpoint, config = self._settle() - return Guide(Client(key, endpoint, self._retry, self._timeout, self._events), config) + client = Client( + key, endpoint, self._retry, self._settled_timeout(), self._events, self._transport + ) + return Guide(client, config) def build_async(self) -> AsyncGuide: """Build an asyncio guide, validating the policy and the origin now. Raises: ConfigError: no API key was set, the `base_url` is malformed or - carries credentials, or the policy's thresholds are out of range. + carries credentials, the policy's thresholds are out of range, or + `transport(...)` was set, which only `build` can use. """ + if self._transport is not None: + detail = "build_async() cannot use a transport; use build() or async_transport()" + raise ConfigError(detail) key, endpoint, config = self._settle() - client = AsyncClient(key, endpoint, self._retry, self._timeout, self._events) + client = AsyncClient( + key, endpoint, self._retry, self._settled_timeout(), self._events, self._async_transport + ) return AsyncGuide(client, config) + def _settled_timeout(self) -> timedelta: + """The timeout to build with: the caller's, or the default they never overrode.""" + return DEFAULT_TIMEOUT if self._timeout is None else self._timeout + def _settle(self) -> tuple[ApiKey, Endpoint, _Config]: """Everything both builds need, with every check done before a socket is opened.""" if self._api_key is None: diff --git a/src/guideme/policy.py b/src/guideme/policy.py index c259c64..797a2d7 100644 --- a/src/guideme/policy.py +++ b/src/guideme/policy.py @@ -12,6 +12,25 @@ from guideme.errors import ConfigError, ProtocolError from guideme.scalars import Confidence, Probability +__all__ = [ + "Answer", + "ChoiceAnswer", + "ChoiceOutcome", + "NoulAnswer", + "NoulOutcome", + "Outcome", + "Policy", + "ScoreAnswer", + "ScoreOutcome", + "Thresholds", + "Verdict", + "VerdictLabel", + "resolve", +] +"""What this module offers, which is the second tier: `guideme.policy` is imported by +its own path rather than re-exported. `resolve` and the shapes it reads and returns are +the whole of it; the bounds constants beside them are the package's own.""" + MAX_OPTIONS = 255 """Most options a choice may carry.""" diff --git a/src/guideme/question.py b/src/guideme/question.py index dc42be0..5b0c10f 100644 --- a/src/guideme/question.py +++ b/src/guideme/question.py @@ -33,6 +33,30 @@ ) from guideme.scalars import Confidence, Key, Probability, Rank +__all__ = [ + "ChoiceQuestion", + "DetailedChoice", + "DetailedNoul", + "DetailedScore", + "NoulQuestion", + "Question", + "Ranked", + "ScoreQuestion", + "Scored", + "choose", + "choose_among", + "noul", + "score", + "score_levels", +] +"""What this module offers a caller, all of it re-exported from `guideme` itself. + +`validate`, `Spec` and the criteria shapes are left out on purpose: they are how the +package carries a question, not how anyone writes one. `Question` is the exception the +`guideme.question` tier exists for, because it is the only way to write down the type +of a stored question. +""" + @final @dataclass(frozen=True, slots=True) diff --git a/src/guideme/receipt.py b/src/guideme/receipt.py new file mode 100644 index 0000000..f7c7dfa --- /dev/null +++ b/src/guideme/receipt.py @@ -0,0 +1,49 @@ +"""What an ask cost and which model answered it, returned beside the answer. + +Its own module rather than a name in `guideme.guide`, because the generated `ask` +surfaces name `Receipt` in their return types and `guideme.guide` imports those. One +module below them is where the type has to live for the graph to stay one-way. It +imports nothing from the package, which is what lets it sit that low. +""" + +from dataclasses import dataclass +from typing import final + + +@final +@dataclass(frozen=True, slots=True) +class Usage: + """What one request cost, in tokens. + + A plain value object rather than the wire model `guideme.api.Usage`, for the reason + `ModelInfo` is one: a caller holding what an ask returned should never have to name a + pydantic type, and a type on this surface should not carry a dependency's methods or + change shape when that dependency has a major release. + """ + + input_tokens: int + """What is billed.""" + + output_tokens: int + """What the judgment came back as. Free, and worth watching anyway.""" + + +@final +@dataclass(frozen=True, slots=True) +class Receipt[T]: + """An answer with what the request cost and which model produced it. + + `Guide.ask` returns the answer alone, which is the shape almost every call wants. + `ask_with_receipt` returns this instead, for cost attribution and for pinning a + policy to the model version that produced the numbers it was tuned against. + """ + + answer: T + """What `ask` would have returned: the caller's shape, answered.""" + + model: str + """The versioned model that answered, even where an alias such as `jev-latest` was asked + for. Log it: thresholds are tuned against one model's numbers.""" + + usage: Usage + """Input and output tokens for the whole request.""" diff --git a/src/guideme/telemetry.py b/src/guideme/telemetry.py index 845e1d5..d732fb6 100644 --- a/src/guideme/telemetry.py +++ b/src/guideme/telemetry.py @@ -24,7 +24,7 @@ from opentelemetry.trace import Span, SpanKind, StatusCode, get_tracer -from guideme.errors import GuidemeError +from guideme.errors import GuidemeError, TransportError from guideme.policy import ChoiceOutcome, NoulOutcome, Outcome, ScoreOutcome, Thresholds if TYPE_CHECKING: @@ -51,7 +51,7 @@ """One per question answered.""" RETRY_EVENT = "guideme.retry" -"""One per throttled attempt, just before the wait. The warning of the Rust SDK.""" +"""One per attempt about to be resent, just before the wait. The warning of the Rust SDK.""" type Events = Literal["span", "log", "both"] """Where an answer or a retry is written: the span, an OTLP log record, or both. @@ -366,20 +366,35 @@ def fail_attempt(span: Span, error_type: str) -> None: span.set_status(StatusCode.ERROR) -def retry_event(span: Span, status: int, attempt: int, delay: timedelta, events: Events) -> None: +def retry_event( + span: Span, status: int | None, attempt: int, delay: timedelta, events: Events +) -> None: """One `guideme.retry`. `attempt` is the ordinal of the resend about to be made. + `status` is the one a response arrived with, or `None` when the connection failed: + refused, reset, or a TLS handshake that did not complete. No timeout is resent, so no + timeout reaches here. Those two cases carry different attributes, and exactly one of + the two is on every event: a status, or `error.type` naming the transport. Neither is + ever absent and neither is ever a placeholder, because a dashboard grouping retries by + cause has to be able to tell a throttled API from an unreachable one. + The log record is the `WARN` the Rust SDK logs; a span event carries no severity, so on the span its presence is the signal. """ delay_ms = delay // _MILLISECOND + cause: dict[str, Attribute] = ( + {"error.type": TransportError.kind} + if status is None + else {"http.response.status_code": status} + ) attributes: dict[str, Attribute] = { - "http.response.status_code": status, + **cause, "guideme.retry.attempt": attempt, "guideme.retry.delay_ms": delay_ms, } + reached = "could not reach TypeSafe" if status is None else f"{status} from TypeSafe" if events != "log": span.add_event(RETRY_EVENT, attributes) logs = LOGS if events != "span" and logs is not None: - logs.retry(f"{status} from TypeSafe, retrying in {delay_ms} ms", attributes) + logs.retry(f"{reached}, retrying in {delay_ms} ms", attributes) diff --git a/tests/conftest.py b/tests/conftest.py index 06ca0fc..61eb76f 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -10,6 +10,7 @@ from pathlib import Path from typing import TYPE_CHECKING, Literal, cast, final +import httpx import pytest from hypothesis import HealthCheck, settings from jsonschema import Draft202012Validator @@ -24,7 +25,7 @@ from pytest_httpserver import HTTPServer from pytest_httpserver.httpserver import RequestHandler -from guideme import ApiKey, AsyncGuide, Guide, GuideBuilder, telemetry +from guideme import ApiKey, AsyncGuide, ConfigError, Guide, GuideBuilder, Receipt, telemetry from guideme._json import Json from guideme.api import Answer as WireAnswer from guideme.api import answer_from_wire @@ -35,6 +36,7 @@ ChoiceOutcome, NoulOutcome, Outcome, + Policy, ScoreOutcome, ) from guideme.question import choose_among, noul, score_levels @@ -226,6 +228,30 @@ def noul_reply(probability: float) -> str: return reply({"q0": {"type": "noul", "noul": probability}}) +MODELS_BODY: Json = { + "models": [ + { + "name": MODEL, + "description": "The current stable Jev.", + "release_date": "2026-02-11", + }, + { + "name": "jev-1.12.0", + "description": "The Jev before it.", + "release_date": "2025-11-04", + }, + ] +} +"""A `GET /v1/models` body, as the docs describe one.""" + + +def answering_offline(request: httpx.Request) -> httpx.Response: + """The whole server one `httpx.MockTransport` test needs: the bearer, then a noul.""" + assert request.headers["authorization"] == f"Bearer {TEST_KEY}" + assert request.url.path == EVALUATE + return httpx.Response(200, text=noul_reply(0.95), headers={"content-type": JSON}) + + def expect_post(httpserver: HTTPServer) -> RequestHandler: """The handler every `POST /v1/systemone` of one test goes to.""" return httpserver.expect_request(EVALUATE, method="POST") @@ -240,6 +266,13 @@ def expect_post(httpserver: HTTPServer) -> RequestHandler: type Configure = Callable[[GuideBuilder], GuideBuilder] """A test's extra builder settings, applied after the ones every test shares.""" +type Handler = Callable[[httpx.Request], httpx.Response] +"""What an injected `httpx.MockTransport` answers one request with. + +A handler may raise instead, which is how a test reaches the connection failures no +local server can produce on demand. +""" + def kind_of(param: object) -> Kind: """Narrow a fixture parameter to a kind. Anything else is a failure, never a default.""" @@ -275,6 +308,34 @@ def client_vars(guide: Guide | AsyncGuide) -> str: return repr(vars(inner)) +BAD_PATCH = Policy(yes_above=2.0) +"""A policy patch that cannot settle: a threshold outside `0..=1`. + +`with_policy` has to refuse it *before* taking a hold on the pool. Settling after the hold +was taken leaked one per refusal, and nothing downstream could tell: the guide that would +have held it was never built, so no close was ever coming for it. +""" + + +def refuse_patches(guide: Guide | AsyncGuide, times: int) -> None: + """Offer `BAD_PATCH` to `with_policy` `times` times, requiring each to be refused.""" + for _ in range(times): + with pytest.raises(ConfigError): + _ = guide.with_policy(BAD_PATCH) + + +def pool_holders(guide: Guide | AsyncGuide) -> int: + """Holds left on a guide's connection pool. + + Private state, read on purpose: "the pool was closed exactly once" is a statement + about the count, and a guide that leaks one looks identical from the outside to a + guide that released it. + """ + # pylint: disable=protected-access # the lifecycle proof reads what is private on purpose + client = guide._client # noqa: SLF001 # pyright: ignore[reportPrivateUsage] -- lifecycle proof + return client._pool.holders # noqa: SLF001 # pyright: ignore[reportPrivateUsage] -- same + + def configured(base_url: str, configure: Configure | None = None) -> GuideBuilder: """The builder every local-server test starts from: the test key, that origin, 10 ms backoff.""" builder = ( @@ -301,6 +362,18 @@ def async_entry(guide: AsyncGuide) -> Callable[[object, Json], Awaitable[object] return cast("Callable[[object, Json], Awaitable[object]]", guide.ask) +def _receipt_entry(guide: Guide) -> Callable[[object, Json], Receipt[object]]: + """`Guide.ask_with_receipt`, widened for the same reason as `_entry`.""" + return cast("Callable[[object, Json], Receipt[object]]", guide.ask_with_receipt) + + +def _async_receipt_entry( + guide: AsyncGuide, +) -> Callable[[object, Json], Awaitable[Receipt[object]]]: + """`AsyncGuide.ask_with_receipt`, widened for the same reason as `_entry`.""" + return cast("Callable[[object, Json], Awaitable[Receipt[object]]]", guide.ask_with_receipt) + + @final @dataclass(frozen=True, slots=True) class Runner: @@ -333,18 +406,95 @@ async def _ask_async(self, shape: object, state: Json, configure: Configure | No finally: await guide.close() - def models(self) -> tuple[ModelInfo, ...]: + def offline(self, shape: object, state: Json, handler: Handler) -> object: + """Ask through a caller-supplied transport: no server, no socket, no port. + + Which setter carries it is the one thing the two kinds cannot share, because a + builder refuses to build a guide of one kind from the other kind's transport. + """ + transport = httpx.MockTransport(handler) + if self.kind == SYNC: + with self._builder(lambda builder: builder.transport(transport)).build() as guide: + return _entry(guide)(shape, state) + return asyncio.run(self._offline_async(shape, state, transport)) + + async def _offline_async( + self, shape: object, state: Json, transport: httpx.MockTransport + ) -> object: + def settle(builder: GuideBuilder) -> GuideBuilder: + return builder.async_transport(transport) + + async with self._builder(settle).build_async() as guide: + return await async_entry(guide)(shape, state) + + def receipt(self, shape: object, state: Json) -> Receipt[object]: + """Ask `shape` about `state` and return the whole receipt, not only the answer.""" + if self.kind == SYNC: + with self._builder(None).build() as guide: + return _receipt_entry(guide)(shape, state) + return asyncio.run(self._receipt_async(shape, state)) + + async def _receipt_async(self, shape: object, state: Json) -> Receipt[object]: + async with self._builder(None).build_async() as guide: + return await _async_receipt_entry(guide)(shape, state) + + def paired( + self, shape: object, state: Json, closes: int, refused: int + ) -> tuple[list[object], int]: + """Three asks across a guide and one derived from it, closing the derived one. + + Both are entered as context managers, so the derived guide is closed `closes` + times in all: `closes - 1` by hand inside its block, and once more by leaving it. + The parent only ever releases at the very end, so all three asks must answer, and + that is the whole invariant. One close of the derived guide must not take the + parent's hold, and neither must a second: a guide closed twice releases once. + + `refused` policy patches are offered to `with_policy` first and must each be + refused; a refusal that took a hold on the way to raising shows up in the count + at the end, because no guide exists to release it. + + Returns the three answers and the holds left on the pool once both guides have + left their blocks, which must be zero. Without that count a guide that leaked a + hold would pass: the asks all answer either way, and the difference between + releasing once and never releasing at all is only visible here. + """ + if self.kind == SYNC: + with self._builder(None).build() as guide: + refuse_patches(guide, refused) + with guide.with_policy(Policy()) as derived: + first = _entry(derived)(shape, state) + for _ in range(closes - 1): + derived.close() + second = _entry(guide)(shape, state) + answers = [first, second, _entry(guide)(shape, state)] + return (answers, pool_holders(guide)) + return asyncio.run(self._paired_async(shape, state, closes, refused)) + + async def _paired_async( + self, shape: object, state: Json, closes: int, refused: int + ) -> tuple[list[object], int]: + async with self._builder(None).build_async() as guide: + refuse_patches(guide, refused) + async with guide.with_policy(Policy()) as derived: + first = await async_entry(derived)(shape, state) + for _ in range(closes - 1): + await derived.close() + second = await async_entry(guide)(shape, state) + answers = [first, second, await async_entry(guide)(shape, state)] + return (answers, pool_holders(guide)) + + def models(self, configure: Configure | None = None) -> tuple[ModelInfo, ...]: """`GET /v1/models` through this kind's executor.""" if self.kind == SYNC: - guide = self._builder(None).build() + guide = self._builder(configure).build() try: return guide.models() finally: guide.close() - return asyncio.run(self._models_async()) + return asyncio.run(self._models_async(configure)) - async def _models_async(self) -> tuple[ModelInfo, ...]: - guide = self._builder(None).build_async() + async def _models_async(self, configure: Configure | None) -> tuple[ModelInfo, ...]: + guide = self._builder(configure).build_async() try: return await guide.models() finally: diff --git a/tests/test_retries.py b/tests/test_retries.py new file mode 100644 index 0000000..49d645a --- /dev/null +++ b/tests/test_retries.py @@ -0,0 +1,433 @@ +"""What happens when a request fails, is resent, or goes through a caller's transport. + +`test_wire.py` holds the checks that say what guideme puts on the wire and reads back. +This module holds the ones about a request that does not simply succeed: the status-to-error +table, the retry policy on both endpoints, what is and is not resent before a response +arrives, an injected transport, and the connection pool two guides share. They are one +subject because they are one code path — `Client._fetch` and the pure `step` beneath it — +and every one of them needs a server that misbehaves on purpose. +""" + +import asyncio +import json +import time +from collections.abc import Callable +from dataclasses import dataclass +from datetime import timedelta +from typing import final + +import httpx +import pytest +from pytest_httpserver import HTTPServer + +from guideme import ( + AuthError, + GuideBuilder, + GuidemeError, + InvalidError, + OverloadedError, + RateLimitedError, + TransportError, + UnexpectedStatusError, +) +from guideme.api.client import EVALUATE, MODELS +from guideme.question import noul +from guideme.telemetry import ASK_SPAN, RETRY_EVENT + +from .conftest import ( + JSON, + MODELS_BODY, + TICKET, + Configure, + Handler, + Recorded, + Runner, + answering_offline, + as_list, + as_object, + async_entry, + attributes, + closed_port, + configured, + expect_post, + noul_reply, +) + +RETRIES = 1 +"""Retries each failing-status case allows, so an exhausted one is exactly two requests.""" + +DETAIL = '{"detail":"questions.q0.criteria: must not be empty"}' +"""A 422 body, which the error must carry verbatim and the span must not.""" + +type Serve = Callable[[HTTPServer], Configure | None] +"""How one failure case arranges the server, returning any builder change it needs.""" + +type Check = Callable[[GuidemeError], None] +"""What one failure case asserts about the raised error beyond its class and kind.""" + +type Drive = Callable[[Runner, Configure], object] +"""Which call one failure case makes. Both endpoints map a status the same way.""" + +OVER_THE_CAP = "3600" +"""A `retry-after` far past the 30 s guideme will wait. The call fails on the first attempt +carrying that duration, so the caller decides whether an hour is worth waiting.""" + +AN_HOUR = timedelta(seconds=3600) +"""`OVER_THE_CAP` as the error must carry it.""" + + +def _status( + status: int, + body: str = "", + retry_after: str | None = None, + path: str = EVALUATE, + method: str = "POST", +) -> Serve: + headers = None if retry_after is None else {"retry-after": retry_after} + + def serve(httpserver: HTTPServer) -> Configure | None: + httpserver.expect_request(path, method=method).respond_with_data( + body, status=status, headers=headers + ) + + return serve + + +def _asking(runner: Runner, configure: Configure) -> object: + return runner.ask(noul("Urgent?"), TICKET, configure) + + +def _listing(runner: Runner, configure: Configure) -> object: + return runner.models(configure) + + +def _refused(_httpserver: HTTPServer) -> Configure | None: + return lambda builder: builder.base_url(f"http://127.0.0.1:{closed_port()}") + + +def _nothing_more(_error: GuidemeError) -> None: + """The class and the kind are the whole contract for this status.""" + + +def _carries_the_body(error: GuidemeError) -> None: + assert isinstance(error, InvalidError) + assert error.detail == DETAIL + + +def _parsed_the_retry_after(error: GuidemeError) -> None: + assert isinstance(error, RateLimitedError) + assert error.retry_after == timedelta(seconds=0) + + +def _kept_the_retry_after(error: GuidemeError) -> None: + assert isinstance(error, OverloadedError) + assert error.retry_after == timedelta(seconds=0) + + +def _carried_an_hour(error: GuidemeError) -> None: + assert isinstance(error, RateLimitedError | OverloadedError) + assert error.retry_after == AN_HOUR + + +@final +@dataclass(frozen=True, slots=True) +class Failure: + """One failing call: how the server behaves, and everything the caller must see.""" + + serve: Serve + expected: type[GuidemeError] + kind: str + served: int + check: Check + drive: Drive = _asking + """Which call to make. A status means the same thing on both endpoints, and the cases + that say so are the ones where it would be cheapest for them to have drifted apart.""" + + asks: bool = True + """Whether a `guideme.ask` span is expected. `models()` opens none, and proving its + absence is as much a statement as reading the failed one's `error.type`.""" + + +FAILURES = [ + Failure(_status(401), AuthError, "auth", 1, _nothing_more), + Failure(_status(422, body=DETAIL), InvalidError, "invalid", 1, _carries_the_body), + Failure( + _status(429, retry_after="0"), + RateLimitedError, + "rate_limited", + RETRIES + 1, + _parsed_the_retry_after, + ), + Failure( + _status(500, body="upstream exploded"), + UnexpectedStatusError, + "unexpected_status", + 1, + _nothing_more, + ), + Failure( + _status(529, retry_after="0"), + OverloadedError, + "overloaded", + RETRIES + 1, + _kept_the_retry_after, + ), + Failure(_refused, TransportError, "transport", 0, _nothing_more), + Failure( + _status(429, retry_after=OVER_THE_CAP), + RateLimitedError, + "rate_limited", + 1, + _carried_an_hour, + ), + Failure( + _status(529, retry_after=OVER_THE_CAP), + OverloadedError, + "overloaded", + 1, + _carried_an_hour, + ), + Failure( + _status(429, retry_after=OVER_THE_CAP, path=MODELS, method="GET"), + RateLimitedError, + "rate_limited", + 1, + _carried_an_hour, + drive=_listing, + asks=False, + ), + Failure( + _status(529, retry_after=OVER_THE_CAP, path=MODELS, method="GET"), + OverloadedError, + "overloaded", + 1, + _carried_an_hour, + drive=_listing, + asks=False, + ), +] +"""Every status the contract defines, then the four that prove a `retry-after` past the cap +is not waited for: it fails on the first attempt and hands the caller the duration, on both +endpoints and for both statuses that carry the header.""" + +FAILURE_IDS = [ + "auth", + "invalid", + "rate_limited", + "unexpected_status", + "overloaded", + "transport", + "rate_limited_over_the_cap", + "overloaded_over_the_cap", + "rate_limited_over_the_cap_on_models", + "overloaded_over_the_cap_on_models", +] + + +@pytest.mark.parametrize("failure", FAILURES, ids=FAILURE_IDS) +def test_every_failure_raises_its_typed_error_and_marks_the_span( + httpserver: HTTPServer, runner: Runner, spans: Recorded, failure: Failure +) -> None: + extra = failure.serve(httpserver) + + def configure(builder: GuideBuilder) -> GuideBuilder: + settled = builder.max_retries(RETRIES) + return settled if extra is None else extra(settled) + + with pytest.raises(failure.expected) as raised: + _ = failure.drive(runner, configure) + assert raised.value.kind == failure.kind + failure.check(raised.value) + assert len(httpserver.log) == failure.served + if failure.asks: + assert attributes(spans.one(ASK_SPAN))["error.type"] == failure.kind + else: + assert not spans.named(ASK_SPAN) + + +BACKOFF = timedelta(milliseconds=300) +"""Long enough to measure that a retry really waited, short enough to pay for twice.""" + + +def _waiting(builder: GuideBuilder) -> GuideBuilder: + """Long enough a backoff that a resend cannot be mistaken for a fast first answer.""" + return builder.backoff(BACKOFF) + + +def _throttled_ask(httpserver: HTTPServer, runner: Runner) -> None: + httpserver.expect_oneshot_request(EVALUATE, method="POST").respond_with_data("", status=429) + expect_post(httpserver).respond_with_data(noul_reply(0.95), content_type=JSON) + assert runner.ask(noul("Urgent?"), TICKET, _waiting) is True + + +def _throttled_models(httpserver: HTTPServer, runner: Runner) -> None: + httpserver.expect_oneshot_request(MODELS, method="GET").respond_with_data("", status=429) + httpserver.expect_request(MODELS, method="GET").respond_with_data( + json.dumps(MODELS_BODY), content_type=JSON + ) + assert len(runner.models(_waiting)) == len(as_list(as_object(MODELS_BODY)["models"])) + + +THROTTLED = [_throttled_ask, _throttled_models] +"""Both endpoints, each throttled once. The API docs promise a retry on either.""" + + +@pytest.mark.parametrize("throttled", THROTTLED, ids=["evaluate", "models"]) +def test_a_429_is_retried_after_waiting_out_the_backoff( + httpserver: HTTPServer, runner: Runner, throttled: Callable[[HTTPServer, Runner], None] +) -> None: + started = time.monotonic() + throttled(httpserver, runner) + assert time.monotonic() - started >= BACKOFF.total_seconds() + assert len(httpserver.log) == 2 + + +CONCURRENT = 2 +"""Asks issued at once, which must wait out their retries together rather than in turn.""" + +ADVERTISED = 1.0 +"""Seconds the server puts in `retry-after`; whole seconds are all the header expresses.""" + +MARGIN = 0.2 +"""Slack below the sequential time, so the assertion fails on serialisation, not on load.""" + + +async def _two_asks(base_url: str) -> list[object]: + guide = configured(base_url).build_async() + try: + entry = async_entry(guide) + return list( + await asyncio.gather(entry(noul("Urgent?"), TICKET), entry(noul("Urgent?"), TICKET)) + ) + finally: + await guide.close() + + +def test_two_async_asks_wait_out_their_retries_at_the_same_time(httpserver: HTTPServer) -> None: + for _ in range(CONCURRENT): + httpserver.expect_oneshot_request(EVALUATE, method="POST").respond_with_data( + "", status=429, headers={"retry-after": "1"} + ) + expect_post(httpserver).respond_with_data(noul_reply(0.95), content_type=JSON) + + started = time.monotonic() + assert asyncio.run(_two_asks(httpserver.url_for(""))) == [True, True] + elapsed = time.monotonic() - started + + assert elapsed >= ADVERTISED + assert elapsed < CONCURRENT * ADVERTISED - MARGIN + assert len(httpserver.log) == 2 * CONCURRENT + + +PAIRINGS = [(1, 0), (2, 0), (1, 1)] +"""`(closes, refused)` for each way a hold can go wrong. + +`closes` is how many times the derived guide is closed: once is the ordinary case, twice is +the caller's mistake and must release once rather than spend the parent's hold too. +`refused` is how many policy patches are offered to `with_policy` and rejected first, which +must leave the count untouched — a refusal that had already taken a hold leaks it, because +the guide that would have released it was never built.""" + + +@pytest.mark.parametrize( + ("closes", "refused"), + PAIRINGS, + ids=["closed_once", "closed_twice", "after_a_refused_patch"], +) +def test_a_pool_is_held_once_per_guide_and_closes_when_the_last_one_releases( + httpserver: HTTPServer, runner: Runner, closes: int, refused: int +) -> None: + expect_post(httpserver).respond_with_data(noul_reply(0.95), content_type=JSON) + answers, held = runner.paired(noul("Urgent?"), TICKET, closes, refused) + assert answers == [True, True, True] + assert len(httpserver.log) == 3 + assert not held + + +def test_an_injected_transport_answers_an_ask_with_no_server(runner: Runner) -> None: + assert runner.offline(noul("Urgent?"), TICKET, answering_offline) is True + + +def _failing(failure: type[httpx.RequestError], detail: str) -> Handler: + """A handler that raises the way one case asks instead of answering.""" + + def fail(request: httpx.Request) -> httpx.Response: + raise failure(detail, request=request) + + return fail + + +@final +@dataclass(frozen=True, slots=True) +class BeforeAResponse: + """A failure that arrives with no response at all, and how guideme must treat it.""" + + fail: Handler + """What the transport raises. Each case raises it once and answers after that.""" + + attempts: int + """Calls the transport sees: one when the failure ends the ask, two when it is resent.""" + + answered: bool + """Whether an answer comes back, which only a resent failure can produce.""" + + +NEVER_REACHED = [ + BeforeAResponse(_failing(httpx.ConnectError, "connection refused"), 2, answered=True), + BeforeAResponse(_failing(httpx.ConnectTimeout, "connect timed out"), 1, answered=False), + BeforeAResponse(_failing(httpx.ReadTimeout, "read timed out"), 1, answered=False), + BeforeAResponse(_failing(httpx.PoolTimeout, "waited for a connection"), 1, answered=False), + BeforeAResponse(_failing(httpx.RemoteProtocolError, "server hung up"), 1, answered=False), +] +"""One case that is resent and four that are not. A refused connection never reached a +server, so nothing was judged; everything else here either did arrive, or is a timeout, and +no timeout is resent whatever phase it names. `client.resend_after` carries the reasoning.""" + + +@final +@dataclass(slots=True) +class _Flaky: + """A transport that fails its first call the way one case asks, then answers.""" + + fail: Handler + calls: int = 0 + + def __call__(self, request: httpx.Request) -> httpx.Response: + self.calls += 1 + if self.calls == 1: + return self.fail(request) + return answering_offline(request) + + +@pytest.mark.parametrize( + "case", + NEVER_REACHED, + ids=[ + "connect_error", + "connect_timeout", + "read_timeout", + "pool_timeout", + "remote_protocol_error", + ], +) +def test_only_a_failed_connection_is_resent_and_no_timeout_ever_is( + runner: Runner, spans: Recorded, case: BeforeAResponse +) -> None: + flaky = _Flaky(case.fail) + if case.answered: + assert runner.offline(noul("Urgent?"), TICKET, flaky) is True + else: + with pytest.raises(TransportError): + _ = runner.offline(noul("Urgent?"), TICKET, flaky) + assert flaky.calls == case.attempts + + resends = [ + event + for span in spans.named(f"POST {EVALUATE}") + for event in spans.events(span, RETRY_EVENT) + ] + assert len(resends) == case.attempts - 1 + for event in resends: + carried = attributes(event) + assert carried["error.type"] == "transport" + assert "http.response.status_code" not in carried diff --git a/tests/test_surface.py b/tests/test_surface.py index 966e471..f5e5d5d 100644 --- a/tests/test_surface.py +++ b/tests/test_surface.py @@ -3,6 +3,7 @@ from pathlib import Path import guideme +from guideme import api from .conftest import REPO_ROOT @@ -12,8 +13,12 @@ "AsyncGuide", "AuthError", "Choice", + "ChoiceQuestion", "Confidence", "ConfigError", + "DetailedChoice", + "DetailedNoul", + "DetailedScore", "Guide", "GuideBuilder", "GuidemeError", @@ -22,18 +27,23 @@ "Levels", "Model", "ModelInfo", + "NoulQuestion", "OverloadedError", "Policy", "Probability", "ProtocolError", + "Question", "Rank", "Ranked", "RateLimitedError", + "Receipt", + "ScoreQuestion", "Scored", "Thresholds", "TransportError", "UnexpectedStatusError", "UnsureError", + "Usage", "Verdict", "choose", "choose_among", @@ -66,6 +76,15 @@ def _imports(path: Path) -> set[str]: return found +def _bound_by_import(path: Path) -> set[str]: + """Every name a module got by importing it, under whatever alias it was bound to.""" + found: set[str] = set() + for node in ast.walk(ast.parse(path.read_text(encoding="utf-8"))): + if isinstance(node, ast.Import | ast.ImportFrom): + found.update(alias.asname or alias.name.split(".")[0] for alias in node.names) + return found + + def test_public_surface_is_exactly_the_documented_list() -> None: assert set(guideme.__all__) == DOCUMENTED assert list(guideme.__all__) == sorted(guideme.__all__) @@ -79,6 +98,15 @@ def test_public_surface_is_exactly_the_documented_list() -> None: code = "\n".join(re.findall(r"`{1,3}[^`]+`{1,3}", prose, re.DOTALL)) assert not [name for name in guideme.__all__ if name not in code] + # The second tier, `guideme.api`, is a module a caller imports by its own path, so + # what it re-exports is a promise too. It owns every name it offers: a list carrying + # `BaseModel` or `Mapping` would make pydantic's surface and the standard library's + # look like this package's, and a reader could not tell the mirror from what the + # mirror is built out of. + assert list(api.__all__) == sorted(api.__all__) + assert all(hasattr(api, name) for name in api.__all__) + assert not set(api.__all__) & _bound_by_import(PACKAGE / "api" / "__init__.py") + def test_httpx_and_pydantic_stay_behind_the_api_package() -> None: for path in sorted(PACKAGE.rglob("*.py")): diff --git a/tests/test_wire.py b/tests/test_wire.py index f6201c5..8d54225 100644 --- a/tests/test_wire.py +++ b/tests/test_wire.py @@ -1,11 +1,18 @@ -import asyncio +"""What guideme puts on the wire, and what it reads back. + +The schemas, the docs examples, the request a shape builds, the errors a malformed +response raises, the unsure ladder, and what a configuration mistake refuses. Its sibling +`test_retries.py` holds the other half: a request that fails, is resent, or goes through +a caller's transport. +""" + import json -import time from collections.abc import Callable from dataclasses import dataclass from datetime import timedelta from typing import cast, final +import httpx import pytest from hypothesis import given from hypothesis import strategies as st @@ -14,26 +21,19 @@ from guideme import ( ApiKey, - AuthError, Choice, Confidence, ConfigError, Guide, GuideBuilder, - GuidemeError, - InvalidError, Key, Levels, Model, - OverloadedError, Policy, Probability, ProtocolError, Rank, Ranked, - RateLimitedError, - TransportError, - UnexpectedStatusError, UnsureError, choose, fallback, @@ -41,7 +41,7 @@ score, ) from guideme.api import NoulAnswer, Request, Response, question_to_wire, request_to_wire -from guideme.api.client import EVALUATE, MODELS +from guideme.api.client import MODELS from guideme.ask import Plan, encode from guideme.guide import KEY_VAR, ModelInfo from guideme.question import Question, choose_among, noul, score_levels @@ -51,18 +51,16 @@ FIXTURES, JSON, MODEL, + MODELS_BODY, TEST_KEY, TICKET, WIRE_ANSWER, - Configure, Json, Recorded, Runner, + answering_offline, as_object, - async_entry, attributes, - closed_port, - configured, expect_post, load_json, narrow, @@ -159,150 +157,6 @@ def test_every_request_guideme_builds_matches_the_schema_and_is_keyed_q0_to_qn( assert Request.model_validate_json(request.model_dump_json(by_alias=True)) == request -RETRIES = 1 -"""Retries each failing-status case allows, so an exhausted one is exactly two requests.""" - -DETAIL = '{"detail":"questions.q0.criteria: must not be empty"}' -"""A 422 body, which the error must carry verbatim and the span must not.""" - -type Serve = Callable[[HTTPServer], Configure | None] -"""How one failure case arranges the server, returning any builder change it needs.""" - -type Check = Callable[[GuidemeError], None] -"""What one failure case asserts about the raised error beyond its class and kind.""" - - -def _status(status: int, body: str = "", retry_after: str | None = None) -> Serve: - headers = None if retry_after is None else {"retry-after": retry_after} - - def serve(httpserver: HTTPServer) -> Configure | None: - expect_post(httpserver).respond_with_data(body, status=status, headers=headers) - - return serve - - -def _refused(_httpserver: HTTPServer) -> Configure | None: - return lambda builder: builder.base_url(f"http://127.0.0.1:{closed_port()}") - - -def _nothing_more(_error: GuidemeError) -> None: - """The class and the kind are the whole contract for this status.""" - - -def _carries_the_body(error: GuidemeError) -> None: - assert isinstance(error, InvalidError) - assert error.detail == DETAIL - - -def _parsed_the_retry_after(error: GuidemeError) -> None: - assert isinstance(error, RateLimitedError) - assert error.retry_after == timedelta(seconds=0) - - -@final -@dataclass(frozen=True, slots=True) -class Failure: - """One failing call: how the server behaves, and everything the caller must see.""" - - serve: Serve - expected: type[GuidemeError] - kind: str - served: int - check: Check - - -FAILURES = [ - Failure(_status(401), AuthError, "auth", 1, _nothing_more), - Failure(_status(422, body=DETAIL), InvalidError, "invalid", 1, _carries_the_body), - Failure( - _status(429, retry_after="0"), - RateLimitedError, - "rate_limited", - RETRIES + 1, - _parsed_the_retry_after, - ), - Failure( - _status(500, body="upstream exploded"), - UnexpectedStatusError, - "unexpected_status", - 1, - _nothing_more, - ), - Failure(_status(529), OverloadedError, "overloaded", RETRIES + 1, _nothing_more), - Failure(_refused, TransportError, "transport", 0, _nothing_more), -] - - -@pytest.mark.parametrize("failure", FAILURES, ids=[case.kind for case in FAILURES]) -def test_every_failure_raises_its_typed_error_and_marks_the_ask_span( - httpserver: HTTPServer, runner: Runner, spans: Recorded, failure: Failure -) -> None: - extra = failure.serve(httpserver) - - def configure(builder: GuideBuilder) -> GuideBuilder: - settled = builder.max_retries(RETRIES) - return settled if extra is None else extra(settled) - - with pytest.raises(failure.expected) as raised: - _ = runner.ask(noul("Urgent?"), TICKET, configure) - assert raised.value.kind == failure.kind - failure.check(raised.value) - assert len(httpserver.log) == failure.served - assert attributes(spans.one(ASK_SPAN))["error.type"] == failure.kind - - -BACKOFF = timedelta(milliseconds=300) -"""Long enough to measure that a retry really waited, short enough to pay for twice.""" - - -def test_a_429_is_retried_after_waiting_out_the_backoff( - httpserver: HTTPServer, runner: Runner -) -> None: - httpserver.expect_oneshot_request(EVALUATE, method="POST").respond_with_data("", status=429) - expect_post(httpserver).respond_with_data(noul_reply(0.95), content_type=JSON) - started = time.monotonic() - assert runner.ask(noul("Urgent?"), TICKET, lambda builder: builder.backoff(BACKOFF)) is True - assert time.monotonic() - started >= BACKOFF.total_seconds() - assert len(httpserver.log) == 2 - - -CONCURRENT = 2 -"""Asks issued at once, which must wait out their retries together rather than in turn.""" - -ADVERTISED = 1.0 -"""Seconds the server puts in `retry-after`; whole seconds are all the header expresses.""" - -MARGIN = 0.2 -"""Slack below the sequential time, so the assertion fails on serialisation, not on load.""" - - -async def _two_asks(base_url: str) -> list[object]: - guide = configured(base_url).build_async() - try: - entry = async_entry(guide) - return list( - await asyncio.gather(entry(noul("Urgent?"), TICKET), entry(noul("Urgent?"), TICKET)) - ) - finally: - await guide.close() - - -def test_two_async_asks_wait_out_their_retries_at_the_same_time(httpserver: HTTPServer) -> None: - for _ in range(CONCURRENT): - httpserver.expect_oneshot_request(EVALUATE, method="POST").respond_with_data( - "", status=429, headers={"retry-after": "1"} - ) - expect_post(httpserver).respond_with_data(noul_reply(0.95), content_type=JSON) - - started = time.monotonic() - assert asyncio.run(_two_asks(httpserver.url_for(""))) == [True, True] - elapsed = time.monotonic() - started - - assert elapsed >= ADVERTISED - assert elapsed < CONCURRENT * ADVERTISED - MARGIN - assert len(httpserver.log) == 2 * CONCURRENT - - def _spent(input_tokens: int, output_tokens: int) -> str: """A reply whose single noul answer is fine and whose `usage` is what is under test.""" return json.dumps( @@ -314,6 +168,24 @@ def _spent(input_tokens: int, output_tokens: int) -> str: ) +SECOND = timedelta(seconds=1) +"""Any timeout at all: what these cases prove is the refusal, never the duration.""" + + +SPENT = (296, 20) +"""The token counts `_spent` reports, and what a receipt must hand back unchanged.""" + + +def test_a_receipt_carries_the_model_and_the_usage_the_body_reported( + httpserver: HTTPServer, runner: Runner +) -> None: + expect_post(httpserver).respond_with_data(_spent(*SPENT), content_type=JSON) + receipt = runner.receipt(noul("Urgent?"), TICKET) + assert receipt.answer is True + assert receipt.model == MODEL + assert (receipt.usage.input_tokens, receipt.usage.output_tokens) == SPENT + + VIOLATIONS = [noul_reply(1.5), _spent(-1, 20), _spent(296, -1)] """Bodies the schema refuses: a probability outside the unit, then a negative token count either way round. `spec/schema/response.json` sets `minimum: 0` on both counts, and an @@ -630,6 +502,38 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: _ = GuideBuilder().api_key(ApiKey("k")).events(unknown) +def _a_timeout_beside_a_transport(_monkeypatch: pytest.MonkeyPatch) -> None: + _ = GuideBuilder().transport(httpx.MockTransport(answering_offline)).timeout(SECOND) + + +def _a_transport_beside_a_timeout(_monkeypatch: pytest.MonkeyPatch) -> None: + _ = GuideBuilder().timeout(SECOND).transport(httpx.MockTransport(answering_offline)) + + +def _an_async_transport_beside_a_timeout(_monkeypatch: pytest.MonkeyPatch) -> None: + _ = GuideBuilder().timeout(SECOND).async_transport(httpx.MockTransport(answering_offline)) + + +def _both_transports(_monkeypatch: pytest.MonkeyPatch) -> None: + mock = httpx.MockTransport(answering_offline) + _ = GuideBuilder().transport(mock).async_transport(mock) + + +def _both_transports_the_other_way_round(_monkeypatch: pytest.MonkeyPatch) -> None: + mock = httpx.MockTransport(answering_offline) + _ = GuideBuilder().async_transport(mock).transport(mock) + + +def _an_async_transport_built_as_sync(_monkeypatch: pytest.MonkeyPatch) -> None: + builder = GuideBuilder().api_key(ApiKey("k")) + _ = builder.async_transport(httpx.MockTransport(answering_offline)).build() + + +def _a_sync_transport_built_as_async(_monkeypatch: pytest.MonkeyPatch) -> None: + builder = GuideBuilder().api_key(ApiKey("k")) + _ = builder.transport(httpx.MockTransport(answering_offline)).build_async() + + @pytest.mark.parametrize( "build", [ @@ -650,6 +554,13 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: _events_both_without_the_logs_api, _events_log_with_a_drifted_log_record, _events_given_an_unknown_mode, + _a_timeout_beside_a_transport, + _a_transport_beside_a_timeout, + _an_async_transport_beside_a_timeout, + _both_transports, + _both_transports_the_other_way_round, + _an_async_transport_built_as_sync, + _a_sync_transport_built_as_async, ], ids=[ "credentialed_base_url", @@ -669,6 +580,13 @@ def _events_given_an_unknown_mode(_monkeypatch: pytest.MonkeyPatch) -> None: "events_both_without_the_logs_api", "events_log_with_a_drifted_log_record", "events_given_an_unknown_mode", + "a_timeout_beside_a_transport", + "a_transport_beside_a_timeout", + "an_async_transport_beside_a_timeout", + "both_transports", + "both_transports_the_other_way_round", + "an_async_transport_built_as_sync", + "a_sync_transport_built_as_async", ], ) def test_a_configuration_mistake_is_refused_before_a_guide_exists( @@ -776,23 +694,6 @@ def test_the_request_carries_the_bearer_token_and_matches_the_schema( assert body["state"] == TICKET -MODELS_BODY: Json = { - "models": [ - { - "name": MODEL, - "description": "The current stable Jev.", - "release_date": "2026-02-11", - }, - { - "name": "jev-1.12.0", - "description": "The Jev before it.", - "release_date": "2025-11-04", - }, - ] -} -"""A `GET /v1/models` body, as the docs describe one.""" - - def test_the_model_list_comes_back_as_values_under_its_own_span( httpserver: HTTPServer, runner: Runner, spans: Recorded ) -> None: diff --git a/uv.lock b/uv.lock index 8562ffc..803e117 100644 --- a/uv.lock +++ b/uv.lock @@ -421,7 +421,7 @@ wheels = [ [[package]] name = "guideme" -version = "0.1.1" +version = "0.2.0" source = { editable = "." } dependencies = [ { name = "httpx" },