diff --git a/README.md b/README.md index b89ffcb..e9ae05a 100644 --- a/README.md +++ b/README.md @@ -3,11 +3,12 @@ modelito Modelito is a compact, dependency-light Python library that provides provider- agnostic abstractions and connectors for large language models (LLMs). It -offers lightweight shims for OpenAI, Claude, Gemini, oMLX, local Ollama deployments, -and local OpenAI-compatible servers (llama.cpp, vLLM, LM Studio), plus -utilities for token counting, timeout estimation, and small helpers to manage -Ollama servers when needed. The library is designed for easy integration into -applications and CI pipelines. +offers lightweight shims for OpenAI, Claude, Gemini, local Ollama deployments, +BaseRT, vllm-mlx, oMLX, and generic OpenAI-compatible servers (llama.cpp, +vLLM, LM Studio, and similar runtimes), plus utilities for token counting, +timeout estimation, provider readiness, and small helpers to manage Ollama +servers when needed. The library is designed for easy integration into +applications and CI pipelines. Quick start ----------- @@ -78,6 +79,7 @@ pip install dist/*.whl See the `docs/` folder for more details: - [ARCHITECTURE.md](docs/ARCHITECTURE.md) — Core design, Provider Protocol, and SDK hierarchy - [USAGE.md](docs/USAGE.md) — Usage guide and examples +- [LOCAL-RUNTIMES.md](docs/LOCAL-RUNTIMES.md) — Local runtime profiles, capabilities, and benchmarking - [local-openai-compatible.md](docs/local-openai-compatible.md) — Using local OpenAI-compatible servers - [INSTALL.md](docs/INSTALL.md), [API.md](docs/API.md) — Installation and API reference - [RELEASE.md](docs/RELEASE.md) — Release checklist and publication steps @@ -121,6 +123,10 @@ Provided shims and utilities: - `ClaudeProvider` — will use the official Anthropic SDK when installed, falling back to deterministic behavior otherwise. - `GeminiProvider`, `GrokProvider` — lightweight shims. +- `BaseRTProvider` — thin BaseRT preset built on + `OpenAICompatibleHTTPProvider`. +- `VLLMMLXProvider` — thin vllm-mlx preset built on + `OpenAICompatibleHTTPProvider`. - `OMLXProvider` — thin oMLX preset built on `OpenAICompatibleHTTPProvider`. - `OllamaProvider` — HTTP-aware provider that can call a local Ollama HTTP API through stdlib helpers and can fall back to the local Ollama CLI or @@ -132,11 +138,10 @@ The client layer recognises the same provider stack through `ChatProvider`, `MessageInput`, and structured response helpers such as `Client.chat()` and `Client.chat_json()`. -`OpenAICompatibleHTTPProvider`, `OMLXProvider`, `OpenAIProvider`, and -`OllamaProvider` also -expose `raw_complete()` and `raw_stream()` for OpenAI-compatible passthrough. -`Client.chat_parsed()` remains the structured JSON convenience path for Python -applications. +`OpenAICompatibleHTTPProvider`, `BaseRTProvider`, `VLLMMLXProvider`, +`OMLXProvider`, `OpenAIProvider`, and `OllamaProvider` expose `raw_complete()` +and `raw_stream()` for OpenAI-compatible passthrough. `Client.chat_parsed()` +remains the structured JSON convenience path for Python applications. For quick diagnostics, use the provider readiness API or CLI: @@ -154,6 +159,51 @@ dict conversion. python -m modelito doctor --provider omlx --model omlx ``` +Local runtime profiles +---------------------- + +For applications that require local execution rather than Modelito's general +provider auto-selection, use `local_client()` or `select_local_runtime()`: + +```py +from modelito import local_client + +client = local_client( + profile="mac-performance", + models={ + "basert": "my-base-model", + "vllm-mlx": "my-vllm-model", + "omlx": "my-omlx-model", + "ollama": "my-ollama-tag", + }, +) +``` + +The profiles are: + +- `portable`: Ollama as the common cross-platform path; +- `mac-performance`: on Apple Silicon, try BaseRT, vllm-mlx, oMLX, then + Ollama; +- `auto`: use `mac-performance` on Apple Silicon and `portable` elsewhere. + +That order is a deployment starting point, not a universal speed ranking. +`prefer=` can override it after benchmarking the target workload. Local +selection is strict: it does not silently fall back to a hosted provider or to +the deterministic offline shim. + +`local_runtime_capabilities()` exposes conservative metadata for streaming, +prefix caching, cancellation, structured output, tool calls, and model +discovery. Model- or configuration-dependent features are marked +`conditional`; uncertain claims are marked `unknown`. + +For latency-sensitive local work, `modelito-benchmark-local` measures first +request TTFT, warm-prefix TTFT, first useful streamed phrase, estimated decode +rate, context-growth latency, cancellation/stream-close behavior, and optional +process RSS against an already-running OpenAI-compatible server. Raw MLX-LM is +supported as a benchmark reference without being added to automatic runtime +selection. See [docs/LOCAL-RUNTIMES.md](docs/LOCAL-RUNTIMES.md) for the caveats +and current upstream-source audit. + Server mode for non-Python clients: ```sh @@ -207,9 +257,9 @@ Example `~/.pi/agent/models.json` provider entry: ``` Tool-calling workflows require raw passthrough support. Modelito currently -implements that on `OpenAICompatibleHTTPProvider`, `OMLXProvider`, -`OpenAIProvider`, and `OllamaProvider` via Ollama's -`/v1/chat/completions` endpoint. +implements that on `OpenAICompatibleHTTPProvider`, `BaseRTProvider`, +`VLLMMLXProvider`, `OMLXProvider`, `OpenAIProvider`, and `OllamaProvider` via +their OpenAI-compatible chat-completions paths. The package also exposes a small Ollama administration layer for local model operations, including install backend detection, remote catalog metadata, @@ -269,8 +319,9 @@ compatible with existing duck-typed providers — it requires only: - `summarize(messages, settings=None)` -> `str` All built-in providers shipped with the package (`OpenAIProvider`, -`ClaudeProvider`, `GeminiProvider`, `OMLXProvider`, `OllamaProvider`, `GrokProvider`) satisfy -the `Provider` protocol structurally. The `Provider` Protocol is decorated with +`ClaudeProvider`, `GeminiProvider`, `BaseRTProvider`, `VLLMMLXProvider`, +`OMLXProvider`, `OllamaProvider`, `GrokProvider`) satisfy the `Provider` +protocol structurally. The `Provider` Protocol is decorated with `@runtime_checkable`, so you can use `isinstance()` checks at runtime when you need to enforce the contract in application code. @@ -399,4 +450,4 @@ These modules are not currently presented as stable top-level package exports in this README. Prefer the documented `Client`, provider adapters, connector, and server entrypoints for application integrations. -See the `tests/` directory for comprehensive coverage and usage examples. +See the `tests/` directory for comprehensive coverage and usage examples. \ No newline at end of file diff --git a/STATUS.md b/STATUS.md index 50b37ef..c486255 100644 --- a/STATUS.md +++ b/STATUS.md @@ -1,225 +1,279 @@ # modelito – Project Status -Last updated: 2026-06-01 12:00 +Last updated: 2026-08-09 ## Project purpose -modelito is a compact, dependency-light Python library that provides provider-agnostic abstractions and connectors for large language model usage. It supports hosted and local providers, lightweight shims, token/timeout helpers, embeddings, streaming normalisation, and Ollama administration utilities. - -## Current implementation state - -Current package metadata version is `1.4.6` in `pyproject.toml`. - -Release `v1.4.5` was tagged and published to GitHub on 2026-05-19, and successfully published to PyPI after configuring the trusted publisher. Release `v1.4.6` is a patch bump adding strict raw_stream non-dict event handling, fallback model propagation, and comprehensive field-preservation tests for raw passthrough. - -The package provides: - -- core provider protocols and dataclasses -- adapters/shims for OpenAI (with local OpenAI-compatible server support), Anthropic/Claude, Gemini, Grok, oMLX, and Ollama -- synchronous, asynchronous, streaming, and embedding provider surfaces -- `OllamaConnector` and provider registry helpers -- Ollama install detection, local service helpers, remote catalog metadata, lifecycle/download tracking, and model readiness helpers -- optional SDK-backed behaviour with deterministic fallback support for offline tests/examples -- provider readiness diagnostics via `check_provider_ready()` and `python -m modelito doctor` -- optional OpenAI-compatible server mode via `modelito-serve` for non-Python clients such as Pi, with bind settings kept separate from provider backend configuration -- raw OpenAI chat-completions passthrough via `RawChatProvider`, `OpenAICompatibleHTTPProvider`, `OMLXProvider`, `OpenAIProvider`, and `OllamaProvider` with strict non-dict event handling and field preservation for tool calling -- shared readiness probes in `modelito/probes.py`, with `modelito-doctor` as a console script and `flatten_message_inputs` exported from the package root for convenience -- comprehensive docs under `docs/` including architecture, usage, API reference, install guide, and local server integration -- concise release checklist documentation in `docs/RELEASE.md` -- pytest, ruff, mypy, build, twine, CI, and publishing workflows -- composable `RecordingProvider` / `ReplayProvider` wrappers in `modelito/recording.py` that persist request/response pairs to a JSONL cassette for offline replay; stdlib-only, zero extra dependencies - -Release `v1.4.3` was tagged in git and published to PyPI after the local OpenAI-compatible server support landed in `v1.4.2`. - -## Active focus - -Phase 4 server-contract hardening and provider cleanup pass complete: - -1. `modelito-serve` now keeps bind `host`/`port` separate from provider backend configuration, and its raw streaming path forwards provider generators lazily instead of buffering them. -2. `OpenAIProvider.raw_complete()` and `raw_stream()` are strict in strict mode, and `OpenAIProvider.chat()` now returns `Response` metadata for server use. -3. `OllamaProvider` accepts dict-style `MessageInput` values consistently through the shared flattening helper. -4. Server endpoints now return OpenAI-style error payloads (`{"error": {...}}`) with mapped HTTP status codes for provider, timeout, connection, not-found, bad-response, and bad-request/internal failures. -5. Chat payload validation now rejects missing/malformed `messages` fields while preserving string-message backward compatibility. -6. Tool-gated fallback checks now trigger on both `tools` and `tool_choice` when raw passthrough is unavailable in strict mode. -7. Embeddings input validation rejects unsupported shapes, and embeddings output validation enforces vector count/type correctness while normalizing numeric values to floats. -8. Request parsing now rejects malformed JSON and non-object request bodies for `/v1/chat/completions` and `/v1/embeddings`, returning OpenAI-style 400 payloads. -9. `ModelitoBadResponseError` now maps to 502 (bad upstream/provider response) and is classified as `modelito_bad_response`. -10. Raw-provider non-dict completion payloads are treated as upstream bad responses (`ModelitoBadResponseError`) and return 502 errors. -11. Regression tests now cover lazy raw streaming, runtime config separation (server bind host/port are not forwarded to provider constructors), payload validators, error-shape/status helpers, malformed JSON handling, OpenAI provider raw/chat behavior, and Ollama dict/string message normalization. -12. The legacy dict-style docs checker now ignores tests because dict `MessageInput` compatibility is intentionally tested there, and the README example now uses `Message(...)`. -13. README wording was tightened for Ollama extra semantics and `--profile`/`--profile-path` path handling, and Ollama raw passthrough is documented as supported via `/v1/chat/completions`. -14. Publish workflow now includes a tag/version gate plus explicit trusted-publishing prerequisites and PyPI environment URL metadata. -15. `ChatProvider` — `@runtime_checkable` Protocol in `modelito/provider.py` formalising the `chat()` interface returning `Response`; exported from package root. -16. `MessageInput` type alias (`Union[Message, str, Mapping[str, Any]]`) added to `provider.py` and exported; `Client` method signatures broadened from `Iterable[Message]` to `Iterable[MessageInput]`; `SyncProvider.summarize()` and `AsyncProvider.acomplete()` signatures similarly broadened. -17. Provider readiness diagnostics added through `check_provider_ready()` / `ProviderStatus` and the `python -m modelito doctor` CLI. -18. OpenAI-compatible server support is provided through `modelito.serve`, `modelito-serve`, `RawChatProvider`, raw passthrough on the OpenAI-compatible HTTP base, and the hosted OpenAI provider. -19. Latest review-feedback fixes made fallback streaming lazy, scoped warning headers to tool fallbacks, shared readiness probes between client and doctor, and cleaned up package exports. - -- `modelito-serve` exposes `/v1/models`, `/v1/chat/completions`, and `/v1/embeddings` for Pi and other OpenAI-compatible consumers. -- Pi/tool-calling should use raw-capable providers (`OpenAICompatibleHTTPProvider`, `OMLXProvider`, hosted `OpenAIProvider`, or `OllamaProvider`) when full OpenAI-compatible payload fidelity is required. -- Validation status is recorded in the current session snapshot under Recent changes. +Modelito is a compact, dependency-light Python library for provider-agnostic +LLM access. It supports hosted and local providers, OpenAI-compatible serving, +streaming and structured responses, embeddings, readiness probes, +token/timeout helpers, Ollama administration, and deterministic/offline-friendly +test fallbacks. + +## Current state + +- Package metadata version: `1.4.6`. +- Python: 3.10–3.12. +- Licence: MIT. +- Hosted providers include OpenAI, Anthropic/Claude, Gemini, and Grok. +- Local providers include Ollama, BaseRT, vllm-mlx, oMLX, and generic + OpenAI-compatible HTTP servers. +- `modelito-serve` exposes OpenAI-compatible models, chat-completions, and + embeddings endpoints. +- `modelito-doctor` and `check_provider_ready()` provide read-only provider + diagnostics. +- `modelito-benchmark-local` measures conversational latency for already-running + OpenAI-compatible local runtimes. +- Raw OpenAI chat-completions passthrough is supported by raw-capable providers + for tool-calling and metadata-preserving integrations. +- Ollama helpers cover installation detection, local service control, model + lifecycle/readiness, download, preload, and diagnostics while keeping + mutating operations explicit. +- `RecordingProvider` / `ReplayProvider` provide JSONL cassette recording and + deterministic replay. + +## Local runtime policy + +The explicit local-only surface is separate from the established +`Client(provider="auto")` contract: + +- `portable`: Ollama as the common macOS/Linux/Windows path; +- `mac-performance`: on Apple Silicon, BaseRT → vllm-mlx → oMLX → Ollama; +- `auto`: `mac-performance` on Apple Silicon and `portable` elsewhere; +- `MODELITO_LOCAL_PROFILE` environment configuration; +- `select_local_runtime()` for read-only selection and diagnostics; +- `local_client()` for a strict local-only client with no hosted or + deterministic fallback; +- provider-specific model, endpoint, and API-key mappings; +- explicit `prefer=` ordering so benchmark results can override defaults; +- `local_runtime_capabilities()` for conservative runtime-family capability + metadata. + +The candidate order is a deployment starting point, not a universal performance +claim. Modelito does not encode a claim that any one local runtime is fastest +for every model or workload. + +## Supported local runtime roles + +### Ollama + +The portable path. Current Apple-Silicon releases can use MLX and cache/snapshot +optimisations, while later 2026 releases also use llama.cpp paths to broaden +model and hardware support. + +### BaseRT + +An Apple-Silicon native-Metal runtime exposed through an OpenAI-compatible +server. Modelito integrates only through the local HTTP API. + +### vllm-mlx + +An Apple-Silicon MLX server with OpenAI-compatible serving, caching/batching, +structured-output and cancellation capabilities in current upstream releases. + +### oMLX + +An MLX-native server oriented towards persistent conversational/agent +workloads, including continuous batching and tiered prefix/KV caching. + +### Generic OpenAI-compatible + +The existing `OpenAICompatibleHTTPProvider` remains the escape hatch for +llama.cpp, LM Studio, vLLM, MLX-LM's HTTP server, and other compatible +endpoints. Arbitrary endpoints are not auto-selected. + +### MLX-LM reference + +Raw MLX-LM remains a benchmark/reference path rather than another automatic +runtime provider. Its prompt caching is useful for repeated conversational +contexts. The benchmark CLI recognises `mlx-lm` as a label for comparisons. + +## Conversational benchmark + +`modelito-benchmark-local` records: + +- first-request TTFT; +- first phrase-like streamed latency; +- estimated decode tokens/s; +- warm-prefix TTFT over repeated requests; +- context-growth TTFT; +- client stream-close latency and a post-cancellation probe; +- optional sampled server process RSS. + +The benchmark embeds its own caveats. First-request TTFT is only a cold-model +measurement when the server/model was actually cold. Token rate uses Modelito's +token-count helper, client stream-close time does not prove server-side +cancellation acknowledgement, and process RSS can under-report Metal/unified +memory on macOS. + +This benchmark is intended to compare equivalent workloads on the target +machine. It does not replace runtime-specific instrumentation such as +vllm-mlx's `bench-serve` command. ## Architecture overview -modelito exposes a small common provider protocol, concrete adapters, connectors, and helper modules. Optional provider SDKs are used when installed; otherwise providers fall back to deterministic behaviour. Detailed architecture policy lives in `docs/ARCHITECTURE.md`. The SVGs below are intentionally compact and current-state oriented. Ollama service and API helpers provide local-model administration while remaining explicitly gated for integration tests. +The primary application surface is `modelito.Client`, backed by registered +provider adapters. Providers implement a small common interface and may expose +richer raw/streaming capabilities when available. The local-runtime selector is +policy layered above those providers rather than another provider protocol. ### Architecture diagram - - modelito architecture - Core protocols connect clients, Pi and OpenAI-compatible callers, modelito-serve, provider adapters, shared probes, model metadata helpers, and package documentation/tests. - - Pi / OpenAIcompatible clients - modelito-serve/v1/models/v1/chat/completions - Clientchat / json / parsedpackage-root exports - Raw-capable pathRawChatProvider,OpenAICompat, OMLX, OpenAI - Fallback pathsummarize / streamResponse / dict output - Hosted adaptersOpenAI, Claude,Gemini, Grok - Shared probesmodelito/probes.pymodelito-doctor - Local Ollamaprovider and adminraw passthrough via /v1 - Metadata helpersModelMetadata,get_model_info, infer - docs and testsCI, build, releasecurrent-state snapshot - + + modelito current architecture + Applications use Client or modelito-serve. Explicit local deployment may pass through the local runtime selector, which probes BaseRT, vllm-mlx, oMLX, or Ollama before constructing a strict provider. Hosted and generic providers remain available through the normal registry. + + ApplicationsClient / HTTP callers + modelito-serveOpenAI-compatible API + local runtime policyselect / probe / prefer + Provider registryhosted + generic + Shared readiness probesactual model availability + Hosted providersOpenAI / ClaudeGemini / Grok + Strict local providersBaseRT / vllm-mlxoMLX / OllamaOpenAI-compatible + Common surfaceschat / stream / rawstructured / embed + -### Flow chart +### Local request flow - - modelito provider request flow - Pi or OpenAI-compatible clients call modelito-serve or the Client API, modelito chooses a raw-capable or fallback provider path, shared probes assist detection, and responses are returned as parsed data or text. + + strict local runtime request flow + Local intent resolves a profile, orders candidates, probes model availability, selects a concrete model, constructs a strict provider, and then uses the normal Client API. If no candidate is usable, selection raises explicitly. - Pi / OpenAIclients - modelito-serve/v1/models - /chat/completions/embeddings - Raw-capableproviders - RawChatProviderOpenAICompat - OMLX / OpenAIraw passthrough - Responseor parsed dict - Fallback pathsummarize / stream - Non-rawResponse / text - Shared probesmodelito-doctor - Package rootexports - + local intentprofile + model + orderedcandidates + readiness +model discovery + resolved localprovider/model + strict providerno shim fallback + normal ClientAPI + no usable candidate → explicit selection error + -## Setup and run instructions - -Install latest release: - -```bash -pip install modelito -``` +## Setup and verification -Development setup: +Development installation: ```bash -pip install -e .[dev] -pip install -r dev-requirements.txt -pip install -e .[ollama,tokenization,openai,anthropic] +python -m pip install -e '.[dev]' ``` -Validation: +Typical checks: ```bash -pytest -q +python scripts/check_no_legacy_dicts.py ruff check . +black --check . mypy modelito --ignore-missing-imports +pytest -q --ignore=tests/integration tests python -m build -python -m twine check dist/* ``` -## Configuration and environment variables +Local runtime policy, benchmark usage, capability caveats, and upstream sources +are documented in `docs/LOCAL-RUNTIMES.md`. -- `RUN_OLLAMA_INTEGRATION=1`: enables Ollama integration tests. -- `ALLOW_OLLAMA_INSTALL=1`: permits integration tests to attempt Ollama installation. -- `ALLOW_OLLAMA_DOWNLOAD=1`: permits remote model downloads during integration tests. -- `ALLOW_OLLAMA_UPDATE=1`: permits update flows during integration tests. -- Provider SDK/API keys are optional and should be supplied through environment or external secret-management mechanisms. +## Recent changes -## Important files and directories +- Added explicit `portable`, `mac-performance`, and `auto` local-runtime + profiles without altering the established `Client(provider="auto")` contract. +- Added BaseRT and vllm-mlx OpenAI-compatible provider presets alongside oMLX + and Ollama. +- Added provider-specific model/endpoint/key mappings, readiness-based model + resolution, benchmark-overridable `prefer=` ordering, and conservative + capability metadata. +- Added `modelito-benchmark-local` for first-turn, warm-prefix, context-growth, + decode, cancellation-close, and approximate RSS measurements. +- Added a strict-aware Ollama surface so local-only clients propagate runtime + failures instead of returning deterministic fallback text; the package-root + export now uses the same class as the provider registry. +- Explicitly configured the historical Ruff lint contract after Ruff 0.16 + expanded its unconfigured default rule set. This prevents an unpinned tool + update from silently redefining repository lint policy. +- Restored repository-wide pending work and inline current-state diagrams in + this status snapshot. + +## Current decisions + +1. Do not encode a universal local-runtime performance ranking. +2. Keep Ollama as the portable path. +3. Keep a Mac-oriented path because Apple-Silicon-specific runtimes expose + materially different caching, serving, and execution strategies. +4. Preserve existing `Client(provider="auto")` behaviour. +5. A local-only client must fail explicitly when no requested local backend or + model is ready. +6. Provider-specific model identifiers and endpoints must be expressible + independently. +7. Local runtime defaults remain benchmark-overridable with `prefer=`. +8. Capability metadata is conservative: `conditional` and `unknown` are used + instead of guessing model- or version-specific support. +9. Speech/VAD/TTS/ASR orchestration does not belong in Modelito's LLM runtime + abstraction. +10. No release, tag, or version bump is part of this work. -- `modelito/`: package source. -- `modelito/recording.py`: `RecordingProvider`, `ReplayProvider`, `CassetteFormatError`, `ReplayMissError`. -- `tests/`: test suite. -- `docs/`: user and API documentation. -- `docs/assets/`: standalone SVG diagram assets reused by the README. -- `examples/`: usage examples. -- `pyproject.toml`: package metadata and build configuration. -- `.github/workflows/ci.yml`: lint/type/test/doc workflow. -- `.github/workflows/integration-ollama.yml`: dedicated self-hosted Ollama integration workflow. -- `.github/workflows/publish.yml`: PyPI publishing workflow. +## Pending tasks -## Recent changes +These repository-wide tasks remain open and are not superseded by the local +runtime work: -- **Raw passthrough follow-up hardening**: `OllamaProvider.raw_stream()` now treats valid JSON non-object SSE events as malformed responses in strict mode (`ModelitoBadResponseError`) instead of silently ignoring them. Added targeted tests for strict non-dict rejection, explicit `/v1/chat/completions` endpoint usage in raw streaming, raw stream input immutability, and preservation of `response_format` plus common generation fields (`temperature`, `top_p`, `max_tokens`, `stop`) for both `raw_complete()` and `raw_stream()`. Updated `OpenAICompatibleHTTPProvider` fallback chat payload to preserve the requested `model` value and aligned OMLX fallback expectation accordingly. -- **OllamaProvider raw passthrough (raw_complete/raw_stream)**: Implemented `raw_complete(payload)` and `raw_stream(payload)` methods on `OllamaProvider` to enable OpenAI-compatible passthrough via `/v1/chat/completions` endpoint. Both methods preserve all OpenAI request fields (tools, tool_choice, response_format, etc.) and work with `modelito-serve` for tool-calling workflows with local Ollama models. Comprehensive test suite added: 13 test functions in `tests/test_ollama_raw_provider.py` covering protocol conformance, default model injection, explicit model preservation, tool preservation, endpoint routing, response handling, error modes (strict/non-strict), and SSE streaming. Integration test added to `tests/test_serve.py` verifying tool-calling through modelito-serve. -- **recording.py audit**: Audited 4 type:ignore comments. Removed 3 unnecessary `type:ignore[attr-defined]` comments on imports of `Response`, `Message`, `flatten_message_inputs` (all properly exported from messages.py). Kept 1 necessary `type:ignore[return-value]` at return statement due to mypy type-narrowing limitation with tuple unpacking in conditional assignment. -- **Namespaced public helpers documentation (Option A)**: Added new "Namespaced public helpers" section to docs/API.md explaining the pattern and providing example code for `RecordingProvider` and `ReplayProvider` from `modelito.recording`. Documentation already used correct `from modelito.recording import ...` imports consistently. -- **LiteLLM documentation note**: Added "Future adapters: LiteLLM" section to docs/USAGE.md explaining LiteLLM as planned optional adapter (`modelito[litellm]`), noting core maintains `dependencies = []`, and positioning as future extra alongside other provider-specific integrations. -- **OpenAI-compatible raw passthrough documentation**: Added "OpenAI-compatible raw passthrough" section to docs/API.md with detailed signature documentation, availability list (OpenAIProvider, OMLXProvider, OpenAICompatibleHTTPProvider, OllamaProvider), tool preservation explanation, and Python example showing Ollama with function calling. Added "OpenAI-compatible raw passthrough (tool calling)" section to docs/USAGE.md with practical example showing tool definitions, payload construction, and response handling. -- **Code quality fixes**: Legacy-dict docs checker now allows raw OpenAI-compatible passthrough payload examples while continuing to reject legacy `summarize([{"role": ...}])` patterns in docs/examples. Fixed 9 linting issues in test file (removed unused imports json/patch/BytesIO/Message, removed unused result variables, removed late import). All 84 files reformatted by ruff for consistency. -- **Validation completed this session**: - - `python scripts/check_no_legacy_dicts.py` → clean - - `ruff check .` → clean - - `ruff format .` → formatted 84 files - - `mypy modelito --ignore-missing-imports` → clean (38 source files) - - `pytest -q` → 351 passed, 3 skipped - - `from modelito import OllamaProvider; from modelito.provider import RawChatProvider; print(isinstance(OllamaProvider(model='test'), RawChatProvider))` → True -- Dependencies unchanged: `dependencies = []` confirmed in `pyproject.toml` (core remains zero-dependency) - -- Added `modelito/recording.py`: composable `RecordingProvider` and `ReplayProvider` wrappers. `RecordingProvider` wraps any modelito provider, delegates calls to it, and appends request/response pairs as JSONL records to a cassette file. `ReplayProvider` reads the cassette and returns stored responses without touching the network. Both support `list_models()`, `summarize()`, and `chat()`; `stream()` and `embed()` raise `NotImplementedError` in v1. Message normalisation (str, dict, `Message`, iterables, generators) is handled consistently; generator exhaustion is avoided by materialising once. Replay is model-agnostic by default; `CassetteFormatError` and `ReplayMissError` are raised on malformed cassettes and cache misses respectively. 73 tests in `tests/test_recording_provider.py`; full suite: 329 passed, 3 skipped. -- Latest cleanup pass fixed fallback streaming laziness, provider warning header scope, shared probe reuse, and `flatten_message_inputs` / `modelito-doctor` export surface. -- Added a concise release checklist document and linked it from the README docs index. -- Install-helper Unix test now accepts apt-based Linux install commands as well as script install commands. -- Python 3.10 test compatibility fixed by using `tomli` as fallback for `tomllib` (available from Python 3.11+). -- Model metadata registry was made conservative and typed using a frozen dataclass, stale hardcoded entries were removed/downgraded, and modern model-family inference was added. -- Provider APIs remain the source of truth for model capabilities; static metadata is now explicitly treated as best-effort fallback only. -- Model metadata inference no longer treats every `o...` model name as OpenAI; only `gpt-*` and known OpenAI reasoning prefixes (`o1*`, `o3*`, `o4*`) infer OpenAI. -- Model metadata helpers are now exported from the package root (`ModelMetadata`, `get_model_info`, `get_model_metadata`, `infer_model_metadata`). -- Release checklist now includes explicit trusted publishing requirements, tag/publish commands, and clean-environment install checks. -- README now embeds standalone current-state SVG assets from `docs/assets/`, and the architecture/usage docs call out `modelito-serve`, shared probes, and raw-capable providers including Ollama. -- Current release line is `1.4.6` (patch bump: strict raw_stream non-dict handling, OpenAI-compatible fallback model propagation, comprehensive field-preservation tests). -- Current oMLX stack uses `OpenAICompatibleHTTPProvider` with strict-mode typed error handling. -- Current provider typing includes `ChatProvider`, `MessageInput`, and `OpenAIMessageDict` exports, with `Client` chat-related methods accepting broadened message input types; provider protocols are aligned so `SyncProvider`, `AsyncProvider`, `StreamingProvider`, and `ChatProvider` all accept `Iterable[MessageInput]`. -- `Client.chat_json()` now supports optional stronger schema validation via `strict_schema=True` using dataclass construction or Pydantic-style `model_validate`/`parse_obj` hooks, while preserving lightweight key-presence checks by default. -- Validation completed locally in this session: - - `python scripts/check_no_legacy_dicts.py` -> clean - - `ruff check .` -> clean - - `mypy modelito --ignore-missing-imports` -> clean (38 source files) - - `pytest -q --ignore=tests/integration tests` -> `350 passed, 1 skipped` - - `python -c "import modelito; print(modelito.__version__)"` -> `1.4.6` - - `python -m build` -> succeeded; `1.4.6` wheel and sdist validated - - `python -m twine check dist/*` -> all packages passed (1.4.3, 1.4.4, 1.4.5) -- Trusted publishing note: GitHub Actions workflow is configured correctly (`pyproject.toml` version matches tag `v1.4.6`). PyPI trusted publisher has been configured for repository `krahd/modelito` with environment `pypi`; re-run of the failed workflow succeeded, and v1.4.5 is now live on PyPI. -- Historical release narratives are maintained in `CHANGELOG.md`; STATUS.md is kept as a current-state snapshot. +- `ClaudeProvider` still has no `raw_complete()` / `raw_stream()` surface, so it + cannot serve as a Pi tool-calling backend through the Modelito HTTP server. + Anthropic tool-call response translation requires a dedicated follow-up. +- `GeminiProvider` and `GrokProvider` still have no `chat()` implementation; + they remain lower-priority compatibility shims. -## Pending tasks +## Remaining empirical work -- ClaudeProvider still has no raw_complete()/raw_stream() — it can't serve as a pi tool-calling backend through the modelito HTTP server. Implementing these requires the Anthropic SDK's tool-call response format, which is non-trivial and would need a dedicated follow-up. -- GeminiProvider and GrokProvider have no chat() — same pattern as Claude, lower priority since they're compatibility shims. +The repository now contains the benchmark needed for workload-specific local +selection, but a real Apple-Silicon performance ranking must be measured on the +actual target machine with comparable model families, quantisations, runtime +configuration, and cache state. GitHub-hosted Linux CI cannot provide that +evidence. +Until those measurements exist, `mac-performance` is intentionally a curated +candidate order with an explicit `prefer=` override rather than an empirical +winner table. ## Next steps -1. Keep reviewing provider additions against the portable-common-surface rule. -2. Continue monitoring Ollama raw passthrough behaviour and keep docs/tests aligned with OpenAI-compatible payload expectations. +1. Keep the current CI green and resolve all PR review findings before merging + the local-runtime work. +2. Run the conversational benchmark on the target Apple-Silicon machine with + equivalent models/configurations and record results separately from runtime + marketing benchmarks. +3. Keep reviewing provider additions against the portable-common-surface rule. +4. Continue monitoring Ollama raw passthrough behaviour and keep docs/tests + aligned with OpenAI-compatible payload expectations. +5. Address Claude raw passthrough and Gemini/Grok chat surfaces in dedicated, + bounded follow-ups rather than expanding the local-runtime PR. ## Longer-term steps 1. Maintain a small stable provider protocol surface. 2. Keep hosted SDK dependencies optional. -3. Expand provider-specific helpers only when they are clearly useful and well-contained. - - -## Decisions and rationale - -- API key storage should not move into a built-in encrypted database in the core package. +3. Expand provider-specific helpers only when they are clearly useful and + well-contained. + +## Risks and constraints + +- Local backend performance and model support change quickly upstream. +- Model identifiers and formats differ between runtimes. +- BaseRT, vllm-mlx, and oMLX are Apple-Silicon-oriented; hosted Linux CI can + validate adapters and policy but not their native execution. +- Readiness probes establish availability, not latency, quality, memory + pressure, cache effectiveness, or thermal behaviour. +- Capability metadata can become stale and should be reviewed when upstream + runtime behaviour materially changes. +- Deterministic fallbacks remain useful for tests but are inappropriate for the + strict local-only runtime path. + +## Standing rationale + +- API key storage should not move into a built-in encrypted database in the core + package. - Cloud-provider integrations should remain lightweight shims by default. -- The core value of the package is provider-agnostic normalisation, optional local tooling, and dependency-light embeddability. -- Consider persistent lifecycle-storage support if downstream tooling needs cross-process tracking. - -- Consider optional pluggable key-provider interfaces only if secret-storage demand grows. -- Deeper cloud-provider features should remain optional unless they map cleanly across providers. -- CI intentionally excludes integration tests by path/flags to keep default hosted CI fast and safe. +- The core value of the package is provider-agnostic normalisation, optional + local tooling, and dependency-light embeddability. +- CI intentionally excludes integration tests by path/flags to keep default + hosted CI fast and safe. -Last updated: 2026-06-01 12:00 +Last updated: 2026-08-09 diff --git a/docs/API.md b/docs/API.md index ae15cc6..5a0a099 100644 --- a/docs/API.md +++ b/docs/API.md @@ -19,6 +19,9 @@ surface. The primary exports (also visible via `from modelito import *`) are: - `Provider`, `SyncProvider`, `AsyncProvider`, `StreamingProvider`, `EmbeddingProvider`, `ChatProvider`, `RawChatProvider` — structural provider protocols for legacy, chat-first, and raw OpenAI-compatible code. - `Message`, `Response`, `MessageInput`, `OpenAIMessageDict` — message and response dataclasses / type helpers. - `ProviderStatus`, `check_provider_ready()`, `format_provider_status()` — readiness diagnostics helpers for local and hosted providers. +- `LocalRuntimeSelection`, `select_local_runtime()`, `local_client()` — explicit local-only runtime selection and client construction. +- `LocalRuntimeCapabilities`, `local_runtime_capabilities()` — conservative capability metadata for supported local runtime families. +- `LOCAL_PROFILE_AUTO`, `LOCAL_PROFILE_PORTABLE`, `LOCAL_PROFILE_MAC_PERFORMANCE`, `LOCAL_PROFILES`, `normalize_local_profile()`, `local_provider_candidates()`, `is_macos_apple_silicon()` — local deployment-profile helpers. - `OpenAICompatibleHTTPProvider` — shared HTTP base class for local OpenAI-compatible runtimes. - `OllamaProvider` — HTTP-aware provider that will call a local Ollama HTTP API when available (via the bundled `ollama_service` helpers). If the HTTP @@ -26,6 +29,8 @@ surface. The primary exports (also visible via `from modelito import *`) are: fallback (using `run_ollama_command`) before exposing a safe deterministic `summarize()` fallback useful for tests. - `OpenAIProvider` — SDK-backed hosted OpenAI provider; can also target hosted OpenAI-compatible APIs via `base_url`. +- `BaseRTProvider` — thin preset for a local BaseRT OpenAI-compatible server. +- `VLLMMLXProvider` — thin preset for a local vllm-mlx OpenAI-compatible server. - `OMLXProvider` — thin preset for local oMLX runtimes, built on `OpenAICompatibleHTTPProvider`. - `GeminiProvider`, `GrokProvider`, `ClaudeProvider` — minimal provider shims with the legacy `list_models()` / `summarize()` surface. - `EmbeddingProvider` — structural protocol for provider implementations that expose `embed(texts, **kwargs)`. @@ -115,6 +120,40 @@ Important `OllamaConnector` methods - `complete(conv_id: Optional[str], new_messages: Optional[Iterable]=None, settings: Optional[dict]=None) -> Response` — typed convenience wrapper returning a `Response` dataclass. - `acomplete(conv_id: Optional[str], new_messages: Optional[Iterable]=None, settings: Optional[dict]=None) -> Response` — asynchronous variant. +Local runtime selection +----------------------- + +`local_client(model=None, *, models=None, profile=None, prefer=None, host=None, port=None, base_url=None, base_urls=None, api_key=None, api_keys=None, probe_timeout=1.5, **provider_kwargs)` +: Construct a strict local-only `Client` after probing the configured local + runtime candidates. It does not fall back to hosted providers or to the + deterministic offline shim when no requested local runtime/model is ready. + +`select_local_runtime(...) -> LocalRuntimeSelection` +: Run the same read-only local selection logic without constructing a client. + The result records the resolved profile, provider, model, and endpoint. + +`local_runtime_capabilities(provider: str) -> LocalRuntimeCapabilities` +: Return coarse runtime-family metadata for streaming, prefix caching, + cancellation, structured output, tool calls, and model discovery. Values are + `yes`, `no`, `conditional`, or `unknown`; they do not replace feature-testing + the actual model/runtime configuration. + +Profiles: + +- `portable` — Ollama as the cross-platform path. +- `mac-performance` — Apple Silicon only; default candidate order is BaseRT, + vllm-mlx, oMLX, then Ollama. +- `auto` — resolves to `mac-performance` on Apple Silicon and `portable` + elsewhere. + +The candidate order is a deployment policy, not a universal performance +ranking. Pass `prefer=` to reorder candidates after benchmarking the target +workload. `models`, `base_urls`, and `api_keys` can map provider names to +runtime-specific values. + +See `docs/LOCAL-RUNTIMES.md` for benchmark usage, capability caveats, and the +upstream-source audit. + Provider shims -------------- @@ -196,8 +235,9 @@ OpenAI-compatible request payloads to supported providers. These methods preserv all request fields (including `tools`, `tool_choice`, `response_format`, etc.) and return raw OpenAI-compatible response dicts or streams without transformation. -**Availability**: `OpenAIProvider`, `OMLXProvider`, `OpenAICompatibleHTTPProvider`, -and `OllamaProvider` support raw passthrough. +**Availability**: `OpenAIProvider`, `BaseRTProvider`, `VLLMMLXProvider`, +`OMLXProvider`, `OpenAICompatibleHTTPProvider`, and `OllamaProvider` support raw +passthrough. `raw_complete(payload: dict[str, Any]) -> dict[str, Any]` : Send a raw OpenAI-compatible request payload and return the complete response @@ -310,10 +350,11 @@ For higher-level tooling, `ollama_service` now exposes two small dataclasses: CLI usage --------- -`modelito` exposes two small module-level CLIs useful during development: +`modelito` exposes small module-level CLIs useful during development: - `python -m modelito doctor` — diagnose provider readiness and report setup hints. - `modelito-serve` — optional OpenAI-compatible server (`/v1/models`, `/v1/chat/completions`, `/v1/embeddings`) requiring `pip install "modelito[serve]"`. +- `modelito-benchmark-local` — compare conversational latency against an already-running local OpenAI-compatible server; see `docs/LOCAL-RUNTIMES.md` for measurement caveats. - `python -m modelito.ollama_service` — minimal Ollama lifecycle CLI (`start`, `stop`, `install`, `inspect`, `pull`, `list-local`, `list-remote`, `version`). - `python -m modelito.timeout_cli` — print estimated timeouts and diagnostic details for a model. - `python -m modelito.timeout_calibrate` — write calibration prompts and (optionally) exercise a local Ollama server to collect timing samples. @@ -431,4 +472,4 @@ Performance & Caching: - Optional in-memory response caching: `ResponseCache`. - Batching utilities for embeddings and batchable operations: `batch_iterable`. -See the `tests/` directory for usage examples and coverage for all features. +See the `tests/` directory for usage examples and coverage for all features. \ No newline at end of file diff --git a/docs/LOCAL-RUNTIMES.md b/docs/LOCAL-RUNTIMES.md new file mode 100644 index 0000000..5186f2b --- /dev/null +++ b/docs/LOCAL-RUNTIMES.md @@ -0,0 +1,301 @@ +# Local runtime profiles + +Modelito separates local deployment policy from provider APIs. The selector +answers a narrow question: given an application's deployment intent, which +already-running local backend and model are ready to use? + +The profiles are deliberately not performance rankings: + +- **portable**: use Ollama as the common macOS/Linux/Windows path; +- **mac-performance**: on Apple Silicon, try BaseRT, vllm-mlx, oMLX, then + Ollama; +- **auto**: choose `mac-performance` on Apple Silicon and `portable` + elsewhere. + +The `mac-performance` order is a practical starting policy. It does **not** +mean that BaseRT, vllm-mlx, oMLX, or Ollama is universally faster than the +others. Model family, quantisation, prompt length, cache state, concurrency, +memory pressure, and runtime version all matter. Benchmark the target workload +and pass `prefer=` when a different order is appropriate. + +## Python + +```python +from modelito import local_client + +portable = local_client(model="my-ollama-tag", profile="portable") +``` + +Provider-specific model identifiers are supported because an Ollama tag, an +MLX/Hugging Face model ID, and a BaseRT model ID should not be assumed to be +interchangeable: + +```python +from modelito import local_client + +client = local_client( + profile="mac-performance", + models={ + "basert": "my-base-model", + "vllm-mlx": "my-vllm-model", + "omlx": "my-omlx-model", + "ollama": "my-ollama-tag", + }, +) +``` + +`local_client()` is strict about local execution: it does not silently fall +back to a hosted provider or Modelito's deterministic offline shim. Existing +`Client(provider="auto")` behaviour is unchanged. + +OpenAI-compatible local backends can also have provider-specific endpoints and +keys: + +```python +client = local_client( + profile="mac-performance", + models={ + "basert": "my-base-model", + "vllm-mlx": "my-vllm-model", + "omlx": "my-omlx-model", + }, + base_urls={ + "basert": "http://127.0.0.1:8080/v1", + "vllm-mlx": "http://127.0.0.1:8000/v1", + "omlx": "http://127.0.0.1:8001/v1", + }, + api_keys={"basert": "local-key"}, +) +``` + +vllm-mlx and oMLX both commonly use port 8000 by default, so run them on +distinct ports when comparing them simultaneously. + +To override the default order after benchmarking a machine: + +```python +client = local_client( + profile="mac-performance", + models={ + "basert": "my-base-model", + "vllm-mlx": "my-vllm-model", + "omlx": "my-omlx-model", + "ollama": "my-ollama-tag", + }, + prefer=["vllm-mlx", "omlx", "ollama", "basert"], +) +``` + +Use `select_local_runtime()` when an application wants to inspect the decision +before constructing a client: + +```python +from modelito import select_local_runtime + +selection = select_local_runtime( + profile="auto", + models={ + "basert": "my-base-model", + "vllm-mlx": "my-vllm-model", + "omlx": "my-omlx-model", + "ollama": "my-ollama-tag", + }, +) +print(selection.provider, selection.model, selection.endpoint) +``` + +## Runtime capabilities + +`local_runtime_capabilities()` exposes conservative, coarse runtime metadata: + +```python +from modelito import local_runtime_capabilities + +caps = local_runtime_capabilities("vllm-mlx") +print(caps.streaming, caps.prefix_cache, caps.tool_calls) +``` + +Capabilities describe a runtime family, not every model or configuration. +`conditional` means that model support, a parser, a chat template, packaging, +or server configuration can affect the feature. `unknown` means Modelito does +not currently make the claim. + +| Runtime | Streaming | Prefix cache | Cancellation | Structured output | Tool calls | Model discovery | +| --- | --- | --- | --- | --- | --- | --- | +| BaseRT | yes | yes | unknown | unknown | yes | yes | +| vllm-mlx | yes | yes | yes | yes | conditional | yes | +| oMLX | yes | yes | yes | conditional | conditional | yes | +| Ollama | yes | yes | unknown | yes | conditional | yes | + +Do not use this table as a substitute for feature-testing the actual model and +server version used by an application. + +## Runtime notes + +### BaseRT + +BaseRT is a native Metal runtime for Apple Silicon with an OpenAI-compatible +server. Current upstream material documents chat/completions, embeddings, +transcription, tool calls, continuous batching, paged KV cache, and prefix +caching. Base Compute also publishes performance comparisons against MLX and +llama.cpp; those are upstream benchmarks, not Modelito's evidence that BaseRT +wins on an arbitrary machine or workload. + +Default Modelito endpoint: `http://127.0.0.1:8080/v1`. + +### vllm-mlx + +vllm-mlx is an Apple-Silicon MLX server with OpenAI-compatible serving. Current +upstream documentation includes continuous batching, paged/prefix KV caching, +structured JSON output, tool-call parsers, request cancellation, and a +`bench-serve` command. Several features are configuration- or model-dependent. + +Default Modelito endpoint: `http://localhost:8000/v1`. + +### oMLX + +oMLX is an MLX-native server oriented towards persistent local workflows. It +supports OpenAI- and Anthropic-compatible APIs, continuous batching, prefix +sharing, and a tiered hot-RAM/SSD KV cache. Tool calling and structured output +are available but can depend on the model's chat template and parser support. +Current releases also include request-abort handling for streaming API paths. + +Default Modelito endpoint: `http://localhost:8000/v1`. + +### Ollama + +Ollama is the portability path and should not be treated as merely a slow +fallback. Current Apple-Silicon releases include an MLX engine and cache +snapshot/prefix-reuse work aimed at repeated conversational and agent contexts. +Which model formats and execution paths are available can change between +releases, so Modelito does not assume a fixed performance relationship between +Ollama and the dedicated Apple-Silicon servers. + +### Generic OpenAI-compatible servers + +`OpenAICompatibleHTTPProvider` remains the generic route for llama.cpp, LM +Studio, vLLM, MLX-LM's HTTP server, and other compatible endpoints. Arbitrary +servers are **not** part of automatic local-runtime selection because Modelito +cannot infer an unknown endpoint safely. + +## MLX-LM as a benchmark reference + +Raw `mlx-lm` is intentionally not another `local_client()` provider. It is a +useful reference implementation for Apple-Silicon measurements and supports +prompt caching for repeated contexts and multi-turn dialogue. Its built-in HTTP +server is OpenAI-like, but MLX-LM itself cautions that the server is not +recommended for production use because it provides only basic security checks. + +The benchmark CLI therefore accepts `--provider mlx-lm` as a **measurement +label/reference path**, without making it part of Modelito's runtime-selection +policy. + +## Conversational benchmark + +Modelito includes a small cross-runtime benchmark for an already-running +OpenAI-compatible server: + +```bash +modelito-benchmark-local \ + --provider vllm-mlx \ + --model my-model \ + --repetitions 3 \ + --json \ + --output vllm-mlx.json +``` + +Other examples: + +```bash +modelito-benchmark-local --provider basert --model my-model --json +modelito-benchmark-local --provider omlx --model my-model --json +modelito-benchmark-local --provider ollama --model my-model --json +modelito-benchmark-local --provider mlx-lm --model my-model --json +modelito-benchmark-local \ + --provider openai-compatible \ + --base-url http://127.0.0.1:9000/v1 \ + --model my-model \ + --json +``` + +If `--model` is omitted, the benchmark asks `/models` for the first advertised +model. To sample the server process's RSS, add `--pid `. + +The benchmark records: + +- first-request time to first streamed token (TTFT); +- time to the first phrase-like chunk useful to a UI; +- estimated decode tokens per second; +- repeated warm-prefix TTFT; +- TTFT as synthetic conversational history grows; +- client stream-close latency followed by a post-cancellation probe; +- approximate peak process RSS when a server PID is supplied. + +The JSON embeds the measurement caveats. In particular: + +1. `first_request` is a cold-model result only if the server/model was actually + cold before the benchmark began. +2. Decode token rate uses Modelito's token-count helper, not the runtime's + native tokenizer accounting, so use it for within-workload comparison rather + than precision benchmarking. +3. Client stream-close latency does not prove that the server acknowledged or + completed cancellation internally. +4. Process RSS can under-report Metal/unified-memory use on macOS. +5. The benchmark measures one client workload. It does not replace a runtime's + own throughput/concurrency benchmark. + +For vllm-mlx specifically, its upstream `bench-serve` command is a useful +second measurement because it reports TTFT, TPOT, throughput, cache deltas, and +Metal memory using runtime-specific instrumentation. + +## What to compare for conversational systems + +For a latency-sensitive conversational application, compare at least: + +- first-turn and warm-turn TTFT; +- prompt-processing time as context grows; +- prefix-cache effectiveness across repeated history; +- sustained generation speed; +- memory pressure while other local models are resident; +- cancellation behaviour; +- output quality for the actual language and register; +- thermal behaviour over a long session. + +Do not select a runtime solely from a headline tokens-per-second figure. + +## Environment + +`MODELITO_LOCAL_PROFILE` sets the profile used when `profile=` is omitted. +Accepted values are: + +- `auto` +- `portable` +- `mac-performance` + +Aliases `mac`, `macos`, and `apple-silicon` resolve to `mac-performance`. +Provider aliases `vllm_mlx`/`vllmmlx` resolve to `vllm-mlx`, and `om` resolves +to `omlx`. + +## Upstream sources reviewed in August 2026 + +The runtime descriptions above were checked against current upstream material: + +- BaseRT: and + +- vllm-mlx: and + +- oMLX: +- Ollama MLX/cache work: and + +- MLX-LM prompt caching and HTTP server: + and + + +These sources justify exposing several local paths and benchmark overrides. +They do not establish a universal runtime winner. + +## Licensing note + +Modelito integrates these runtimes through their documented local HTTP APIs and +does not redistribute their engines. Applications that redistribute or bundle a +runtime should review that runtime's current licence independently. \ No newline at end of file diff --git a/modelito/__init__.py b/modelito/__init__.py index 2683291..81ae09b 100644 --- a/modelito/__init__.py +++ b/modelito/__init__.py @@ -5,7 +5,7 @@ """ try: - from importlib.metadata import version, PackageNotFoundError + from importlib.metadata import PackageNotFoundError, version except Exception: __version__ = "1.4.6" else: @@ -14,96 +14,113 @@ except PackageNotFoundError: __version__ = "1.4.6" -from .tokenizer import count_tokens -from .timeout import estimate_remote_timeout, estimate_remote_timeout_details -from .plumbing import ( - ErrorEnvelope, - ResponseEnvelope, - TransportPolicy, - envelope_error, - envelope_ok, - normalize_network_error, - retry_with_backoff, -) -from .connector import OllamaConnector +from .basert import BaseRTProvider +from .claude import ClaudeProvider +from .client import Client from .config import load_config, parse_host_port +from .connector import OllamaConnector +from .doctor import ProviderStatus, check_provider_ready, format_provider_status +from .embeddings import Embedder, StubEmbeddingProvider, embed_texts from .exceptions import ( LLMProviderError, - ModelitoConnectionError, - ModelitoTimeoutError, ModelitoBadResponseError, - ModelitoProviderError, + ModelitoConnectionError, ModelitoModelNotFoundError, + ModelitoProviderError, + ModelitoTimeoutError, ) -from .openai_compat import OpenAICompatibleHTTPProvider -from .ollama_service import server_is_up, endpoint_url -from .ollama_service import ensure_ollama_running +from .gemini import GeminiProvider +from .grok import GrokProvider +from .local_runtime import ( + LOCAL_PROFILE_AUTO, + LOCAL_PROFILE_MAC_PERFORMANCE, + LOCAL_PROFILE_PORTABLE, + LOCAL_PROFILES, + LocalRuntimeCapabilities, + LocalRuntimeSelection, + is_macos_apple_silicon, + local_client, + local_provider_candidates, + local_runtime_capabilities, + normalize_local_profile, + select_local_runtime, +) +from .messages import Message, Messages, Response, flatten_message_inputs +from .model_metadata import ( + ModelMetadata, + get_model_info, + get_model_metadata, + infer_model_metadata, +) +from .normalization import normalize_metadata, normalize_models from .ollama_service import ( - RemoteModelCatalogEntry, ModelLifecycleState, ReadinessResult, + RemoteModelCatalogEntry, + async_ensure_model_ready, + async_ensure_model_ready_detailed, + change_ollama_config, + clear_model_lifecycle_state, + delete_model, detect_install_method, + download_model, + download_model_progress, + endpoint_url, + ensure_model_loaded, + ensure_model_ready, + ensure_model_ready_detailed, + ensure_ollama_running, + ensure_ollama_running_verbose, + find_ollama_listener_pids, + get_model_lifecycle_state, get_ollama_binary, install_ollama, - start_ollama, - stop_ollama, - update_ollama, + install_service, list_local_models, - list_remote_models, + list_model_lifecycle_states, list_remote_model_catalog, - download_model, - download_model_progress, - delete_model, - serve_model, - change_ollama_config, + list_remote_models, ollama_binary_candidates, - resolve_ollama_command, - ollama_installed, - run_ollama_command, - start_detached_ollama_serve, - wait_until_ready, - preload_model, - ensure_model_ready, - ensure_model_ready_detailed, - ensure_model_loaded, ollama_health_check, + ollama_installed, ollama_readiness_probe, + preload_model, + resolve_ollama_command, + run_ollama_command, running_model_names, - get_model_lifecycle_state, - list_model_lifecycle_states, - clear_model_lifecycle_state, - find_ollama_listener_pids, + serve_model, + server_is_up, + start_detached_ollama_serve, + start_ollama, + stop_ollama, stop_service, - install_service, - ensure_ollama_running_verbose, - async_ensure_model_ready, - async_ensure_model_ready_detailed, + update_ollama, + wait_until_ready, ) -from .ollama import OllamaProvider -from .gemini import GeminiProvider -from .grok import GrokProvider -from .openai import OpenAIProvider -from .claude import ClaudeProvider +from .ollama_strict import OllamaProvider from .omlx import OMLXProvider +from .openai import OpenAIProvider +from .openai_compat import OpenAICompatibleHTTPProvider +from .plumbing import ( + ErrorEnvelope, + ResponseEnvelope, + TransportPolicy, + envelope_error, + envelope_ok, + normalize_network_error, + retry_with_backoff, +) from .provider import ( - Provider, - EmbeddingProvider, ChatProvider, - RawChatProvider, + EmbeddingProvider, MessageInput, OpenAIMessageDict, + Provider, + RawChatProvider, ) -from .client import Client -from .doctor import ProviderStatus, check_provider_ready, format_provider_status -from .embeddings import Embedder, StubEmbeddingProvider, embed_texts -from .messages import Message, Messages, Response, flatten_message_inputs -from .model_metadata import ( - ModelMetadata, - get_model_info, - get_model_metadata, - infer_model_metadata, -) -from .normalization import normalize_models, normalize_metadata +from .timeout import estimate_remote_timeout, estimate_remote_timeout_details +from .tokenizer import count_tokens +from .vllm_mlx import VLLMMLXProvider __all__ = [ "__version__", @@ -124,6 +141,8 @@ "OpenAIProvider", "ClaudeProvider", "OMLXProvider", + "BaseRTProvider", + "VLLMMLXProvider", "Provider", "EmbeddingProvider", "ChatProvider", @@ -134,6 +153,18 @@ "ModelLifecycleState", "ReadinessResult", "Client", + "LOCAL_PROFILE_AUTO", + "LOCAL_PROFILE_PORTABLE", + "LOCAL_PROFILE_MAC_PERFORMANCE", + "LOCAL_PROFILES", + "LocalRuntimeSelection", + "LocalRuntimeCapabilities", + "is_macos_apple_silicon", + "normalize_local_profile", + "local_provider_candidates", + "local_runtime_capabilities", + "select_local_runtime", + "local_client", "ProviderStatus", "check_provider_ready", "format_provider_status", diff --git a/modelito/basert.py b/modelito/basert.py new file mode 100644 index 0000000..e493e0d --- /dev/null +++ b/modelito/basert.py @@ -0,0 +1,42 @@ +"""BaseRT provider with OpenAI-compatible local runtime integration. + +BaseRT exposes an OpenAI-compatible HTTP server on Apple Silicon. This provider +reuses Modelito's shared OpenAI-compatible transport and only supplies BaseRT +specific defaults. +""" + +from __future__ import annotations + +from typing import Optional + +from .openai_compat import OpenAICompatibleHTTPProvider + + +class BaseRTProvider(OpenAICompatibleHTTPProvider): + """Provider for a local ``basert serve`` endpoint. + + Args: + base_url: Base URL of the BaseRT OpenAI-compatible endpoint. + Defaults to ``http://127.0.0.1:8080/v1``. + model: Loaded BaseRT model identifier. + api_key: Optional bearer token matching ``basert serve --api-key``. + timeout: HTTP timeout in seconds. + strict: When ``True``, raise typed Modelito errors instead of falling + back to deterministic offline behaviour. + """ + + def __init__( + self, + base_url: Optional[str] = None, + model: Optional[str] = None, + api_key: Optional[str] = None, + timeout: float = 20.0, + strict: bool = False, + ) -> None: + super().__init__( + base_url=base_url or "http://127.0.0.1:8080/v1", + model=model or "basert", + api_key=api_key, + timeout=timeout, + strict=strict, + ) diff --git a/modelito/doctor.py b/modelito/doctor.py index 9503721..5499488 100644 --- a/modelito/doctor.py +++ b/modelito/doctor.py @@ -21,9 +21,30 @@ def _normalize_provider_name(provider: str) -> str: name = str(provider or "").strip().lower() - if name == "om": - return "omlx" - return name + aliases = { + "om": "omlx", + "vllm_mlx": "vllm-mlx", + "vllmmlx": "vllm-mlx", + } + return aliases.get(name, name) + + +def _probe_basert( + model: Optional[str], + base_url: Optional[str], + api_key: Optional[str], + probe_timeout: float, +) -> ProviderStatus: + return probes.probe_basert_status(model, base_url, api_key, probe_timeout) + + +def _probe_vllm_mlx( + model: Optional[str], + base_url: Optional[str], + api_key: Optional[str], + probe_timeout: float, +) -> ProviderStatus: + return probes.probe_vllm_mlx_status(model, base_url, api_key, probe_timeout) def _probe_omlx( @@ -64,7 +85,10 @@ def _probe_openai( endpoint=endpoint, models=models, reason="" if ready else "requested model not found or API unavailable", - setup_hint="Set OPENAI_API_KEY and optional OPENAI_BASE_URL if you are targeting a hosted OpenAI-compatible API.", + setup_hint=( + "Set OPENAI_API_KEY and optional OPENAI_BASE_URL if you are " + "targeting a hosted OpenAI-compatible API." + ), ) except Exception as exc: return probes.build_status( @@ -73,7 +97,10 @@ def _probe_openai( endpoint=base_url or "https://api.openai.com/v1", models=[], reason="OpenAI provider unavailable", - setup_hint="Set OPENAI_API_KEY and optional OPENAI_BASE_URL if you are targeting a hosted OpenAI-compatible API.", + setup_hint=( + "Set OPENAI_API_KEY and optional OPENAI_BASE_URL if you are " + "targeting a hosted OpenAI-compatible API." + ), details={"error": str(exc)}, ) @@ -86,9 +113,11 @@ def _probe_generic_provider(provider: str, model: Optional[str]) -> ProviderStat provider, False, reason=f"Unknown provider: {provider}", - setup_hint="Pick one of the built-in providers or configure a valid provider profile.", + setup_hint=( + "Pick one of the built-in providers or configure a valid " + "provider profile." + ), ) - models = [] try: models = list(resolved.list_models()) except Exception as exc: @@ -122,6 +151,27 @@ def _probe_generic_provider(provider: str, model: Optional[str]) -> ProviderStat ) +def _probe_local_candidate( + provider: str, + model: Optional[str], + *, + host: Optional[str], + port: Optional[int], + base_url: Optional[str], + api_key: Optional[str], + probe_timeout: float, +) -> ProviderStatus: + if provider == "basert": + return _probe_basert(model, base_url, api_key, probe_timeout) + if provider == "vllm-mlx": + return _probe_vllm_mlx(model, base_url, api_key, probe_timeout) + if provider == "omlx": + return _probe_omlx(model, base_url, api_key, probe_timeout) + if provider == "ollama": + return _probe_ollama(model, host, port, probe_timeout) + return _probe_generic_provider(provider, model) + + def check_provider_ready( provider: str, model: Optional[str] = None, @@ -136,42 +186,60 @@ def check_provider_ready( """Diagnose whether a provider looks ready to use. The helper is read-only: it does not install, download, or mutate state. + ``auto`` follows the same local candidate family as ``local_client()``. """ normalized = _normalize_provider_name(provider) if normalized == "auto": - default_prefer: List[str] = list( - prefer or (["omlx", "ollama"] if _is_macos_apple_silicon() else ["ollama"]) - ) - for candidate in default_prefer: - candidate_name = _normalize_provider_name(candidate) - if candidate_name == "omlx": - status = _probe_omlx(model, base_url, api_key, probe_timeout) - elif candidate_name == "ollama": - status = _probe_ollama(model, host, port, probe_timeout) - else: - status = _probe_generic_provider(candidate_name, model) + if prefer: + default_prefer = [_normalize_provider_name(name) for name in prefer] + elif _is_macos_apple_silicon(): + default_prefer = ["basert", "vllm-mlx", "omlx", "ollama"] + else: + default_prefer = ["ollama"] + + for candidate_name in default_prefer: + status = _probe_local_candidate( + candidate_name, + model, + host=host, + port=port, + base_url=base_url, + api_key=api_key, + probe_timeout=probe_timeout, + ) if status.ready: return status + if _is_macos_apple_silicon(): return probes.build_status( "auto", False, reason="No local backend was ready on macOS Apple Silicon", setup_hint=( - "Start oMLX at http://localhost:8000/v1 or Ollama at http://127.0.0.1:11434, then ensure the requested model is available." + "Start BaseRT, vllm-mlx, oMLX, or Ollama and ensure the " + "requested model is available." ), ) return probes.build_status( "auto", False, reason="No suitable provider was ready", - setup_hint="Configure a provider profile, environment override, or a local backend.", + setup_hint=( + "Configure a provider profile, environment override, or a local " + "backend." + ), ) - if normalized == "omlx": - return _probe_omlx(model, base_url, api_key, probe_timeout) - if normalized == "ollama": - return _probe_ollama(model, host, port, probe_timeout) + if normalized in {"basert", "vllm-mlx", "omlx", "ollama"}: + return _probe_local_candidate( + normalized, + model, + host=host, + port=port, + base_url=base_url, + api_key=api_key, + probe_timeout=probe_timeout, + ) if normalized == "openai": return _probe_openai(model, base_url, api_key) @@ -223,7 +291,9 @@ def build_parser() -> argparse.ArgumentParser: doctor.add_argument("--host", default=None, help="Provider host override") doctor.add_argument("--port", type=int, default=None, help="Provider port override") doctor.add_argument( - "--base-url", default=None, help="OpenAI/oMLX-compatible base URL override" + "--base-url", + default=None, + help="OpenAI-compatible provider base URL override", ) doctor.add_argument("--api-key", default=None, help="Optional API key override") doctor.add_argument( diff --git a/modelito/local_benchmark.py b/modelito/local_benchmark.py new file mode 100644 index 0000000..0049d66 --- /dev/null +++ b/modelito/local_benchmark.py @@ -0,0 +1,560 @@ +"""Conversational benchmark for running local OpenAI-compatible runtimes. + +The benchmark is deliberately runtime-neutral. It measures observable client +latency for a representative conversation rather than declaring a universal +provider winner. Results should be compared on the same machine, model family, +quantisation, context, and server configuration. +""" + +from __future__ import annotations + +import argparse +import json +import statistics +import subprocess +import threading +import time +from dataclasses import asdict, dataclass +from datetime import datetime, timezone +from typing import Any, Dict, Iterable, List, Optional, Sequence +from urllib.request import Request, urlopen + +from .tokenizer import count_tokens + +_PROVIDER_BASE_URLS = { + "basert": "http://127.0.0.1:8080/v1", + "vllm-mlx": "http://127.0.0.1:8000/v1", + "omlx": "http://127.0.0.1:8000/v1", + "ollama": "http://127.0.0.1:11434/v1", + "mlx-lm": "http://127.0.0.1:8080/v1", +} +_PROVIDER_ALIASES = { + "vllm_mlx": "vllm-mlx", + "vllmmlx": "vllm-mlx", + "mlx_lm": "mlx-lm", + "generic": "openai-compatible", +} +_SUPPORTED_PROVIDERS = tuple(_PROVIDER_BASE_URLS) + ("openai-compatible",) + + +@dataclass(frozen=True) +class StreamMetrics: + """Observable metrics for one streamed chat request.""" + + ttft_ms: Optional[float] + first_useful_ms: Optional[float] + elapsed_ms: float + output_chars: int + estimated_output_tokens: int + estimated_decode_tokens_per_s: Optional[float] + cancel_close_ms: Optional[float] = None + + +def _normalise_provider(provider: str) -> str: + value = str(provider or "").strip().lower() + return _PROVIDER_ALIASES.get(value, value) + + +def _headers(api_key: Optional[str]) -> Dict[str, str]: + headers = { + "Accept": "text/event-stream", + "Content-Type": "application/json", + } + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + return headers + + +def _extract_stream_text(event: Any) -> str: + if not isinstance(event, dict): + return "" + choices = event.get("choices") + if not isinstance(choices, list) or not choices: + return "" + first = choices[0] + if not isinstance(first, dict): + return "" + delta = first.get("delta") + if isinstance(delta, dict): + content = delta.get("content") + if isinstance(content, str): + return content + text = first.get("text") + return text if isinstance(text, str) else "" + + +def _is_useful_phrase(text: str) -> bool: + """Return whether streamed text has become visibly phrase-like. + + This is a UI-oriented heuristic rather than a linguistic metric: terminal + punctuation/newline qualifies immediately; otherwise a multi-word fragment + of at least 24 characters is considered useful enough to display. + """ + + stripped = text.strip() + if not stripped: + return False + if any(mark in stripped for mark in (".", "!", "?", "\n")): + return True + return len(stripped) >= 24 and " " in stripped + + +def _discover_model( + base_url: str, api_key: Optional[str], timeout: float +) -> str: + request = Request( + f"{base_url.rstrip('/')}/models", + headers=_headers(api_key), + method="GET", + ) + with urlopen(request, timeout=timeout) as response: + payload = json.loads(response.read().decode("utf-8")) + models = payload.get("data") if isinstance(payload, dict) else None + if isinstance(models, list): + for item in models: + if isinstance(item, dict) and isinstance(item.get("id"), str): + return item["id"] + raise RuntimeError("The server returned no model IDs from /models") + + +def _stream_chat( + base_url: str, + model: str, + messages: Iterable[Dict[str, str]], + *, + api_key: Optional[str], + timeout: float, + max_tokens: int, + stop_after_first: bool = False, +) -> StreamMetrics: + payload = { + "model": model, + "messages": list(messages), + "stream": True, + "max_tokens": int(max_tokens), + "temperature": 0, + } + request = Request( + f"{base_url.rstrip('/')}/chat/completions", + data=json.dumps(payload).encode("utf-8"), + headers=_headers(api_key), + method="POST", + ) + + start = time.perf_counter() + first_token_at: Optional[float] = None + first_useful_at: Optional[float] = None + cancel_close_ms: Optional[float] = None + parts: List[str] = [] + + response = urlopen(request, timeout=timeout) + try: + while True: + raw_line = response.readline() + if not raw_line: + break + line = raw_line.decode("utf-8", errors="ignore").strip() + if not line: + continue + if line.startswith("data:"): + line = line[5:].lstrip() + if line == "[DONE]": + break + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + text = _extract_stream_text(event) + if not text: + continue + now = time.perf_counter() + if first_token_at is None: + first_token_at = now + parts.append(text) + accumulated = "".join(parts) + if first_useful_at is None and _is_useful_phrase(accumulated): + first_useful_at = now + if stop_after_first: + close_start = time.perf_counter() + response.close() + cancel_close_ms = (time.perf_counter() - close_start) * 1000.0 + break + finally: + response.close() + + end = time.perf_counter() + output = "".join(parts) + token_count = count_tokens(output) if output else 0 + ttft_ms = ( + (first_token_at - start) * 1000.0 if first_token_at is not None else None + ) + useful_ms = ( + (first_useful_at - start) * 1000.0 if first_useful_at is not None else None + ) + decode_tps: Optional[float] = None + if first_token_at is not None and token_count > 1 and end > first_token_at: + decode_tps = (token_count - 1) / (end - first_token_at) + + return StreamMetrics( + ttft_ms=ttft_ms, + first_useful_ms=useful_ms, + elapsed_ms=(end - start) * 1000.0, + output_chars=len(output), + estimated_output_tokens=token_count, + estimated_decode_tokens_per_s=decode_tps, + cancel_close_ms=cancel_close_ms, + ) + + +def _median(values: Iterable[Optional[float]]) -> Optional[float]: + present = [float(value) for value in values if value is not None] + return statistics.median(present) if present else None + + +def _summarise_samples(samples: Sequence[StreamMetrics]) -> Dict[str, Any]: + return { + "samples": [asdict(sample) for sample in samples], + "median_ttft_ms": _median(sample.ttft_ms for sample in samples), + "median_first_useful_ms": _median( + sample.first_useful_ms for sample in samples + ), + "median_estimated_decode_tokens_per_s": _median( + sample.estimated_decode_tokens_per_s for sample in samples + ), + } + + +def _conversation(turns: int) -> List[Dict[str, str]]: + messages: List[Dict[str, str]] = [ + { + "role": "system", + "content": ( + "Answer concisely. Preserve the markers supplied by the user so " + "the final question can refer to the latest one." + ), + } + ] + for index in range(1, turns + 1): + marker = f"MARKER-{index:02d}-ALPHA" + messages.extend( + [ + { + "role": "user", + "content": ( + f"Conversation turn {index}. Remember marker {marker}. " + "Reply only that you have recorded it." + ), + }, + {"role": "assistant", "content": f"Recorded {marker}."}, + ] + ) + messages.append( + { + "role": "user", + "content": "In one short sentence, state the latest marker you recorded.", + } + ) + return messages + + +def _read_rss_kb(pid: int) -> Optional[int]: + try: + result = subprocess.run( + ["ps", "-o", "rss=", "-p", str(pid)], + check=False, + capture_output=True, + text=True, + timeout=1.0, + ) + text = result.stdout.strip() + return int(text.splitlines()[0]) if text else None + except (OSError, ValueError, subprocess.SubprocessError): + return None + + +class _PeakRSSSampler: + def __init__(self, pid: Optional[int], interval: float = 0.05) -> None: + self.pid = pid + self.interval = interval + self.peak_kb: Optional[int] = None + self._stop = threading.Event() + self._thread: Optional[threading.Thread] = None + + def start(self) -> None: + if self.pid is None: + return + self._thread = threading.Thread(target=self._run, daemon=True) + self._thread.start() + + def _run(self) -> None: + assert self.pid is not None + while not self._stop.is_set(): + rss = _read_rss_kb(self.pid) + if rss is not None and (self.peak_kb is None or rss > self.peak_kb): + self.peak_kb = rss + self._stop.wait(self.interval) + + def stop(self) -> Optional[float]: + if self._thread is None: + return None + self._stop.set() + self._thread.join(timeout=2.0) + return self.peak_kb / 1024.0 if self.peak_kb is not None else None + + +def run_benchmark( + *, + provider: str, + base_url: str, + model: str, + api_key: Optional[str], + timeout: float, + max_tokens: int, + repetitions: int, + context_turns: Sequence[int], + pid: Optional[int], +) -> Dict[str, Any]: + """Run the conversational benchmark against one already-running server.""" + + sampler = _PeakRSSSampler(pid) + sampler.start() + try: + first_request = _stream_chat( + base_url, + model, + _conversation(1), + api_key=api_key, + timeout=timeout, + max_tokens=max_tokens, + ) + + warm_messages = _conversation(max(4, max(context_turns, default=4))) + warm_baseline = _stream_chat( + base_url, + model, + warm_messages, + api_key=api_key, + timeout=timeout, + max_tokens=max_tokens, + ) + warm_reuse = [ + _stream_chat( + base_url, + model, + warm_messages, + api_key=api_key, + timeout=timeout, + max_tokens=max_tokens, + ) + for _ in range(repetitions) + ] + + scaling: List[Dict[str, Any]] = [] + for turns in context_turns: + samples = [ + _stream_chat( + base_url, + model, + _conversation(turns), + api_key=api_key, + timeout=timeout, + max_tokens=max_tokens, + ) + for _ in range(repetitions) + ] + scaling.append({"turns": turns, **_summarise_samples(samples)}) + + cancellation_stream = _stream_chat( + base_url, + model, + [ + { + "role": "user", + "content": ( + "Write a long numbered explanation of local language-model " + "inference, with at least twenty items." + ), + } + ], + api_key=api_key, + timeout=timeout, + max_tokens=max(256, max_tokens), + stop_after_first=True, + ) + post_cancel_probe = _stream_chat( + base_url, + model, + [{"role": "user", "content": "Reply with the single word ready."}], + api_key=api_key, + timeout=timeout, + max_tokens=min(max_tokens, 32), + ) + finally: + peak_rss_mb = sampler.stop() + + return { + "schema_version": 1, + "generated_at": datetime.now(timezone.utc).isoformat(), + "provider": provider, + "base_url": base_url, + "model": model, + "repetitions": repetitions, + "first_request": asdict(first_request), + "warm_prefix": { + "baseline": asdict(warm_baseline), + "reuse": _summarise_samples(warm_reuse), + }, + "context_scaling": scaling, + "cancellation": { + "client_stream_close": asdict(cancellation_stream), + "post_cancel_probe": asdict(post_cancel_probe), + }, + "peak_process_rss_mb": peak_rss_mb, + "measurement_notes": { + "first_request": ( + "This is the first request made by this benchmark process. It is a " + "cold-model metric only if the runtime/model was actually cold." + ), + "tokens_per_second": ( + "Output tokens are estimated with Modelito's tokenizer helper; this " + "is not the runtime's native tokenizer accounting." + ), + "first_useful": ( + "A UI heuristic: terminal punctuation/newline, or a multi-word " + "fragment of at least 24 characters." + ), + "cancellation": ( + "cancel_close_ms measures how quickly the client closes the HTTP " + "stream. It does not prove server-side cancellation acknowledgement." + ), + "rss": ( + "If --pid is supplied, peak_process_rss_mb samples process RSS via " + "ps. On macOS this can under-report native Metal/unified-memory use." + ), + }, + } + + +def _parse_context_turns(value: str) -> List[int]: + out: List[int] = [] + for item in value.split(","): + item = item.strip() + if not item: + continue + turns = int(item) + if turns < 1: + raise ValueError("context turns must be positive integers") + if turns not in out: + out.append(turns) + if not out: + raise ValueError("at least one context-turn value is required") + return out + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="modelito-benchmark-local", + description=( + "Benchmark conversational latency for a running local " + "OpenAI-compatible LLM server." + ), + ) + parser.add_argument( + "--provider", + required=True, + help=( + "Runtime label: basert, vllm-mlx, omlx, ollama, mlx-lm, or " + "openai-compatible" + ), + ) + parser.add_argument("--base-url", default=None, help="Override API base URL") + parser.add_argument("--model", default=None, help="Model ID; auto-detect if omitted") + parser.add_argument("--api-key", default=None, help="Optional bearer token") + parser.add_argument("--timeout", type=float, default=120.0) + parser.add_argument("--max-tokens", type=int, default=96) + parser.add_argument("--repetitions", type=int, default=3) + parser.add_argument("--context-turns", default="1,2,4,8") + parser.add_argument( + "--pid", + type=int, + default=None, + help="Optional server PID for approximate process-RSS sampling", + ) + parser.add_argument("--output", default=None, help="Write JSON result to this path") + parser.add_argument( + "--json", action="store_true", help="Print the complete JSON result" + ) + return parser + + +def _format_number(value: Optional[float]) -> str: + return "n/a" if value is None else f"{value:.1f}" + + +def _print_summary(result: Dict[str, Any]) -> None: + first = result["first_request"] + warm = result["warm_prefix"]["reuse"] + print(f"provider: {result['provider']}") + print(f"model: {result['model']}") + print(f"first request TTFT: {_format_number(first['ttft_ms'])} ms") + print( + "warm-prefix median TTFT: " + f"{_format_number(warm['median_ttft_ms'])} ms" + ) + print( + "warm-prefix median estimated decode: " + f"{_format_number(warm['median_estimated_decode_tokens_per_s'])} tok/s" + ) + peak = result.get("peak_process_rss_mb") + print(f"peak sampled process RSS: {_format_number(peak)} MB") + print("See JSON output for context-scaling, first-useful, and cancellation metrics.") + + +def main(argv: Optional[List[str]] = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + + provider = _normalise_provider(args.provider) + if provider not in _SUPPORTED_PROVIDERS: + parser.error( + "--provider must be one of: " + ", ".join(_SUPPORTED_PROVIDERS) + ) + if args.repetitions < 1: + parser.error("--repetitions must be at least 1") + if args.max_tokens < 2: + parser.error("--max-tokens must be at least 2") + try: + context_turns = _parse_context_turns(args.context_turns) + except ValueError as exc: + parser.error(str(exc)) + + base_url = args.base_url or _PROVIDER_BASE_URLS.get(provider) + if not base_url: + parser.error("--base-url is required for openai-compatible providers") + base_url = base_url.rstrip("/") + model = args.model or _discover_model(base_url, args.api_key, args.timeout) + + result = run_benchmark( + provider=provider, + base_url=base_url, + model=model, + api_key=args.api_key, + timeout=args.timeout, + max_tokens=args.max_tokens, + repetitions=args.repetitions, + context_turns=context_turns, + pid=args.pid, + ) + rendered = json.dumps(result, indent=2, sort_keys=True) + if args.output: + with open(args.output, "w", encoding="utf-8") as handle: + handle.write(rendered + "\n") + if args.json: + print(rendered) + else: + _print_summary(result) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) \ No newline at end of file diff --git a/modelito/local_runtime.py b/modelito/local_runtime.py new file mode 100644 index 0000000..ceb25d7 --- /dev/null +++ b/modelito/local_runtime.py @@ -0,0 +1,401 @@ +"""Local runtime selection helpers. + +Profiles express deployment intent without claiming that one local backend is +universally fastest. ``portable`` uses Ollama as the common cross-platform +path. ``mac-performance`` uses Apple-Silicon-oriented backends before Ollama. +Applications should benchmark representative workloads and may override the +candidate order with ``prefer=``. +""" + +from __future__ import annotations + +import os +import platform +from dataclasses import dataclass +from typing import Any, List, Literal, Mapping, Optional, Sequence + +from .probes import ( + ProviderStatus, + probe_basert_status, + probe_ollama_status, + probe_omlx_status, + probe_vllm_mlx_status, +) + +LOCAL_PROFILE_AUTO = "auto" +LOCAL_PROFILE_PORTABLE = "portable" +LOCAL_PROFILE_MAC_PERFORMANCE = "mac-performance" +LOCAL_PROFILES = ( + LOCAL_PROFILE_AUTO, + LOCAL_PROFILE_PORTABLE, + LOCAL_PROFILE_MAC_PERFORMANCE, +) +LOCAL_PROVIDERS = ("basert", "vllm-mlx", "omlx", "ollama") + +CapabilityState = Literal["yes", "no", "conditional", "unknown"] + +_ALIASES = { + "mac": LOCAL_PROFILE_MAC_PERFORMANCE, + "macos": LOCAL_PROFILE_MAC_PERFORMANCE, + "apple": LOCAL_PROFILE_MAC_PERFORMANCE, + "apple-silicon": LOCAL_PROFILE_MAC_PERFORMANCE, + "apple_silicon": LOCAL_PROFILE_MAC_PERFORMANCE, + "cross-platform": LOCAL_PROFILE_PORTABLE, + "cross_platform": LOCAL_PROFILE_PORTABLE, +} + +_PROVIDER_ALIASES = { + "om": "omlx", + "vllm_mlx": "vllm-mlx", + "vllmmlx": "vllm-mlx", +} + + +@dataclass(frozen=True) +class LocalRuntimeSelection: + """Resolved local runtime and model.""" + + profile: str + provider: str + model: Optional[str] + endpoint: Optional[str] = None + + +@dataclass(frozen=True) +class LocalRuntimeCapabilities: + """Coarse upstream capabilities relevant to local runtime selection. + + Values describe the runtime family, not every model/configuration. A + ``conditional`` capability depends on model, parser, packaging, or runtime + configuration. ``unknown`` means Modelito deliberately does not claim the + capability from the upstream evidence currently recorded in the project. + """ + + streaming: CapabilityState + prefix_cache: CapabilityState + cancellation: CapabilityState + structured_output: CapabilityState + tool_calls: CapabilityState + model_discovery: CapabilityState + notes: str = "" + + +_LOCAL_RUNTIME_CAPABILITIES = { + "basert": LocalRuntimeCapabilities( + streaming="yes", + prefix_cache="yes", + cancellation="unknown", + structured_output="unknown", + tool_calls="yes", + model_discovery="yes", + notes="Native Metal runtime; exact features depend on the served model and BaseRT release.", + ), + "vllm-mlx": LocalRuntimeCapabilities( + streaming="yes", + prefix_cache="yes", + cancellation="yes", + structured_output="yes", + tool_calls="conditional", + model_discovery="yes", + notes="Tool calling depends on model support and server parser/configuration.", + ), + "omlx": LocalRuntimeCapabilities( + streaming="yes", + prefix_cache="yes", + cancellation="yes", + structured_output="conditional", + tool_calls="conditional", + model_discovery="yes", + notes="Structured output and tool calling can depend on installation and model/chat-template support.", + ), + "ollama": LocalRuntimeCapabilities( + streaming="yes", + prefix_cache="yes", + cancellation="unknown", + structured_output="yes", + tool_calls="conditional", + model_discovery="yes", + notes="Model support and engine choice vary by Ollama release and model format.", + ), +} + + +def is_macos_apple_silicon() -> bool: + """Return whether the current host is macOS on Apple Silicon.""" + + try: + return platform.system() == "Darwin" and platform.machine().lower() in { + "arm64", + "aarch64", + } + except Exception: + return False + + +def normalize_local_profile(profile: Optional[str]) -> str: + """Return the canonical local-runtime profile name. + + ``None`` resolves from ``MODELITO_LOCAL_PROFILE`` and then to ``auto``. + Unknown names fail explicitly rather than silently changing provider + selection behaviour. + """ + + raw = profile if profile is not None else os.getenv("MODELITO_LOCAL_PROFILE") + value = str(raw or LOCAL_PROFILE_AUTO).strip().lower() + value = _ALIASES.get(value, value) + if value not in LOCAL_PROFILES: + allowed = ", ".join(LOCAL_PROFILES) + raise ValueError( + f"Unknown local runtime profile: {profile!r}. Expected one of: {allowed}" + ) + return value + + +def local_provider_candidates( + profile: Optional[str], *, is_macos_apple_silicon: bool +) -> List[str]: + """Return the ordered local providers for *profile*. + + ``mac-performance`` is intentionally restricted to Apple Silicon. The + order is a starting policy, not a benchmark result for an arbitrary model, + machine, or workload. Use ``prefer=`` after measuring the target workload. + """ + + normalized = normalize_local_profile(profile) + if normalized == LOCAL_PROFILE_AUTO: + normalized = ( + LOCAL_PROFILE_MAC_PERFORMANCE + if is_macos_apple_silicon + else LOCAL_PROFILE_PORTABLE + ) + + if normalized == LOCAL_PROFILE_PORTABLE: + return ["ollama"] + + if not is_macos_apple_silicon: + raise ValueError( + "The mac-performance local runtime profile requires macOS on Apple Silicon. " + "Use profile='portable' on other platforms." + ) + return ["basert", "vllm-mlx", "omlx", "ollama"] + + +def _normalize_provider(name: str) -> str: + value = str(name or "").strip().lower() + return _PROVIDER_ALIASES.get(value, value) + + +def local_runtime_capabilities(provider: str) -> LocalRuntimeCapabilities: + """Return coarse capability metadata for a supported local runtime. + + The metadata is intentionally conservative and does not replace a runtime + probe or an application-specific feature test. + """ + + normalized = _normalize_provider(provider) + try: + return _LOCAL_RUNTIME_CAPABILITIES[normalized] + except KeyError as exc: + allowed = ", ".join(LOCAL_PROVIDERS) + raise ValueError( + f"Unknown local runtime provider: {provider!r}. Expected one of: {allowed}" + ) from exc + + +def _ordered_candidates( + profile: str, + prefer: Optional[Sequence[str]], + *, + on_macos_apple_silicon: bool, +) -> List[str]: + defaults: List[str] = local_provider_candidates( + profile, is_macos_apple_silicon=on_macos_apple_silicon + ) + requested: List[str] = [_normalize_provider(x) for x in (prefer or [])] + unknown = [x for x in requested if x not in LOCAL_PROVIDERS] + if unknown: + raise ValueError( + "Local runtime preference can only contain basert, vllm-mlx, omlx or " + "ollama; got: " + ", ".join(unknown) + ) + apple_only = [x for x in requested if x in {"basert", "vllm-mlx", "omlx"}] + if not on_macos_apple_silicon and apple_only: + raise ValueError("BaseRT, vllm-mlx and oMLX require macOS on Apple Silicon") + + ordered: List[str] = [] + for candidate in requested + defaults: + if candidate not in ordered: + ordered.append(candidate) + return ordered + + +def _model_for_provider( + provider: str, + model: Optional[str], + models: Optional[Mapping[str, str]], +) -> Optional[str]: + if not models: + return model + normalized = {_normalize_provider(k): v for k, v in models.items()} + return normalized.get(provider, model) + + +def _mapped_value( + provider: str, + default: Optional[str], + mapping: Optional[Mapping[str, str]], +) -> Optional[str]: + if not mapping: + return default + normalized = {_normalize_provider(k): v for k, v in mapping.items()} + return normalized.get(provider, default) + + +def _probe_local_provider( + provider: str, + model: Optional[str], + *, + host: Optional[str], + port: Optional[int], + base_url: Optional[str], + api_key: Optional[str], + probe_timeout: float, +) -> ProviderStatus: + if provider == "basert": + return probe_basert_status(model, base_url, api_key, probe_timeout) + if provider == "vllm-mlx": + return probe_vllm_mlx_status(model, base_url, api_key, probe_timeout) + if provider == "omlx": + return probe_omlx_status(model, base_url, api_key, probe_timeout) + if provider == "ollama": + return probe_ollama_status(model, host, port, probe_timeout) + raise ValueError(f"Unsupported local provider: {provider}") + + +def select_local_runtime( + model: Optional[str] = None, + *, + models: Optional[Mapping[str, str]] = None, + profile: Optional[str] = None, + prefer: Optional[Sequence[str]] = None, + host: Optional[str] = None, + port: Optional[int] = None, + base_url: Optional[str] = None, + base_urls: Optional[Mapping[str, str]] = None, + api_key: Optional[str] = None, + api_keys: Optional[Mapping[str, str]] = None, + probe_timeout: float = 1.5, + _is_macos_apple_silicon: Optional[bool] = None, +) -> LocalRuntimeSelection: + """Select a ready local runtime without falling back to a hosted provider. + + ``models`` may map provider names to provider-specific model identifiers. + ``base_urls`` and ``api_keys`` provide the same per-provider distinction for + OpenAI-compatible local servers. When no model is requested, the selector + binds to the first model reported by a ready local server; a server with no + loaded models is not considered usable. + """ + + resolved_profile = normalize_local_profile(profile) + on_mac = ( + is_macos_apple_silicon() + if _is_macos_apple_silicon is None + else bool(_is_macos_apple_silicon) + ) + candidates: List[str] = _ordered_candidates( + resolved_profile, prefer, on_macos_apple_silicon=on_mac + ) + diagnostics: List[str] = [] + + for provider in candidates: + candidate_model = _model_for_provider(provider, model, models) + candidate_base_url = _mapped_value(provider, base_url, base_urls) + candidate_api_key = _mapped_value(provider, api_key, api_keys) + status = _probe_local_provider( + provider, + candidate_model, + host=host, + port=port, + base_url=candidate_base_url, + api_key=candidate_api_key, + probe_timeout=probe_timeout, + ) + if status.ready: + resolved_model = candidate_model + if not resolved_model: + resolved_model = status.models[0] if status.models else None + if resolved_model: + return LocalRuntimeSelection( + profile=resolved_profile, + provider=provider, + model=resolved_model, + endpoint=status.endpoint, + ) + diagnostics.append( + f"{provider} at {status.endpoint or 'unknown endpoint'}: no loaded models" + ) + continue + reason = status.reason or "not ready" + endpoint = status.endpoint or "unknown endpoint" + diagnostics.append(f"{provider} at {endpoint}: {reason}") + + detail = "; ".join(diagnostics) if diagnostics else "no local providers probed" + raise ValueError( + f"No ready local runtime for profile '{resolved_profile}'. {detail}. " + "Start a local backend and ensure its requested model is available." + ) + + +def local_client( + model: Optional[str] = None, + *, + models: Optional[Mapping[str, str]] = None, + profile: Optional[str] = None, + prefer: Optional[Sequence[str]] = None, + host: Optional[str] = None, + port: Optional[int] = None, + base_url: Optional[str] = None, + base_urls: Optional[Mapping[str, str]] = None, + api_key: Optional[str] = None, + api_keys: Optional[Mapping[str, str]] = None, + probe_timeout: float = 1.5, + **provider_kwargs: Any, +): + """Return a :class:`modelito.Client` bound to a ready local runtime. + + The returned provider is strict by default so a runtime failure is not + silently replaced by Modelito's deterministic offline fallback. + """ + + selection = select_local_runtime( + model, + models=models, + profile=profile, + prefer=prefer, + host=host, + port=port, + base_url=base_url, + base_urls=base_urls, + api_key=api_key, + api_keys=api_keys, + probe_timeout=probe_timeout, + ) + + from .client import Client + + kwargs = dict(provider_kwargs) + kwargs.setdefault("strict", True) + if selection.provider in {"basert", "vllm-mlx", "omlx"}: + selected_base_url = _mapped_value(selection.provider, base_url, base_urls) + selected_api_key = _mapped_value(selection.provider, api_key, api_keys) + if selected_base_url is not None: + kwargs["base_url"] = selected_base_url + if selected_api_key is not None: + kwargs["api_key"] = selected_api_key + else: + if host is not None: + kwargs["host"] = host + if port is not None: + kwargs["port"] = port + + return Client(provider=selection.provider, model=selection.model, **kwargs) \ No newline at end of file diff --git a/modelito/ollama_strict.py b/modelito/ollama_strict.py new file mode 100644 index 0000000..046194c --- /dev/null +++ b/modelito/ollama_strict.py @@ -0,0 +1,86 @@ +"""Strict-aware Ollama provider surface. + +The historical :mod:`modelito.ollama` provider intentionally keeps deterministic +fallbacks for offline-friendly use. Local-runtime profiles have a stronger +contract: when ``strict=True`` a failed local model request must fail rather than +silently returning the prompt. This subclass preserves the legacy non-strict +behaviour while routing strict chat through Ollama's OpenAI-compatible raw +surfaces, which already classify and raise transport/provider errors. +""" + +from __future__ import annotations + +from typing import Any, Iterable, Optional + +from .exceptions import ModelitoBadResponseError +from .messages import flatten_message_inputs +from .ollama import OllamaProvider as _LegacyOllamaProvider +from .provider import MessageInput + + +class OllamaProvider(_LegacyOllamaProvider): + """Ollama provider that enforces ``strict=True`` for summary and streaming.""" + + @staticmethod + def _strict_messages(messages: Iterable[MessageInput]) -> list[dict[str, Any]]: + return flatten_message_inputs(messages) + + def summarize( + self, + messages: Iterable[MessageInput], + settings: Optional[dict[str, Any]] = None, + ) -> str: + if not self.strict: + return super().summarize(messages, settings=settings) + + payload: dict[str, Any] = {"messages": self._strict_messages(messages)} + if settings: + payload.update(settings) + response = self.raw_complete(payload) + choices = response.get("choices") + if not isinstance(choices, list) or not choices: + raise ModelitoBadResponseError( + "Ollama strict summarize returned no completion choices" + ) + first = choices[0] + if not isinstance(first, dict): + raise ModelitoBadResponseError( + "Ollama strict summarize returned an invalid completion choice" + ) + message = first.get("message") + if isinstance(message, dict) and isinstance(message.get("content"), str): + return str(message["content"]) + if isinstance(first.get("text"), str): + return str(first["text"]) + raise ModelitoBadResponseError( + "Ollama strict summarize returned no textual completion" + ) + + def stream( + self, + messages: Iterable[MessageInput], + settings: Optional[dict[str, Any]] = None, + ) -> Iterable[str]: + if not self.strict: + yield from super().stream(messages, settings=settings) + return + + payload: dict[str, Any] = {"messages": self._strict_messages(messages)} + if settings: + payload.update(settings) + for event in self.raw_stream(payload): + choices = event.get("choices") if isinstance(event, dict) else None + if not isinstance(choices, list) or not choices: + continue + first = choices[0] + if not isinstance(first, dict): + continue + delta = first.get("delta") + if isinstance(delta, dict) and isinstance(delta.get("content"), str): + content = str(delta["content"]) + if content: + yield content + continue + text = first.get("text") + if isinstance(text, str) and text: + yield text diff --git a/modelito/probes.py b/modelito/probes.py index fd1dfbf..f88857d 100644 --- a/modelito/probes.py +++ b/modelito/probes.py @@ -1,7 +1,7 @@ """Shared provider readiness probes. These helpers are used by both the client auto-selection path and the doctor -diagnostics so the readiness behaviour stays aligned. +diagnostics so readiness behaviour stays aligned. """ from __future__ import annotations @@ -9,8 +9,10 @@ from dataclasses import dataclass, field from typing import Any, Dict, Iterable, List, Optional +from .basert import BaseRTProvider from .ollama_service import DEFAULT_PORT, DEFAULT_URL, list_local_models, server_is_up from .omlx import OMLXProvider +from .vllm_mlx import VLLMMLXProvider @dataclass(frozen=True) @@ -59,6 +61,92 @@ def build_status( _build_status = build_status +def probe_basert_status( + model: Optional[str], + base_url: Optional[str], + api_key: Optional[str], + probe_timeout: float, +) -> ProviderStatus: + endpoint = base_url or "http://127.0.0.1:8080/v1" + try: + provider = BaseRTProvider( + base_url=endpoint, + model=model, + api_key=api_key, + timeout=probe_timeout, + strict=True, + ) + models = provider.list_models() + ready = _model_is_available(model, models) + return _build_status( + "basert", + ready, + endpoint=getattr(provider, "base_url", endpoint), + models=models, + reason="" if ready else "requested model not found", + setup_hint=( + "Start BaseRT with `basert serve --port 8080` and use the " + "same API key in Modelito if `--api-key` is enabled." + ), + ) + except Exception as exc: + return _build_status( + "basert", + False, + endpoint=endpoint, + models=[], + reason="BaseRT server not reachable", + setup_hint=( + "Start BaseRT with `basert serve --port 8080` and use the " + "same API key in Modelito if `--api-key` is enabled." + ), + details={"error": str(exc)}, + ) + + +def probe_vllm_mlx_status( + model: Optional[str], + base_url: Optional[str], + api_key: Optional[str], + probe_timeout: float, +) -> ProviderStatus: + endpoint = base_url or "http://localhost:8000/v1" + try: + provider = VLLMMLXProvider( + base_url=endpoint, + model=model, + api_key=api_key, + timeout=probe_timeout, + strict=True, + ) + models = provider.list_models() + ready = _model_is_available(model, models) + return _build_status( + "vllm-mlx", + ready, + endpoint=getattr(provider, "base_url", endpoint), + models=models, + reason="" if ready else "requested model not found", + setup_hint=( + "Start vllm-mlx with `vllm-mlx serve --port 8000` and use " + "the same API key in Modelito if `--api-key` is enabled." + ), + ) + except Exception as exc: + return _build_status( + "vllm-mlx", + False, + endpoint=endpoint, + models=[], + reason="vllm-mlx server not reachable", + setup_hint=( + "Start vllm-mlx with `vllm-mlx serve --port 8000` and use " + "the same API key in Modelito if `--api-key` is enabled." + ), + details={"error": str(exc)}, + ) + + def probe_omlx_status( model: Optional[str], base_url: Optional[str], @@ -82,7 +170,7 @@ def probe_omlx_status( endpoint=getattr(provider, "base_url", endpoint), models=models, reason="" if ready else "requested model not found", - setup_hint="Start oMLX and download an MLX model via the admin dashboard.", + setup_hint="Start oMLX and make the requested model available to the server.", ) except Exception as exc: return _build_status( @@ -91,7 +179,7 @@ def probe_omlx_status( endpoint=endpoint, models=[], reason="oMLX server not reachable", - setup_hint="Start oMLX and download an MLX model via the admin dashboard.", + setup_hint="Start oMLX and make the requested model available to the server.", details={"error": str(exc)}, ) @@ -100,7 +188,7 @@ def probe_ollama_status( model: Optional[str], host: Optional[str], port: Optional[int], - probe_timeout: float, # kept for API symmetry with probe_omlx_status; server_is_up has no timeout + probe_timeout: float, # kept for API symmetry; server_is_up has no timeout ) -> ProviderStatus: _ = probe_timeout host_value = host or DEFAULT_URL @@ -114,7 +202,10 @@ def probe_ollama_status( endpoint=endpoint, models=[], reason="Ollama server not reachable", - setup_hint="Start Ollama and pull the requested model with `ollama pull `.", + setup_hint=( + "Start Ollama and pull the requested model with " + "`ollama pull `." + ), ) models = list_local_models() @@ -134,6 +225,9 @@ def probe_ollama_status( endpoint=endpoint, models=[], reason="Ollama probe failed", - setup_hint="Start Ollama and pull the requested model with `ollama pull `.", + setup_hint=( + "Start Ollama and pull the requested model with " + "`ollama pull `." + ), details={"error": str(exc)}, - ) + ) \ No newline at end of file diff --git a/modelito/provider_registry.py b/modelito/provider_registry.py index 4d5ba83..477eaf3 100644 --- a/modelito/provider_registry.py +++ b/modelito/provider_registry.py @@ -1,16 +1,18 @@ """Provider and embedder registry helpers for Modelito.""" -from inspect import signature, Parameter +from inspect import Parameter, signature from typing import Any, Dict, List, Optional, Type -from .provider import EmbeddingProvider, SyncProvider -from .openai import OpenAIProvider + +from .basert import BaseRTProvider from .claude import ClaudeProvider from .gemini import GeminiProvider -from .ollama import OllamaProvider +from .mock_provider import MockProvider +from .ollama_strict import OllamaProvider from .omlx import OMLXProvider +from .openai import OpenAIProvider +from .provider import EmbeddingProvider, SyncProvider +from .vllm_mlx import VLLMMLXProvider -# Registry of provider classes -from .mock_provider import MockProvider PROVIDER_REGISTRY: Dict[str, Type] = { "openai": OpenAIProvider, @@ -21,6 +23,9 @@ "ollama": OllamaProvider, "omlx": OMLXProvider, "om": OMLXProvider, + "basert": BaseRTProvider, + "vllm-mlx": VLLMMLXProvider, + "vllm_mlx": VLLMMLXProvider, "mock": MockProvider, } @@ -28,14 +33,7 @@ def get_provider(name: str, **kwargs: Any) -> Optional[SyncProvider]: - """ - Factory to instantiate a provider by name. - Args: - name: Provider name (e.g., 'openai', 'claude', 'gemini', 'ollama') - kwargs: Passed to provider constructor - Returns: - Provider instance or None if not found - """ + """Instantiate a provider by name, or return ``None`` when unknown.""" cls = PROVIDER_REGISTRY.get(name.lower()) if cls is not None: try: @@ -57,7 +55,7 @@ def get_provider(name: str, **kwargs: Any) -> Optional[SyncProvider]: def get_embedder(name: str, **kwargs: Any) -> Optional[EmbeddingProvider]: - """Factory to instantiate an embedder by name.""" + """Instantiate an embedder by name, or return ``None`` when unknown.""" cls = EMBEDDER_REGISTRY.get(name.lower()) if cls is not None: try: @@ -79,10 +77,10 @@ def get_embedder(name: str, **kwargs: Any) -> Optional[EmbeddingProvider]: def list_providers() -> List[str]: - """Return a list of available provider names.""" + """Return the available provider names.""" return list(PROVIDER_REGISTRY.keys()) def list_embedders() -> List[str]: - """Return a list of available embedder names.""" + """Return the available embedder names.""" return list(EMBEDDER_REGISTRY.keys()) diff --git a/modelito/vllm_mlx.py b/modelito/vllm_mlx.py new file mode 100644 index 0000000..36cef51 --- /dev/null +++ b/modelito/vllm_mlx.py @@ -0,0 +1,44 @@ +"""vllm-mlx provider with OpenAI-compatible local runtime integration. + +vllm-mlx exposes an OpenAI-compatible HTTP server on Apple Silicon. This +provider reuses Modelito's shared OpenAI-compatible transport and only supplies +vllm-mlx-specific defaults. +""" + +from __future__ import annotations + +from typing import Optional + +from .openai_compat import OpenAICompatibleHTTPProvider + + +class VLLMMLXProvider(OpenAICompatibleHTTPProvider): + """Provider for a local ``vllm-mlx serve`` endpoint. + + Args: + base_url: Base URL of the vllm-mlx OpenAI-compatible endpoint. + Defaults to ``http://localhost:8000/v1``. + model: Served model identifier. ``"default"`` is used only when a + caller constructs the provider directly without specifying one; + runtime selection discovers the actual model through ``/models``. + api_key: Optional bearer token matching ``vllm-mlx serve --api-key``. + timeout: HTTP timeout in seconds. + strict: When ``True``, raise typed Modelito errors instead of falling + back to deterministic offline behaviour. + """ + + def __init__( + self, + base_url: Optional[str] = None, + model: Optional[str] = None, + api_key: Optional[str] = None, + timeout: float = 20.0, + strict: bool = False, + ) -> None: + super().__init__( + base_url=base_url or "http://localhost:8000/v1", + model=model or "default", + api_key=api_key, + timeout=timeout, + strict=strict, + ) diff --git a/pyproject.toml b/pyproject.toml index 0b9eb62..86e3230 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,13 +5,13 @@ build-backend = "setuptools.build_meta" [project] name = "modelito" version = "1.4.6" -description = "Lightweight Python abstractions and connectors for LLM providers (OpenAI, Claude, Gemini, oMLX, Ollama)." +description = "Lightweight Python abstractions and connectors for hosted and local LLM providers." readme = "README.md" requires-python = ">=3.10" authors = [ { name = "krahd" } ] license = "MIT" license-files = ["LICENSE"] -keywords = ["llm", "provider", "tokenizer", "ollama", "openai", "omlx"] +keywords = ["llm", "provider", "tokenizer", "ollama", "openai", "omlx", "local-inference"] classifiers = [ "Development Status :: 4 - Beta", "Intended Audience :: Developers", @@ -38,6 +38,7 @@ dev = ["pytest", "pytest-asyncio", "pytest-mock", "mypy", "ruff", "black", "toml [project.scripts] modelito-serve = "modelito.serve:main" modelito-doctor = "modelito.doctor:main" +modelito-benchmark-local = "modelito.local_benchmark:main" [project.urls] Homepage = "https://github.com/krahd/modelito" @@ -47,3 +48,9 @@ Changelog = "https://github.com/krahd/modelito/blob/main/CHANGELOG.md" [tool.setuptools.packages.find] where = ["."] include = ["modelito*"] + +# Ruff 0.16 expanded its unconfigured default from 59 to 413 rules. Modelito's +# existing lint contract predates that change; make it explicit so CI does not +# change semantics when the unpinned development dependency advances. +[tool.ruff.lint] +select = ["E4", "E7", "E9", "F"] diff --git a/tests/test_basert.py b/tests/test_basert.py new file mode 100644 index 0000000..aa4909d --- /dev/null +++ b/tests/test_basert.py @@ -0,0 +1,25 @@ +from modelito.basert import BaseRTProvider +from modelito.provider_registry import get_provider, list_providers + + +def test_basert_defaults_to_local_openai_compatible_endpoint(): + provider = BaseRTProvider(model="Qwen/Qwen3-4B", strict=True) + + assert provider.base_url == "http://127.0.0.1:8080/v1" + assert provider.model == "Qwen/Qwen3-4B" + assert provider.strict is True + + +def test_basert_is_registered_provider(): + provider = get_provider( + "basert", + model="Qwen/Qwen3-4B", + base_url="http://127.0.0.1:9000/v1", + api_key="test-key", + strict=True, + ) + + assert "basert" in list_providers() + assert isinstance(provider, BaseRTProvider) + assert provider.base_url == "http://127.0.0.1:9000/v1" + assert provider.api_key == "test-key" diff --git a/tests/test_doctor.py b/tests/test_doctor.py index f411e9b..81ea594 100644 --- a/tests/test_doctor.py +++ b/tests/test_doctor.py @@ -27,6 +27,40 @@ def test_check_provider_ready_omlx_success(monkeypatch): assert "omlx" in status.models +def test_check_provider_ready_vllm_mlx_alias(monkeypatch): + monkeypatch.setattr( + "modelito.probes.probe_vllm_mlx_status", + lambda *args, **kwargs: ProviderStatus( + provider="vllm-mlx", + ready=True, + endpoint="http://localhost:8000/v1", + models=["mlx-model"], + ), + ) + + status = check_provider_ready("vllm_mlx", model="mlx-model") + + assert status.ready is True + assert status.provider == "vllm-mlx" + + +def test_check_provider_ready_basert_success(monkeypatch): + monkeypatch.setattr( + "modelito.probes.probe_basert_status", + lambda *args, **kwargs: ProviderStatus( + provider="basert", + ready=True, + endpoint="http://127.0.0.1:8080/v1", + models=["base-model"], + ), + ) + + status = check_provider_ready("basert", model="base-model") + + assert status.ready is True + assert status.provider == "basert" + + def test_check_provider_ready_ollama_failure(monkeypatch): monkeypatch.setattr( "modelito.probes.probe_ollama_status", @@ -49,25 +83,63 @@ def test_check_provider_ready_ollama_failure(monkeypatch): assert "ollama pull" in status.setup_hint -def test_check_provider_ready_auto_prefers_omlx_on_macos(monkeypatch): +def test_check_provider_ready_auto_follows_mac_runtime_order(monkeypatch): monkeypatch.setattr("modelito.doctor._is_macos_apple_silicon", lambda: True) - monkeypatch.setattr( - "modelito.doctor._probe_omlx", - lambda model, base_url, api_key, probe_timeout: ProviderStatus( - provider="omlx", + seen = [] + + def basert_probe(*args, **kwargs): + seen.append("basert") + return ProviderStatus(provider="basert", ready=False) + + def vllm_probe(*args, **kwargs): + seen.append("vllm-mlx") + return ProviderStatus( + provider="vllm-mlx", ready=True, endpoint="http://localhost:8000/v1", - models=["omlx"], + models=["mlx-model"], + ) + + monkeypatch.setattr("modelito.doctor._probe_basert", basert_probe) + monkeypatch.setattr("modelito.doctor._probe_vllm_mlx", vllm_probe) + monkeypatch.setattr( + "modelito.doctor._probe_omlx", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected oMLX probe") ), ) monkeypatch.setattr( "modelito.doctor._probe_ollama", - lambda *args, **kwargs: ProviderStatus(provider="ollama", ready=False), + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected Ollama probe") + ), ) - status = check_provider_ready("auto", model="omlx") + status = check_provider_ready("auto", model="mlx-model") - assert status.provider == "omlx" + assert status.provider == "vllm-mlx" + assert status.ready is True + assert seen == ["basert", "vllm-mlx"] + + +def test_check_provider_ready_auto_non_mac_only_probes_ollama(monkeypatch): + monkeypatch.setattr("modelito.doctor._is_macos_apple_silicon", lambda: False) + monkeypatch.setattr( + "modelito.doctor._probe_ollama", + lambda *args, **kwargs: ProviderStatus( + provider="ollama", ready=True, models=["local"] + ), + ) + monkeypatch.setattr( + "modelito.doctor._probe_basert", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected BaseRT probe") + ), + ) + + status = check_provider_ready("auto", model="local") + + assert status.provider == "ollama" assert status.ready is True diff --git a/tests/test_local_benchmark.py b/tests/test_local_benchmark.py new file mode 100644 index 0000000..d9e6247 --- /dev/null +++ b/tests/test_local_benchmark.py @@ -0,0 +1,54 @@ +import pytest + +from modelito.local_benchmark import ( + StreamMetrics, + _extract_stream_text, + _is_useful_phrase, + _normalise_provider, + _parse_context_turns, + _summarise_samples, +) + + +def test_extract_stream_text_reads_openai_delta(): + event = {"choices": [{"delta": {"content": "hello"}}]} + + assert _extract_stream_text(event) == "hello" + + +def test_extract_stream_text_ignores_non_text_events(): + assert _extract_stream_text({"choices": [{"delta": {"role": "assistant"}}]}) == "" + assert _extract_stream_text({}) == "" + + +def test_useful_phrase_heuristic(): + assert _is_useful_phrase("Hello.") is True + assert _is_useful_phrase("this is a sufficiently long fragment") is True + assert _is_useful_phrase("short fragment") is False + + +def test_provider_aliases_are_normalised(): + assert _normalise_provider("vllm_mlx") == "vllm-mlx" + assert _normalise_provider("mlx_lm") == "mlx-lm" + assert _normalise_provider("generic") == "openai-compatible" + + +def test_context_turn_parser_deduplicates_and_validates(): + assert _parse_context_turns("1,2,2,4") == [1, 2, 4] + with pytest.raises(ValueError, match="positive"): + _parse_context_turns("0,2") + + +def test_sample_summary_uses_medians(): + samples = [ + StreamMetrics(100.0, 150.0, 500.0, 20, 5, 10.0), + StreamMetrics(200.0, 250.0, 600.0, 20, 5, 20.0), + StreamMetrics(300.0, None, 700.0, 20, 5, 30.0), + ] + + summary = _summarise_samples(samples) + + assert summary["median_ttft_ms"] == 200.0 + assert summary["median_first_useful_ms"] == 200.0 + assert summary["median_estimated_decode_tokens_per_s"] == 20.0 + assert len(summary["samples"]) == 3 diff --git a/tests/test_local_runtime.py b/tests/test_local_runtime.py new file mode 100644 index 0000000..a2a9425 --- /dev/null +++ b/tests/test_local_runtime.py @@ -0,0 +1,371 @@ +import pytest + +from modelito.local_runtime import ( + LOCAL_PROFILE_MAC_PERFORMANCE, + LOCAL_PROFILE_PORTABLE, + LocalRuntimeSelection, + local_client, + local_provider_candidates, + local_runtime_capabilities, + normalize_local_profile, + select_local_runtime, +) +from modelito.probes import ProviderStatus + + +def status(provider, ready, model=None, reason="", endpoint=None): + endpoints = { + "basert": "http://127.0.0.1:8080/v1", + "vllm-mlx": "http://localhost:8000/v1", + "omlx": "http://localhost:8000/v1", + "ollama": "http://127.0.0.1:11434", + } + return ProviderStatus( + provider=provider, + ready=ready, + endpoint=endpoint or endpoints[provider], + models=[model] if ready and model else [], + reason=reason, + setup_hint="", + ) + + +def test_portable_profile_is_ollama_everywhere(): + assert local_provider_candidates( + "portable", is_macos_apple_silicon=False + ) == ["ollama"] + assert local_provider_candidates( + "portable", is_macos_apple_silicon=True + ) == ["ollama"] + + +def test_auto_profile_uses_mac_native_order_on_apple_silicon(): + assert local_provider_candidates("auto", is_macos_apple_silicon=True) == [ + "basert", + "vllm-mlx", + "omlx", + "ollama", + ] + assert local_provider_candidates("auto", is_macos_apple_silicon=False) == [ + "ollama" + ] + + +def test_mac_performance_profile_rejects_non_apple_silicon(): + with pytest.raises(ValueError, match="requires macOS on Apple Silicon"): + local_provider_candidates( + "mac-performance", is_macos_apple_silicon=False + ) + + +def test_profile_alias_and_environment(monkeypatch): + assert normalize_local_profile("mac") == LOCAL_PROFILE_MAC_PERFORMANCE + monkeypatch.setenv("MODELITO_LOCAL_PROFILE", "cross-platform") + assert normalize_local_profile(None) == LOCAL_PROFILE_PORTABLE + + +def test_capabilities_cover_vllm_mlx_alias(): + caps = local_runtime_capabilities("vllm_mlx") + + assert caps.streaming == "yes" + assert caps.prefix_cache == "yes" + assert caps.cancellation == "yes" + assert caps.model_discovery == "yes" + + +def test_capabilities_reject_unknown_provider(): + with pytest.raises(ValueError, match="Unknown local runtime provider"): + local_runtime_capabilities("unknown") + + +def test_select_portable_does_not_probe_mac_only_backends(monkeypatch): + for probe in ( + "probe_basert_status", + "probe_vllm_mlx_status", + "probe_omlx_status", + ): + monkeypatch.setattr( + f"modelito.local_runtime.{probe}", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected Apple-only probe") + ), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda model, host, port, timeout: status("ollama", True, model), + ) + + selection = select_local_runtime( + "gemma4:12b-mlx", + profile="portable", + _is_macos_apple_silicon=True, + ) + + assert selection.provider == "ollama" + assert selection.model == "gemma4:12b-mlx" + + +def test_select_mac_performance_prefers_basert(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_basert_status", + lambda model, base_url, api_key, timeout: status("basert", True, model), + ) + for probe in ( + "probe_vllm_mlx_status", + "probe_omlx_status", + "probe_ollama_status", + ): + monkeypatch.setattr( + f"modelito.local_runtime.{probe}", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected fallback probe") + ), + ) + + selection = select_local_runtime( + "base-model", + profile="mac-performance", + _is_macos_apple_silicon=True, + ) + + assert selection.provider == "basert" + + +def test_select_mac_performance_uses_vllm_mlx_after_basert(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_basert_status", + lambda model, base_url, api_key, timeout: status( + "basert", False, reason="not reachable" + ), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_vllm_mlx_status", + lambda model, base_url, api_key, timeout: status("vllm-mlx", True, model), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_omlx_status", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected oMLX probe") + ), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected Ollama probe") + ), + ) + + selection = select_local_runtime( + "mlx-model", + profile="mac-performance", + _is_macos_apple_silicon=True, + ) + + assert selection.provider == "vllm-mlx" + + +def test_select_mac_performance_falls_back_through_all_backends(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_basert_status", + lambda model, base_url, api_key, timeout: status( + "basert", False, reason="not reachable" + ), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_vllm_mlx_status", + lambda model, base_url, api_key, timeout: status( + "vllm-mlx", False, reason="not reachable" + ), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_omlx_status", + lambda model, base_url, api_key, timeout: status( + "omlx", False, reason="not reachable" + ), + ) + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda model, host, port, timeout: status("ollama", True, model), + ) + + selection = select_local_runtime( + profile="mac-performance", + models={ + "basert": "base-model", + "vllm-mlx": "vllm-model", + "omlx": "omlx-model", + "ollama": "ollama-model", + }, + _is_macos_apple_silicon=True, + ) + + assert selection.provider == "ollama" + assert selection.model == "ollama-model" + + +def test_provider_specific_model_mapping_is_used(monkeypatch): + seen = [] + + def basert_probe(model, base_url, api_key, timeout): + seen.append(("basert", model)) + return status("basert", False, reason="not ready") + + def vllm_probe(model, base_url, api_key, timeout): + seen.append(("vllm-mlx", model)) + return status("vllm-mlx", False, reason="not ready") + + def omlx_probe(model, base_url, api_key, timeout): + seen.append(("omlx", model)) + return status("omlx", False, reason="not ready") + + def ollama_probe(model, host, port, timeout): + seen.append(("ollama", model)) + return status("ollama", True, model) + + monkeypatch.setattr("modelito.local_runtime.probe_basert_status", basert_probe) + monkeypatch.setattr("modelito.local_runtime.probe_vllm_mlx_status", vllm_probe) + monkeypatch.setattr("modelito.local_runtime.probe_omlx_status", omlx_probe) + monkeypatch.setattr("modelito.local_runtime.probe_ollama_status", ollama_probe) + + selection = select_local_runtime( + model="fallback-model", + models={ + "basert": "base-specific", + "vllm_mlx": "vllm-specific", + "om": "omlx-specific", + "ollama": "ollama-specific", + }, + profile="mac-performance", + _is_macos_apple_silicon=True, + ) + + assert seen == [ + ("basert", "base-specific"), + ("vllm-mlx", "vllm-specific"), + ("omlx", "omlx-specific"), + ("ollama", "ollama-specific"), + ] + assert selection.model == "ollama-specific" + + +def test_provider_specific_base_url_and_api_key_mapping_is_used(monkeypatch): + seen = {} + + def basert_probe(model, base_url, api_key, timeout): + seen.update(base_url=base_url, api_key=api_key) + return status("basert", True, model, endpoint=base_url) + + monkeypatch.setattr("modelito.local_runtime.probe_basert_status", basert_probe) + + selection = select_local_runtime( + "base-model", + profile="mac-performance", + base_urls={"basert": "http://127.0.0.1:9000/v1"}, + api_keys={"basert": "local-secret"}, + _is_macos_apple_silicon=True, + ) + + assert selection.provider == "basert" + assert selection.endpoint == "http://127.0.0.1:9000/v1" + assert seen == { + "base_url": "http://127.0.0.1:9000/v1", + "api_key": "local-secret", + } + + +def test_prefer_can_reorder_mac_profile_after_benchmarking(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda model, host, port, timeout: status("ollama", True, model), + ) + for probe in ( + "probe_basert_status", + "probe_vllm_mlx_status", + "probe_omlx_status", + ): + monkeypatch.setattr( + f"modelito.local_runtime.{probe}", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("unexpected earlier probe") + ), + ) + + selection = select_local_runtime( + "ollama-model", + profile="mac-performance", + prefer=["ollama", "vllm-mlx", "basert", "omlx"], + _is_macos_apple_silicon=True, + ) + + assert selection.provider == "ollama" + + +def test_local_prefer_rejects_hosted_provider(): + with pytest.raises(ValueError, match="vllm-mlx"): + select_local_runtime( + profile="portable", + prefer=["openai"], + _is_macos_apple_silicon=False, + ) + + +def test_local_prefer_rejects_mac_only_backend_off_mac(): + with pytest.raises(ValueError, match="require macOS on Apple Silicon"): + select_local_runtime( + profile="portable", + prefer=["vllm-mlx"], + _is_macos_apple_silicon=False, + ) + + +def test_no_ready_local_runtime_raises_with_diagnostics(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda model, host, port, timeout: status( + "ollama", False, reason="requested model not found" + ), + ) + + with pytest.raises(ValueError, match="requested model not found"): + select_local_runtime( + "missing", + profile="portable", + _is_macos_apple_silicon=False, + ) + + +def test_local_client_uses_strict_provider_by_default(monkeypatch): + called = {} + + class DummyProvider: + model = "local-model" + + def list_models(self): + return [self.model] + + def summarize(self, messages, settings=None): + return "ok" + + monkeypatch.setattr( + "modelito.local_runtime.select_local_runtime", + lambda *args, **kwargs: LocalRuntimeSelection( + profile="portable", + provider="ollama", + model="local-model", + endpoint="http://127.0.0.1:11434", + ), + ) + + def fake_get_provider(name, **kwargs): + called["name"] = name + called["kwargs"] = kwargs + return DummyProvider() + + monkeypatch.setattr("modelito.client.get_provider", fake_get_provider) + + client = local_client(profile="portable") + + assert client.provider_name == "DummyProvider" + assert called["name"] == "ollama" + assert called["kwargs"]["strict"] is True + assert called["kwargs"]["model"] == "local-model" diff --git a/tests/test_local_runtime_edge_cases.py b/tests/test_local_runtime_edge_cases.py new file mode 100644 index 0000000..9fbcb43 --- /dev/null +++ b/tests/test_local_runtime_edge_cases.py @@ -0,0 +1,41 @@ +import pytest + +from modelito.local_runtime import select_local_runtime +from modelito.probes import ProviderStatus + + +def test_selector_uses_first_loaded_model_when_none_requested(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda model, host, port, timeout: ProviderStatus( + provider="ollama", + ready=True, + endpoint="http://127.0.0.1:11434", + models=["loaded-model"], + ), + ) + + selection = select_local_runtime( + profile="portable", + _is_macos_apple_silicon=False, + ) + + assert selection.model == "loaded-model" + + +def test_selector_rejects_reachable_server_without_loaded_model(monkeypatch): + monkeypatch.setattr( + "modelito.local_runtime.probe_ollama_status", + lambda model, host, port, timeout: ProviderStatus( + provider="ollama", + ready=True, + endpoint="http://127.0.0.1:11434", + models=[], + ), + ) + + with pytest.raises(ValueError, match="no loaded models"): + select_local_runtime( + profile="portable", + _is_macos_apple_silicon=False, + ) diff --git a/tests/test_ollama_strict.py b/tests/test_ollama_strict.py new file mode 100644 index 0000000..dd148c4 --- /dev/null +++ b/tests/test_ollama_strict.py @@ -0,0 +1,89 @@ +import pytest + +import modelito +from modelito.exceptions import ModelitoBadResponseError, ModelitoConnectionError +from modelito.ollama_strict import OllamaProvider + + +def test_package_root_exports_strict_aware_ollama_provider(): + assert modelito.OllamaProvider is OllamaProvider + + +def test_strict_summarize_uses_raw_completion(monkeypatch): + provider = OllamaProvider(model="local-model", strict=True) + seen = {} + + def raw_complete(payload): + seen.update(payload) + return {"choices": [{"message": {"content": "respuesta local"}}]} + + monkeypatch.setattr(provider, "raw_complete", raw_complete) + + result = provider.summarize([{"role": "user", "content": "hola"}]) + + assert result == "respuesta local" + assert seen["messages"] == [{"role": "user", "content": "hola"}] + + +def test_strict_summarize_propagates_runtime_failure(monkeypatch): + provider = OllamaProvider(model="local-model", strict=True) + + def fail(_payload): + raise ModelitoConnectionError("Ollama disappeared") + + monkeypatch.setattr(provider, "raw_complete", fail) + + with pytest.raises(ModelitoConnectionError, match="Ollama disappeared"): + provider.summarize([{"role": "user", "content": "hola"}]) + + +def test_strict_summarize_rejects_non_text_completion(monkeypatch): + provider = OllamaProvider(model="local-model", strict=True) + monkeypatch.setattr( + provider, + "raw_complete", + lambda _payload: {"choices": [{"message": {"content": None}}]}, + ) + + with pytest.raises(ModelitoBadResponseError, match="no textual completion"): + provider.summarize([{"role": "user", "content": "hola"}]) + + +def test_strict_stream_propagates_runtime_failure(monkeypatch): + provider = OllamaProvider(model="local-model", strict=True) + + def fail(_payload): + raise ModelitoConnectionError("stream failed") + yield # pragma: no cover + + monkeypatch.setattr(provider, "raw_stream", fail) + + with pytest.raises(ModelitoConnectionError, match="stream failed"): + list(provider.stream([{"role": "user", "content": "hola"}])) + + +def test_strict_stream_yields_only_text(monkeypatch): + provider = OllamaProvider(model="local-model", strict=True) + + def events(_payload): + yield {"choices": [{"delta": {"role": "assistant"}}]} + yield {"choices": [{"delta": {"content": "Buen"}}]} + yield {"choices": [{"delta": {"content": " día"}}]} + yield {"choices": [{"delta": {}, "finish_reason": "stop"}]} + + monkeypatch.setattr(provider, "raw_stream", events) + + assert list(provider.stream([{"role": "user", "content": "hola"}])) == [ + "Buen", + " día", + ] + + +def test_non_strict_summarize_preserves_legacy_fallback(monkeypatch): + provider = OllamaProvider(model="local-model", strict=False) + monkeypatch.setattr("modelito.ollama.server_is_up", lambda *_args: False) + monkeypatch.setattr("modelito.ollama.ollama_installed", lambda: False) + + assert provider.summarize([{"role": "user", "content": "sin servidor"}]) == ( + "sin servidor" + ) diff --git a/tests/test_vllm_mlx.py b/tests/test_vllm_mlx.py new file mode 100644 index 0000000..156cf23 --- /dev/null +++ b/tests/test_vllm_mlx.py @@ -0,0 +1,31 @@ +from modelito.provider_registry import get_provider, list_providers +from modelito.vllm_mlx import VLLMMLXProvider + + +def test_vllm_mlx_defaults_to_local_openai_compatible_endpoint(): + provider = VLLMMLXProvider(model="mlx-community/Qwen3-4B-4bit", strict=True) + + assert provider.base_url == "http://localhost:8000/v1" + assert provider.model == "mlx-community/Qwen3-4B-4bit" + assert provider.strict is True + + +def test_vllm_mlx_is_registered_provider(): + provider = get_provider( + "vllm-mlx", + model="mlx-community/Qwen3-4B-4bit", + base_url="http://127.0.0.1:9001/v1", + api_key="test-key", + strict=True, + ) + + assert "vllm-mlx" in list_providers() + assert isinstance(provider, VLLMMLXProvider) + assert provider.base_url == "http://127.0.0.1:9001/v1" + assert provider.api_key == "test-key" + + +def test_vllm_mlx_underscore_alias_is_registered(): + provider = get_provider("vllm_mlx", model="test-model") + + assert isinstance(provider, VLLMMLXProvider)