From b661eacd2298c5ca71f5fa779125e29b1b11a6dd Mon Sep 17 00:00:00 2001 From: NikolaI Baakh Date: Sat, 19 Sep 2026 09:58:23 +0000 Subject: [PATCH 1/2] feat(web): a deployment with no local runtime says so once, instead of warning forever MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The shell had two inline strings for a dead local runtime and rendered every other Ollama control — the pull card, the catalog, the KV-cache card, the context-window card, the `Local (Ollama)` optgroup — as though a runtime always existed. On a deployment that deliberately runs none, "is the ollama service up?" describes a fault that is not there and cannot be fixed. The state now comes from `GET /platform/v1/llm/local-runtime` and never from a failing model list: that query answers `[]` with a 200 in both unhappy states, so its failure no longer carries the information. When the state is `absent` the local half of the Models page collapses into one line, the five Ollama cards are removed rather than disabled, the embedding picker drops its local group so an unrunnable model cannot be chosen, the chat picker drops its Local heading while keeping the core-default row, the first-run welcome stops offering a pull, and a module model slot with nothing to offer says why. `unreachable` keeps today's warning; an older core without the endpoint keeps today's behaviour exactly. Part of #962. --- CHANGELOG.md | 18 ++ docs/user/configuration.md | 32 +++ services/web/package-lock.json | 4 +- services/web/package.json | 2 +- services/web/src/lib/api.ts | 5 + services/web/src/lib/contracts.ts | 22 ++ services/web/src/lib/useLocalRuntime.ts | 63 +++++ services/web/src/screens/ChatScreen.tsx | 68 ++++-- services/web/src/screens/ModelsScreen.tsx | 90 +++++-- services/web/src/screens/ModulesScreen.tsx | 10 + .../web/src/test/ChatLocalRuntime.test.tsx | 140 +++++++++++ .../test/ModelsScreenLocalRuntime.test.tsx | 229 ++++++++++++++++++ .../test/ModuleModelsLocalRuntime.test.tsx | 80 ++++++ .../web/src/test/useLocalRuntime.test.tsx | 94 +++++++ 14 files changed, 823 insertions(+), 34 deletions(-) create mode 100644 services/web/src/lib/useLocalRuntime.ts create mode 100644 services/web/src/test/ChatLocalRuntime.test.tsx create mode 100644 services/web/src/test/ModelsScreenLocalRuntime.test.tsx create mode 100644 services/web/src/test/ModuleModelsLocalRuntime.test.tsx create mode 100644 services/web/src/test/useLocalRuntime.test.tsx diff --git a/CHANGELOG.md b/CHANGELOG.md index d84cbe61..5b98567a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -43,6 +43,24 @@ images to GHCR. hosted-only release *and* proves the guard refuses the half-working one, and `k8s-smoke` upgrades a live release into the mode and asserts the core answers. `core-app` 0.128.0→0.129.0 (MINOR), chart 0.1.2→0.2.0 (MINOR). +- **A deployment with no local AI says so once, instead of warning forever** (#962) — the web + shell had two inline strings for a dead local runtime ("The local runtime is unreachable — is + the ollama service up?", "local runtime unreachable") and rendered every other Ollama control — + the pull card, the catalog, the KV-cache card, the context-window card, the `Local (Ollama)` + optgroup — as though a runtime existed. On a deployment that deliberately runs **none** (hosted + chat, hosted embeddings, `OLLAMA_URL=""`) that reads as a fault the operator is supposed to fix, + when in fact nothing is wrong and there is nothing to fix. The shell now reads the runtime's + state from `GET /platform/v1/llm/local-runtime` — never from a failing model list, which since + this change answers `[]` with a 200 in *both* unhappy states and so carries no information at + all — and when that state is `absent` the local half of the Models page collapses into one + line: local AI is not configured on this deployment, hosted models are unaffected. The five + Ollama cards are removed rather than disabled, the embedding picker drops its local group so an + unrunnable model cannot be chosen, the chat picker drops its `Local` heading while keeping the + core-default row (that default may itself be hosted), the first-run welcome stops offering a + pull, and a module's model slot with nothing left to offer says why instead of looking broken. + `unreachable` keeps today's warning, because that state *is* an error and must still look like + one, and an older core with no such endpoint keeps today's behaviour exactly. + `web` 0.150.0→0.151.0 (MINOR). - **The bound on a turn is the operator's; runaway is caught by behaviour** (#925) — the **Agent cycles** setting stopped at 12, and the route enforced it *silently*: type 40 and 12 was stored. A genuinely long task — search → read → read → summarize → write — ran out of rounds and diff --git a/docs/user/configuration.md b/docs/user/configuration.md index 9b0feccb..a204ecb8 100644 --- a/docs/user/configuration.md +++ b/docs/user/configuration.md @@ -158,3 +158,35 @@ To connect OpenRouter: > **Pausing.** Pausing the LLM runtime protects the local GPU, so it stops local models only. > Hosted chat and hosted embeddings keep working while paused. + +## A deployment with no local AI + +Running **no local runtime at all** — hosted chat, hosted embeddings, no Ollama — is a +supported mode, not a broken install. It is spelled with a blank `OLLAMA_URL`: see +[`config` reference](../reference/config.md) for the variable and +[Kubernetes](../infrastructure/kubernetes.md) for the chart values that produce it. + +What you see in the web UI when the core reports no local runtime: + +- The **Models** page keeps only the hosted half. The catalog, the download tray, the local + model list, the default context window and the KV-cache card are *gone* — not greyed out — + and one line in their place says local AI is not configured on this deployment. Nothing there + could have done anything: every one of those cards is an Ollama control. +- The **Embedding model** picker offers no `Local (Ollama)` group, so you cannot pick a model + that has nothing to run on. Everything on offer is hosted, which means the whole notes, + knowledge and memory corpus goes to that provider — the trade-off described above is no + longer optional here, so make it deliberately. +- The **chat model picker** drops its `Local` heading and lists your hosted models. The *core + default* entry stays, because the core's default may itself be a hosted model. +- A module's **model slots** (Modules page) offer the core default and any saved hosted model; + a slot with neither says why it is empty rather than looking broken. + +None of this is a warning, because nothing is wrong. A deployment that *does* expect a local +runtime and cannot reach it is a different state and still says so, in the words it always +used: "The local runtime is unreachable — is the ollama service up?" If you see that, something +is down; if you see "Local AI is not configured on this deployment", nothing is. + +> **Set a hosted default.** With no local runtime, a stale `LLM_DEFAULT_MODEL` or +> `MEMORY_EMBED_MODEL` naming a local model (`llama3.2`, `nomic-embed-text`) cannot run. Star a +> hosted chat model under *Hosted models*, and pick a hosted embedding model under *Embedding +> model*, so both defaults point somewhere that answers. diff --git a/services/web/package-lock.json b/services/web/package-lock.json index ee777d9c..bf4c75f0 100644 --- a/services/web/package-lock.json +++ b/services/web/package-lock.json @@ -1,12 +1,12 @@ { "name": "epicurus-web", - "version": "0.150.0", + "version": "0.151.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "epicurus-web", - "version": "0.150.0", + "version": "0.151.0", "dependencies": { "@fontsource-variable/inter": "^5.3.0", "@fontsource-variable/jetbrains-mono": "^5.3.0", diff --git a/services/web/package.json b/services/web/package.json index 2d9cf197..fd1f9afa 100644 --- a/services/web/package.json +++ b/services/web/package.json @@ -1,7 +1,7 @@ { "name": "epicurus-web", "private": true, - "version": "0.150.0", + "version": "0.151.0", "type": "module", "description": "The epicurus web shell: chat, model manager, power toggle, and manifest-driven module UI.", "scripts": { diff --git a/services/web/src/lib/api.ts b/services/web/src/lib/api.ts index ff9be3f6..79a355d7 100644 --- a/services/web/src/lib/api.ts +++ b/services/web/src/lib/api.ts @@ -31,6 +31,7 @@ import { FileText, HoverCard, LlmPrefs, + LocalRuntimeStatus, LogEntry, MaintenanceCurrentRun, MaintenanceRunPage, @@ -211,6 +212,10 @@ export const api = { z.array(ModelInfo), `/platform/v1/llm/models${withCapabilities ? "?capabilities=true" : ""}`, ), + // Whether this deployment runs a local runtime at all (#962) — `absent` (deliberately none), + // `unreachable` (one is configured and down), or `ok`. Its own endpoint rather than an + // envelope around the model list, which stays a bare array for its five web consumers. + localRuntime: () => request(LocalRuntimeStatus, "/platform/v1/llm/local-runtime"), // The browsable model catalog the core parses from upstream on a schedule (#269). catalog: () => request(CatalogResponse, "/platform/v1/llm/catalog"), deleteModel: (name: string) => diff --git a/services/web/src/lib/contracts.ts b/services/web/src/lib/contracts.ts index ae8cf8ec..44b0515b 100644 --- a/services/web/src/lib/contracts.ts +++ b/services/web/src/lib/contracts.ts @@ -23,6 +23,28 @@ export const ModelInfo = z.object({ }); export type ModelInfo = z.infer; +/** + * Whether this deployment runs a local model runtime at all (#962). + * + * Three states, because `absent` and `unreachable` are not the same thing and never were: + * `absent` is a **deliberate deployment mode** (`OLLAMA_URL=""` — hosted chat, hosted + * embeddings, no Ollama), so nothing is wrong and there is nothing for the operator to fix; + * `unreachable` is a runtime the deployment expects and cannot reach, which *is* an error and + * still looks like one. `ok` is the common case. The state is read from this endpoint and never + * inferred from a failing model list — since #962 `GET /llm/models` answers `[]` with a 200 in + * both of the unhappy states, so an error there no longer carries the information. + */ +export const LocalRuntimeState = z.enum(["absent", "unreachable", "ok"]); +export type LocalRuntimeState = z.infer; + +export const LocalRuntimeStatus = z.object({ + state: LocalRuntimeState, + // Whether `OLLAMA_URL` carries a value at all. Defaulted for tolerance: the state is what + // every surface renders from, and a core that reports one without the other is still usable. + url_configured: z.boolean().default(true), +}); +export type LocalRuntimeStatus = z.infer; + /** Per-model tuning; null on a field means "inherit" the global / env default. */ export const ModelSettings = z.object({ context_window: z.number().nullable(), diff --git a/services/web/src/lib/useLocalRuntime.ts b/services/web/src/lib/useLocalRuntime.ts new file mode 100644 index 00000000..d66b6a2b --- /dev/null +++ b/services/web/src/lib/useLocalRuntime.ts @@ -0,0 +1,63 @@ +/** + * Does this deployment have a local model runtime? (#962) + * + * One query, one answer, read by every surface that used to assume Ollama existed. It exists + * because the shell had no way to tell three different things apart — a runtime that is *absent + * by design* (`OLLAMA_URL=""`: hosted chat, hosted embeddings, no Ollama), one that is + * configured and **unreachable**, and one that is fine — so it rendered the pull card, the + * catalog, the KV-cache card and a `Local (Ollama)` optgroup on a deployment where none of them + * could ever do anything, and called the whole situation "is the ollama service up?". + * + * Two rules this hook exists to keep: + * + * 1. **The state comes from `/llm/local-runtime`, never from a failing `/llm/models`.** Since + * #962 the model list answers `[]` with a 200 when the runtime is absent *or* unreachable, + * so an error there carries no information at all. Any code that goes back to inferring the + * state from a query failure silently stops working. + * 2. **An older core, or an unreachable one, reads as `ok`.** A core that predates the endpoint + * 404s; treating that as "absent" would hide the local half of the Models page from a + * deployment that has a perfectly good runtime. `ok` is exactly what every surface assumed + * before this change, so an unknown answer keeps today's behaviour. + */ +import { useQuery } from "@tanstack/react-query"; + +import { api } from "@/lib/api"; +import type { LocalRuntimeState } from "@/lib/contracts"; + +export interface LocalRuntime { + /** `absent` | `unreachable` | `ok` — `ok` until the endpoint says otherwise. */ + state: LocalRuntimeState; + /** No local runtime on this deployment, deliberately. The one flag most callers want. */ + absent: boolean; + /** A runtime is configured and cannot be reached. An error, and it still looks like one. */ + unreachable: boolean; + /** The endpoint has answered (or failed) at least once. Gate a *collapse* on this so a + * hosted-only deployment never flashes the local half of the page before hiding it again. */ + settled: boolean; +} + +export function useLocalRuntime(): LocalRuntime { + const status = useQuery({ + queryKey: ["localRuntime"], + queryFn: () => api.localRuntime(), + // A deployment does not grow or lose a runtime between renders; one fetch per session is + // plenty, and a stale answer here is cheaper than polling a constant. + staleTime: 5 * 60_000, + // A 404 from an older core is a final answer, not a blip worth three attempts. + retry: false, + }); + const state: LocalRuntimeState = status.data?.state ?? "ok"; + return { + state, + absent: state === "absent", + unreachable: state === "unreachable", + // `isFetched` and not `!isPending`, because this gates whether the local half of the Models + // page is mounted at all, and those components subscribe to this very query. An errored + // query (an older core's 404) is permanently stale, so each newly-mounted subscriber + // triggers a refetch — which, with no data to fall back on, puts the query back into + // `pending` and unmounts them again: a page that oscillates forever. `isFetched` latches + // true after the first attempt, so the answer can change but the mounting decision can't + // feed back into it. + settled: status.isFetched, + }; +} diff --git a/services/web/src/screens/ChatScreen.tsx b/services/web/src/screens/ChatScreen.tsx index d30ef820..5cb55207 100644 --- a/services/web/src/screens/ChatScreen.tsx +++ b/services/web/src/screens/ChatScreen.tsx @@ -87,6 +87,7 @@ import { } from "@/lib/format"; import { SHARE_CACHE, SHARE_FILE_KEY, SHARE_FILE_NAME_HEADER, SHARE_META_KEY, type ShareMeta } from "@/lib/shareTarget"; import { SUGGESTION_VERB, suggestionTarget } from "@/lib/suggestions"; +import { useLocalRuntime } from "@/lib/useLocalRuntime"; import { fetchSessions, useChat, @@ -616,6 +617,10 @@ function ModelPicker() { const sessionModel = sessions.data?.find((s) => s.id === sessionId)?.model ?? null; const effectiveModel = sessionModel ?? model; const models = useQuery({ queryKey: ["models"], queryFn: () => api.models(), enabled: open }); + // No local runtime (#962) → no Local group at all: no heading, no "local runtime unreachable", + // nothing to pick. The core-default row survives on its own, because the core's default may + // perfectly well be a hosted model and this is the only way back to it. + const runtime = useLocalRuntime(); const providers = useQuery({ queryKey: ["providers"], queryFn: api.providers, enabled: open }); const llmPrefs = useQuery({ queryKey: ["llmPrefs"], queryFn: api.llmPrefs, enabled: open }); const saved = useQuery({ queryKey: ["savedModels"], queryFn: api.savedModels, enabled: open }); @@ -695,25 +700,34 @@ function ModelPicker() {

)} -
-

Local

+ {runtime.absent ? (
choose(null)} /> - {visibleModels.map((m) => ( - choose(m.name)} - /> - ))} - {models.isError && ( -

local runtime unreachable

- )}
-
+ ) : ( +
+

Local

+
+ choose(null)} /> + {visibleModels.map((m) => ( + choose(m.name)} + /> + ))} + {/* Driven by the runtime's own state (#962) — the model list stopped erroring on + an unreachable runtime, so `models.isError` alone would never fire again. It + is kept beside it for an older core, which still 500s here. */} + {(runtime.unreachable || models.isError) && ( +

local runtime unreachable

+ )} +
+
+ )}

@@ -817,6 +831,28 @@ function Welcome() { const active = useDownloads((s) => s.active); const queryClient = useQueryClient(); const suggestions = ["llama3.2", "qwen2.5:0.5b"]; + // The first thing a new operator reads must not be advice they cannot take (#962): with no + // local runtime there is nothing to pull, and "pull a local one" would send them looking for a + // Pull button that this deployment deliberately doesn't have. + const runtime = useLocalRuntime(); + + if (runtime.absent) { + return ( + + +

Welcome to the garden

+

+ No model is configured yet. This deployment runs no local AI, so add a hosted + provider key under{" "} + + Models + {" "} + and pick a model there. +

+ + + ); + } return ( diff --git a/services/web/src/screens/ModelsScreen.tsx b/services/web/src/screens/ModelsScreen.tsx index 74b3ba84..d9fd35fc 100644 --- a/services/web/src/screens/ModelsScreen.tsx +++ b/services/web/src/screens/ModelsScreen.tsx @@ -59,6 +59,7 @@ import type { import { CAPABILITY_META, shownCapabilities } from "@/lib/icons"; import { assessFit, fitFilterOf, type FitFilter } from "@/lib/modelFit"; import { recommendKvCache } from "@/lib/kvCacheFit"; +import { useLocalRuntime } from "@/lib/useLocalRuntime"; import { formatVariantSize, isCloudTag, @@ -426,6 +427,11 @@ export function CatalogBrowser({ installed }: { installed: Set }) { export function LocalModels() { const queryClient = useQueryClient(); + // The runtime's state comes from its own endpoint (#962), not from the model list failing: + // since the core stopped 500-ing, `/llm/models` answers `[]` with a 200 whether the runtime is + // absent or unreachable, so `models.isError` can no longer tell the operator anything. It is + // still honoured for an older core, which does still 500 here. + const runtime = useLocalRuntime(); // Ask for capabilities here so each model can be badged with what it does (tools/vision/…). // Keyed under ["models", …] so the mutations' `["models"]` invalidation still refreshes it. const models = useQuery({ @@ -490,12 +496,12 @@ export function LocalModels() { )}
{models.isLoading && } - {models.isError && ( + {(runtime.unreachable || models.isError) && (

The local runtime is unreachable — is the ollama service up?

)} - {models.data?.length === 0 && ( + {!runtime.unreachable && !models.isError && models.data?.length === 0 && (

None yet. Pull one above — it stays on your disk.

)}
@@ -607,6 +613,32 @@ export function LocalModels() { ); } +// ── No local runtime (#962) ───────────────────────────────────────────────────── + +/** + * What the local half of this page becomes on a deployment that runs **no local runtime** + * (`OLLAMA_URL=""` — hosted chat, hosted embeddings, no Ollama): one calm line. + * + * It replaces — rather than disables — the catalog, the download tray, the local-model list, the + * context-window card and the KV-cache card, because every one of them is an Ollama control and + * a disabled control still reads as "something here is broken; fix it". Nothing is broken: this + * is a supported mode, and the cards below it (hosted providers, hosted models, the embedding + * default) are the whole page here. `unreachable` is the opposite case and keeps its warning. + */ +export function LocalRuntimeAbsent() { + return ( + +

Local AI

+

+ Local AI is not configured on this deployment — there is no local runtime to pull models + into, so the catalog, the local model list and the runtime settings don't apply. + Hosted models are unaffected: set a provider key below and chat and embeddings run + through it as usual. +

+
+ ); +} + // ── Context window ────────────────────────────────────────────────────────────── /** Render a megabyte count as a compact GB string (VRAM / model size). */ @@ -1550,6 +1582,9 @@ export function KvCache() { export function EmbedDefault() { const queryClient = useQueryClient(); const models = useQuery({ queryKey: ["models"], queryFn: () => api.models() }); + // With no local runtime (#962) the local group is not merely empty, it is unofferable: a local + // id chosen here cannot run, and the core refuses it at call time. Better not to offer it. + const runtime = useLocalRuntime(); const llmPrefs = useQuery({ queryKey: ["llmPrefs"], queryFn: api.llmPrefs }); // Saved hosted models (#865): the gateway embeds through a hosted provider the same way it // chats through one, so they belong in this select too. The store holds *any* hosted id and @@ -1601,6 +1636,13 @@ export function EmbedDefault() { ones the catalogue knows to be chat models aren't offered here, and the core refuses one if it slips through. A hosted choice sends the whole notes, knowledge, and memory corpus to that provider. + {runtime.absent && ( + <> + {" "} + This deployment runs no local runtime, so every option here is a hosted one — which + means the corpus leaves the machine whichever you pick. + + )}

{llmPrefs.isLoading ? ( @@ -1616,13 +1658,15 @@ export function EmbedDefault() { > {orphan && } - - {available.map((m) => ( - - ))} - + {!runtime.absent && ( + + {available.map((m) => ( + + ))} + + )} {hosted.length > 0 && ( {hosted.map((m) => ( @@ -1634,7 +1678,11 @@ export function EmbedDefault() { )} - {current && ( + {/* A stale *local* id can still be the stored default on a deployment with no runtime + (an orphan, kept selected above so the select doesn't misreport the core). Its + settings sheet is all Ollama knobs — num_ctx, keep-alive, run-on — so offering it + would be a control that cannot do anything. A hosted id keeps its sheet. */} + {current && (isHostedModelId(current) || !runtime.absent) && (