diff --git a/CHANGELOG.md b/CHANGELOG.md index 06f3cba..758fe3f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -86,6 +86,9 @@ This file records what changes **in the product** – process and session state existing `auth` schema only when the app owns it. ### Fixed +- Processing errors name the actual cause: when the model provider fails (e.g. overloaded), staff read + "Das KI-Modell des Anbieters war nicht verfügbar (z. B. überlastet)." instead of "Der KI-Dienst ist nicht + erreichbar."; a busy or disturbed AI service and an unusable model answer have their own texts (#80). - Request detail: while a failed attempt waits for its retry, the page shows the same facts as the list – last error, attempts and the next retry – instead of "Die Dokumente werden gerade ausgewertet". On the showcase, opening the list or a request picks up due retries (at most once per 30 s and never while diff --git a/src/features/extraction/ai-client.test.ts b/src/features/extraction/ai-client.test.ts index 266b6ea..ea84711 100644 --- a/src/features/extraction/ai-client.test.ts +++ b/src/features/extraction/ai-client.test.ts @@ -56,6 +56,47 @@ describe("AI service client", () => { await expect(client.extract(input)).rejects.toMatchObject({ retryable: true, status }); }); + // #80: the contract's error code names the cause (model provider vs. our service); staff texts depend on it. + it.each([ + [502, { error: { code: "model_error", message: "the model call failed" } }, "model_error"], + [502, { error: { code: "model_output_invalid", message: "m" } }, "model_output_invalid"], + [429, { error: { code: "busy", message: "all extraction slots are busy" } }, "busy"], + [500, { error: { code: "internal_error", message: "m" } }, "internal_error"], + [503, { error: { code: "not-a-contract-code", message: "m" } }, "unavailable"], + ] as const)("keeps the contract error code of a retryable HTTP %i answer", async (status, body, code) => { + handler = (_request, response) => json(response, status, body); + const client = createAiServiceClient({ baseUrl, token: "t".repeat(24), timeoutMs: 2000 }); + + await expect(client.extract(input)).rejects.toMatchObject({ retryable: true, status, code }); + }); + + it("classifies a retryable answer without a JSON error body (e.g. a platform 502 page) as unavailable", async () => { + handler = (_request, response) => { + response.writeHead(502, { "content-type": "text/html" }); + response.end("Bad Gateway"); + }; + const client = createAiServiceClient({ baseUrl, token: "t".repeat(24), timeoutMs: 2000 }); + + await expect(client.extract(input)).rejects.toMatchObject({ retryable: true, status: 502, code: "unavailable" }); + }); + + it("reads at most a small error object: an oversized error body is dropped, not buffered (#82 review)", async () => { + handler = (_request, response) => json(response, 502, { error: { code: "model_error", message: "m".repeat(64 * 1024) } }); + const client = createAiServiceClient({ baseUrl, token: "t".repeat(24), timeoutMs: 2000 }); + + await expect(client.extract(input)).rejects.toMatchObject({ retryable: true, status: 502, code: "unavailable" }); + }); + + it("classifies an error body that stalls after the headers as a timeout (#82 review)", async () => { + handler = (_request, response) => { + response.writeHead(503, { "content-type": "application/json" }); + response.write('{"error":'); // …and never finishes + }; + const client = createAiServiceClient({ baseUrl, token: "t".repeat(24), timeoutMs: 200 }); + + await expect(client.extract(input)).rejects.toMatchObject({ retryable: true, status: 503, code: "timeout" }); + }); + it.each([ [413, "document"], [415, "document"], diff --git a/src/features/extraction/ai-client.ts b/src/features/extraction/ai-client.ts index 061fa3f..cc3c2c1 100644 --- a/src/features/extraction/ai-client.ts +++ b/src/features/extraction/ai-client.ts @@ -1,4 +1,5 @@ import { z } from "zod"; +import type { components } from "./ai-service.contract"; import type { ExtractResponse } from "./types"; // Client of the stateless AI service (contract: contracts/ai-service.openapi.yaml, types generated in @@ -60,6 +61,67 @@ const responseSchema = z.object({ }).loose(); const RETRYABLE_STATUS = new Set([408, 429, 500, 502, 503, 504]); +// Every error code of the contract (a Record, so a new code in the generated types fails the build until +// it is listed). A retryable answer keeps its code, so staff read the real cause (#80): `model_error` +// means the model provider failed, not our service. +const CONTRACT_ERROR_CODES: Record = { + invalid_request: true, + unauthorized: true, + length_required: true, + document_too_large: true, + unsupported_media_type: true, + document_unparseable: true, + document_too_long: true, + busy: true, + model_error: true, + model_output_invalid: true, + internal_error: true, +}; +const errorBody = z.object({ error: z.object({ code: z.string() }) }); +// The contract's error object is a code and a short fixed message; anything larger is not one (#82 review). +const MAX_ERROR_BODY_BYTES = 4 * 1024; + +const isTimeout = (error: unknown) => error instanceof Error && (error.name === "TimeoutError" || error.name === "AbortError"); + +/** The body as text, or null once it exceeds `limit` bytes (the rest is never buffered). */ +async function boundedText(response: Response, limit: number): Promise { + const reader = response.body?.getReader(); + if (!reader) return ""; + const decoder = new TextDecoder(); + let text = ""; + let size = 0; + for (;;) { + const { done, value } = await reader.read(); + if (done) return text + decoder.decode(); + size += value.byteLength; + if (size > limit) { + await reader.cancel().catch(() => undefined); + return null; + } + text += decoder.decode(value, { stream: true }); + } +} + +/** + * The contract code of an error answer; `unavailable` without a valid error object (e.g. a platform + * 502 page, an oversized body); `timeout` when the body stalls until the call's timeout. + */ +async function errorCode(response: Response): Promise { + let text: string | null; + try { + text = await boundedText(response, MAX_ERROR_BODY_BYTES); + } catch (error) { + return isTimeout(error) ? "timeout" : "unavailable"; + } + let body: unknown = null; + try { + body = text === null ? null : JSON.parse(text); + } catch { + // not JSON – no contract code + } + const parsed = errorBody.safeParse(body); + return parsed.success && Object.hasOwn(CONTRACT_ERROR_CODES, parsed.data.error.code) ? parsed.data.error.code : "unavailable"; +} const DOCUMENT_STATUS = new Set([413, 415, 422]); /** @@ -101,12 +163,11 @@ export function createAiServiceClient(settings: AiServiceSettings): AiServiceCli signal: AbortSignal.timeout(settings.timeoutMs), }); } catch (error) { - const timedOut = error instanceof Error && (error.name === "TimeoutError" || error.name === "AbortError"); - throw new AiServiceError(timedOut ? "timeout" : "unreachable", true, "service"); + throw new AiServiceError(isTimeout(error) ? "timeout" : "unreachable", true, "service"); } if (!response.ok) { + if (RETRYABLE_STATUS.has(response.status)) throw new AiServiceError(await errorCode(response), true, "service", response.status); await response.body?.cancel(); - if (RETRYABLE_STATUS.has(response.status)) throw new AiServiceError("unavailable", true, "service", response.status); // Only these say "this document": everything else (400, 401, 403, 404, 405, …) points at the // call itself – a misconfigured URL or client bug must not look like unreadable documents. const scope = DOCUMENT_STATUS.has(response.status) ? "document" : "service"; diff --git a/src/features/jobs/describe-failure.test.ts b/src/features/jobs/describe-failure.test.ts new file mode 100644 index 0000000..8192255 --- /dev/null +++ b/src/features/jobs/describe-failure.test.ts @@ -0,0 +1,32 @@ +import { describe, expect, it } from "vitest"; +import { AiServiceError } from "@/features/extraction"; +import { describeFailure } from "./process-request"; + +// #80: staff read the real cause – a model provider failure is not "our service is unreachable". No text +// promises a retry: it stays when the attempts are used up, and the next retry is shown separately (#70). +describe("describeFailure", () => { + it.each([ + ["model_error", 502, "Das KI-Modell des Anbieters war nicht verfügbar (z. B. überlastet)."], + ["model_output_invalid", 502, "Das KI-Modell hat keine verwertbare Antwort geliefert."], + ["busy", 429, "Der KI-Dienst ist gerade ausgelastet."], + ["internal_error", 500, "Der KI-Dienst ist vorübergehend gestört."], + ["unavailable", 503, "Der KI-Dienst ist vorübergehend gestört."], + ["unreachable", undefined, "Der KI-Dienst ist nicht erreichbar."], + ["timeout", undefined, "Der KI-Dienst hat nicht rechtzeitig geantwortet."], + ] as const)("names the cause of a retryable %s", (code, status, text) => { + expect(describeFailure(new AiServiceError(code, true, "service", status))).toBe(text); + }); + + it("keeps the text for a permanent rejection", () => { + expect(describeFailure(new AiServiceError("rejected", false, "service", 401))).toBe( + "Der KI-Dienst hat die Anfrage abgelehnt – bitte die Administration informieren.", + ); + }); + + it("never promises a retry and never names a host or status line", () => { + for (const code of ["model_error", "model_output_invalid", "busy", "internal_error", "unavailable", "unreachable", "timeout"]) { + const text = describeFailure(new AiServiceError(code, true, "service", 502)); + expect(text).not.toMatch(/Versuch|http|HTTP|\d{3}/); + } + }); +}); diff --git a/src/features/jobs/process-request.ts b/src/features/jobs/process-request.ts index bcfeb2a..da6a32d 100644 --- a/src/features/jobs/process-request.ts +++ b/src/features/jobs/process-request.ts @@ -25,12 +25,21 @@ export class PermanentProcessingError extends Error { // is kept as an original but skipped with a visible note. const AI_KINDS = new Set(["pdf", "eml", "xlsx", "docx", "msg"]); +// Causes of retryable AI failures (#80): the model provider failing is not our service being down. No +// text promises a retry – it stays when the attempts are used up; the next retry is shown separately. +const RETRYABLE_CAUSE: Record = { + timeout: "Der KI-Dienst hat nicht rechtzeitig geantwortet.", + unreachable: "Der KI-Dienst ist nicht erreichbar.", + model_error: "Das KI-Modell des Anbieters war nicht verfügbar (z. B. überlastet).", + model_output_invalid: "Das KI-Modell hat keine verwertbare Antwort geliefert.", + busy: "Der KI-Dienst ist gerade ausgelastet.", +}; + /** Human-readable causes for staff (DR4): no stack traces, no hosts, no document content. */ export function describeFailure(error: unknown): string { if (error instanceof PermanentProcessingError) return error.cause_; if (error instanceof AiServiceError) { - if (error.code === "timeout") return "Der KI-Dienst hat nicht rechtzeitig geantwortet."; - if (error.retryable) return "Der KI-Dienst ist nicht erreichbar."; + if (error.retryable) return Object.hasOwn(RETRYABLE_CAUSE, error.code) ? RETRYABLE_CAUSE[error.code]! : "Der KI-Dienst ist vorübergehend gestört."; return "Der KI-Dienst hat die Anfrage abgelehnt – bitte die Administration informieren."; } return "Unerwarteter Fehler bei der Verarbeitung."; diff --git a/tests/integration/processing.test.ts b/tests/integration/processing.test.ts index 3cd066e..b1d3e7d 100644 --- a/tests/integration/processing.test.ts +++ b/tests/integration/processing.test.ts @@ -188,10 +188,24 @@ describe("processing: worker, AI service, retries and visible errors", () => { await drainUntil(async () => (await requestOf(admin, requestId))?.status === "ERROR"); const request = await requestOf(admin, requestId); - expect(request).toMatchObject({ status: "ERROR", errorStage: "processing", errorMessage: "Der KI-Dienst ist nicht erreichbar.", attempts: 2, nextRetryAt: null }); + // A bare 5xx without the contract's error object: our service is disturbed, not unreachable (#80). + expect(request).toMatchObject({ status: "ERROR", errorStage: "processing", errorMessage: "Der KI-Dienst ist vorübergehend gestört.", attempts: 2, nextRetryAt: null }); expect(request?.errorMessage).not.toMatch(/127\.0\.0\.1|Error|at /); }); + it("names the model provider as the cause when the AI service answers model_error (#80)", async () => { + reply = () => ({ status: 502, body: { error: { code: "model_error", message: "the model call failed" }, requestId: null } }); + const { requestId } = await newRequest(admin, [{ name: "a.eml", kind: "eml", bytes: MAIL }]); + + await drainUntil(async () => (await requestOf(admin, requestId))?.status === "ERROR"); + + expect(await requestOf(admin, requestId)).toMatchObject({ + status: "ERROR", + errorMessage: "Das KI-Modell des Anbieters war nicht verfügbar (z. B. überlastet).", + attempts: 2, + }); + }); + it("treats a timeout as retryable", async () => { reply = () => "hang"; const { requestId, job } = await newRequest(admin, [{ name: "a.eml", kind: "eml", bytes: MAIL }]);