From 040dede1cf9809ae4f6abe7e1d44bed7244716db Mon Sep 17 00:00:00 2001 From: MrJev <330784339+MrJev@users.noreply.github.com> Date: Sun, 20 Sep 2026 18:35:33 +0000 Subject: [PATCH 1/2] benchmarks: confirm --model against the server before running llama.cpp accepts any model name and answers with whatever is loaded, so 'node benchmarks/run-semif.mjs' without --model produced rows labelled qwen3.8-27b-text-64k on a server holding something else. The llama.cpp benchmark now lists GET /v1/models first and stops unless the requested model is served, naming what the server does hold. The report's runtime block records resolvedModel and modelChecked. --skip-model-check runs anyway and records the run as unchecked. --- CHANGELOG.md | 4 +++ benchmarks/README.md | 6 ++++ benchmarks/run-semif.mjs | 56 +++++++++++++++++++++++++++++- tests/benchmark.test.mjs | 42 ++++++++++++++++++++++ tests/fixtures/benchmark-fetch.mjs | 19 +++++++++- 5 files changed, 125 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e6eef40..0099920 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,9 @@ # Changelog +## Unreleased + +- Confirmed the SemIf benchmark's `--model` against the server's `/v1/models` before running, since llama.cpp answers with whatever is loaded however the request names the model. `--skip-model-check` keeps the old behaviour and records the run as unchecked. + ## 0.5.0 - 2026-09-20 - Added label scoring through OpenRouter with `choosekit/openrouter`. diff --git a/benchmarks/README.md b/benchmarks/README.md index c6073cb..5291cbd 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -21,6 +21,12 @@ node benchmarks/run-semif.mjs \ --output benchmarks/results/semif-qwen-labels.json ``` +Before the first row, the script asks the server which models it serves (`GET /v1/models`) and stops +unless `--model` is one of them. llama.cpp answers with whatever is loaded however the request names +the model, so without that check a run labelled `qwen3.8-27b-text-64k` may have been answered by +something else entirely. `--skip-model-check` runs anyway and records `modelChecked: false` with a +null `resolvedModel` in the report's `runtime` block. + Run the Jev comparison with an OpenRouter API key: ```sh diff --git a/benchmarks/run-semif.mjs b/benchmarks/run-semif.mjs index 67fa26a..49f1241 100644 --- a/benchmarks/run-semif.mjs +++ b/benchmarks/run-semif.mjs @@ -89,6 +89,48 @@ function summarize(results, startedAt) { }; } +/** llama.cpp serves /v1/models beside /completion; mirror the adapter's base-URL handling. */ +function modelsURL(baseURL) { + const url = new URL(baseURL); + const path = url.pathname.replace(/\/v1\/?$/, "").replace(/\/$/, ""); + url.pathname = `${path}/v1/models`; + return url.href; +} + +async function servedModelIds(url) { + const response = await fetch(url, { signal: AbortSignal.timeout(30_000) }); + if (!response.ok) throw new Error(`HTTP ${response.status} ${response.statusText}`); + const body = await response.json(); + const entries = Array.isArray(body?.data) ? body.data : []; + const ids = entries.map((entry) => entry?.id).filter((id) => typeof id === "string" && id.length > 0); + if (ids.length === 0) throw new Error("the response listed no models"); + return ids; +} + +/** + * llama.cpp answers with whatever is loaded however the request names the + * model, so an unchecked --model puts a name in the results that may never + * have run. Confirm it against the server before spending an hour on rows. + */ +async function checkModel(baseURL, model) { + const url = modelsURL(baseURL); + let ids; + try { + ids = await servedModelIds(url); + } catch (error) { + const detail = error instanceof Error ? error.message : String(error); + throw new Error(`Could not list the models at ${url}: ${detail}.` + + " Start the server first, or pass --skip-model-check to record the run as unverified."); + } + if (!ids.includes(model)) { + throw new Error(`The server at ${url} serves ${ids.map((id) => `"${id}"`).join(", ")},` + + ` not "${model}". llama.cpp answers with whatever is loaded however the request names the` + + " model, so these results would carry a model name that never ran." + + " Pass --model with one of the ids above, or --skip-model-check to record the run as unverified."); + } + return model; +} + const input = option("--input", "benchmarks/data/semif-authored144.jsonl"); const mode = option("--mode", "labels"); if (mode !== "labels" && mode !== "minimal-prefix") { @@ -98,6 +140,7 @@ const output = option("--output", `benchmarks/results/semif-qwen3.8-27b-producti const baseURL = option("--base-url", process.env.LLAMA_CPP_BASE_URL ?? "http://127.0.0.1:11434/"); const model = option("--model", process.env.LLAMA_CPP_MODEL ?? "qwen3.8-27b-text-64k"); +const skipModelCheck = process.argv.includes("--skip-model-check"); const rawLimit = option("--limit", undefined); const limit = rawLimit === undefined ? undefined : Number.parseInt(rawLimit, 10); @@ -111,6 +154,9 @@ const allRows = source.toString("utf8").trim().split(/\r?\n/).map((line) => JSON if (allRows.length !== 144) throw new Error(`Expected 144 SemIf rows, received ${allRows.length}.`); const rows = limit === undefined ? allRows : allRows.slice(0, limit); mkdirSync(dirname(output), { recursive: true }); +const resolvedModel = skipModelCheck ? null : await checkModel(baseURL, model); +if (skipModelCheck) console.warn("--skip-model-check: the recorded model is the one requested, unconfirmed."); +else console.log(`Model confirmed by the server: ${resolvedModel}`); const choose = fromLlamaCpp({ baseURL, model, mode }); const results = []; const startedAt = performance.now(); @@ -172,7 +218,15 @@ for (let index = 0; index < rows.length; index++) { totalRows: allRows.length, selectedRows: rows.length, }, - runtime: { baseURL, model, mode, adapter: "choosekit/llama-cpp", packageVersion: "0.5.0" }, + runtime: { + baseURL, + model, + resolvedModel, + modelChecked: !skipModelCheck, + mode, + adapter: "choosekit/llama-cpp", + packageVersion: "0.5.0", + }, interpretation: mode === "labels" ? "Package A/B/C label prompt and distinguishing-token likelihoods." : "Package original-key prompt and minimal distinguishing-prefix likelihoods.", diff --git a/tests/benchmark.test.mjs b/tests/benchmark.test.mjs index bd7a1fa..7616faf 100644 --- a/tests/benchmark.test.mjs +++ b/tests/benchmark.test.mjs @@ -58,6 +58,48 @@ test("llama.cpp benchmark creates its output directory before inference", () => assert.equal(report.results.length, 1); assert.equal(report.summary.errors, 1); assert.match(report.results[0].error, /Intentional benchmark test failure/); + assert.equal(report.runtime.modelChecked, true); + assert.equal(report.runtime.resolvedModel, "qwen3.8-27b-text-64k"); +}); + +test("llama.cpp benchmark refuses a model the server does not serve", () => { + const output = temporaryPath("qwen.json"); + const result = run("run-semif.mjs", [ + "--input", input, "--output", output, "--limit", "1", + ], { + CHOOSEKIT_EXPECT_OUTPUT_PARENT: dirname(output), + CHOOSEKIT_SERVED_MODELS: "/gguf/LFM2.5-1.2B-Instruct-Q8_0.gguf", + }); + + assert.notEqual(result.status, 0); + assert.match(result.stderr, /serves "\/gguf\/LFM2\.5-1\.2B-Instruct-Q8_0\.gguf", not "qwen3\.8-27b-text-64k"/); + assert.throws(() => readFileSync(output)); +}); + +test("llama.cpp benchmark explains an unreachable model list", () => { + const output = temporaryPath("qwen.json"); + const result = run("run-semif.mjs", [ + "--input", input, "--output", output, "--limit", "1", + ], { CHOOSEKIT_EXPECT_OUTPUT_PARENT: dirname(output), CHOOSEKIT_MODELS_STATUS: "404" }); + + assert.notEqual(result.status, 0); + assert.match(result.stderr, /Could not list the models at .*\/v1\/models: HTTP 404/); + assert.match(result.stderr, /--skip-model-check/); +}); + +test("llama.cpp benchmark records an unchecked run as unchecked", () => { + const output = temporaryPath("qwen.json"); + const result = run("run-semif.mjs", [ + "--input", input, "--output", output, "--limit", "1", "--skip-model-check", + ], { + CHOOSEKIT_EXPECT_OUTPUT_PARENT: dirname(output), + CHOOSEKIT_SERVED_MODELS: "/gguf/LFM2.5-1.2B-Instruct-Q8_0.gguf", + }); + + assert.equal(result.status, 0, result.stderr); + const report = JSON.parse(readFileSync(output, "utf8")); + assert.equal(report.runtime.modelChecked, false); + assert.equal(report.runtime.resolvedModel, null); }); test("OpenRouter benchmark creates its output directory before inference", () => { diff --git a/tests/fixtures/benchmark-fetch.mjs b/tests/fixtures/benchmark-fetch.mjs index bbbb01b..7ae4dc0 100644 --- a/tests/fixtures/benchmark-fetch.mjs +++ b/tests/fixtures/benchmark-fetch.mjs @@ -1,6 +1,23 @@ import { existsSync } from "node:fs"; -globalThis.fetch = async () => { +/** The llama.cpp benchmark lists /v1/models before inference; everything else fails on purpose. */ +function servedModels() { + const ids = (process.env.CHOOSEKIT_SERVED_MODELS ?? "qwen3.8-27b-text-64k") + .split(",").filter((id) => id.length > 0); + return new Response(JSON.stringify({ object: "list", data: ids.map((id) => ({ id, object: "model" })) }), { + status: 200, + headers: { "content-type": "application/json" }, + }); +} + +globalThis.fetch = async (input) => { + const url = typeof input === "string" ? input : input.url; + if (new URL(url).pathname.endsWith("/v1/models")) { + if (process.env.CHOOSEKIT_MODELS_STATUS) { + return new Response("nope", { status: Number(process.env.CHOOSEKIT_MODELS_STATUS) }); + } + return servedModels(); + } const expectedParent = process.env.CHOOSEKIT_EXPECT_OUTPUT_PARENT; if (!expectedParent || !existsSync(expectedParent)) { throw new Error("Benchmark output directory was not created before the first request."); From 03a1c3e0734a6e8c9bce28bb3f0d0429bed21292 Mon Sep 17 00:00:00 2001 From: NotXf1le <89696340+NotXf1le@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:16:29 +0200 Subject: [PATCH 2/2] benchmarks: tighten model-check messages --- CHANGELOG.md | 3 ++- benchmarks/README.md | 8 +++----- benchmarks/run-semif.mjs | 19 ++++++++----------- tests/benchmark.test.mjs | 3 ++- tests/fixtures/benchmark-fetch.mjs | 2 +- 5 files changed, 16 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0099920..6370546 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,7 +2,8 @@ ## Unreleased -- Confirmed the SemIf benchmark's `--model` against the server's `/v1/models` before running, since llama.cpp answers with whatever is loaded however the request names the model. `--skip-model-check` keeps the old behaviour and records the run as unchecked. +- Added an exact `/v1/models` check to the SemIf llama.cpp benchmark, with + `--skip-model-check` for unverified runs. ## 0.5.0 - 2026-09-20 diff --git a/benchmarks/README.md b/benchmarks/README.md index 02074c1..ac65d0f 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -93,11 +93,9 @@ node benchmarks/run-semif.mjs \ --output benchmarks/results/semif-qwen-labels.json ``` -Before the first row, the script asks the server which models it serves (`GET /v1/models`) and stops -unless `--model` is one of them. llama.cpp answers with whatever is loaded however the request names -the model, so without that check a run labelled `qwen3.8-27b-text-64k` may have been answered by -something else entirely. `--skip-model-check` runs anyway and records `modelChecked: false` in the -report's `runtime` block. +Before running, the script requires the selected model ID to appear in `GET /v1/models`. This prevents +a report from being labelled with a model the server did not expose. Use `--skip-model-check` to bypass +the check; the report then records `modelChecked: false` in its `runtime` block. Run the Jev comparison with an OpenRouter API key: diff --git a/benchmarks/run-semif.mjs b/benchmarks/run-semif.mjs index 41c85d0..460d9df 100644 --- a/benchmarks/run-semif.mjs +++ b/benchmarks/run-semif.mjs @@ -111,11 +111,7 @@ async function servedModelIds(url) { return ids; } -/** - * llama.cpp answers with whatever is loaded however the request names the - * model, so an unchecked --model puts a name in the results that may never - * have run. Confirm it against the server before spending an hour on rows. - */ +/** Verify that the report's model ID is listed by the server before inference. */ async function checkModel(baseURL, model) { const url = modelsURL(baseURL); let ids; @@ -124,13 +120,14 @@ async function checkModel(baseURL, model) { } catch (error) { const detail = error instanceof Error ? error.message : String(error); throw new Error(`Could not list the models at ${url}: ${detail}.` - + " Start the server first, or pass --skip-model-check to record the run as unverified."); + + " Ensure the server exposes /v1/models, or pass --skip-model-check" + + " to record the run as unverified."); } if (!ids.includes(model)) { - throw new Error(`The server at ${url} serves ${ids.map((id) => `"${id}"`).join(", ")},` - + ` not "${model}". llama.cpp answers with whatever is loaded however the request names the` - + " model, so these results would carry a model name that never ran." - + " Pass --model with one of the ids above, or --skip-model-check to record the run as unverified."); + throw new Error(`The server at ${url} does not list "${model}".` + + ` Available models: ${ids.map((id) => `"${id}"`).join(", ")}.` + + " Pass --model with one of the IDs above, or --skip-model-check" + + " to record the run as unverified."); } } @@ -157,7 +154,7 @@ const allRows = source.toString("utf8").trim().split(/\r?\n/).map((line) => JSON if (allRows.length !== 144) throw new Error(`Expected 144 SemIf rows, received ${allRows.length}.`); const rows = limit === undefined ? allRows : allRows.slice(0, limit); mkdirSync(dirname(output), { recursive: true }); -if (skipModelCheck) console.warn("--skip-model-check: the recorded model is the one requested, unconfirmed."); +if (skipModelCheck) console.warn(`--skip-model-check: recording unverified model ID "${model}".`); else { await checkModel(baseURL, model); console.log(`Model ID confirmed in the server catalog: ${model}`); diff --git a/tests/benchmark.test.mjs b/tests/benchmark.test.mjs index 2ecc1c8..a664932 100644 --- a/tests/benchmark.test.mjs +++ b/tests/benchmark.test.mjs @@ -71,7 +71,8 @@ test("llama.cpp benchmark refuses a model the server does not serve", () => { }); assert.notEqual(result.status, 0); - assert.match(result.stderr, /serves "\/gguf\/LFM2\.5-1\.2B-Instruct-Q8_0\.gguf", not "qwen3\.8-27b-text-64k"/); + assert.match(result.stderr, /does not list "qwen3\.8-27b-text-64k"/); + assert.match(result.stderr, /Available models: "\/gguf\/LFM2\.5-1\.2B-Instruct-Q8_0\.gguf"/); assert.throws(() => readFileSync(output)); }); diff --git a/tests/fixtures/benchmark-fetch.mjs b/tests/fixtures/benchmark-fetch.mjs index 7ae4dc0..90cae9c 100644 --- a/tests/fixtures/benchmark-fetch.mjs +++ b/tests/fixtures/benchmark-fetch.mjs @@ -14,7 +14,7 @@ globalThis.fetch = async (input) => { const url = typeof input === "string" ? input : input.url; if (new URL(url).pathname.endsWith("/v1/models")) { if (process.env.CHOOSEKIT_MODELS_STATUS) { - return new Response("nope", { status: Number(process.env.CHOOSEKIT_MODELS_STATUS) }); + return new Response("model list unavailable", { status: Number(process.env.CHOOSEKIT_MODELS_STATUS) }); } return servedModels(); }