diff --git a/.gitignore b/.gitignore index 1533ed3..f01d2bf 100644 --- a/.gitignore +++ b/.gitignore @@ -5,4 +5,5 @@ dist/ !.env.example coverage/ /docs/ +/benchmarks/.data/ /benchmarks/results/ diff --git a/README.md b/README.md index 3cbe7f8..8d9b324 100644 --- a/README.md +++ b/README.md @@ -2,10 +2,37 @@ `choosekit` scores a finite set of choices with a language model and returns a typed decision with a probability distribution. It supports local llama.cpp models and an optional OpenRouter backend. +![SuperGPQA direct-choice benchmark](benchmarks/supergpqa-benchmark.svg) + +The chart compares accuracy with a lower-is-better cost-latency product. The green line and confidence band show local Qwen3.8 27B Q4_XL accuracy; it has no cloud cost coordinate. [Method and reproduction](benchmarks/README.md#supergpqa) + +| Model | Accuracy | Cost / 1,000 decisions | Decisions/s | +|---|---:|---:|---:| +| Granite 4.0 H Micro | 19.3% | $0.0053 | 2.84 | +| Llama 3.1 8B | 19.0% | $0.0061 | 1.33 | +| GLM 4.7 Flash | 25.7% | $0.0172 | 1.18 | +| Gemma 4 26B | 37.6% | $0.0184 | 2.19 | +| Jev 1.13 | 53.6% | $0.0244 | 2.86 | +| Granite 4.2 8B | 23.8% | $0.0293 | 3.15 | +| DeepSeek V4.1 Flash | 45.6% | $0.0544 | 1.40 | +| DeepSeek V4 Pro | 43.2% | $0.3588 | 0.88 | +| GLM 5.2 | 44.6% | $0.3932 | 0.69 | +| Kimi K3 | 59.3% | $0.6243 | 0.73 | + +## Install + +Library: + ```sh npm install choosekit ``` +MCP server: + +```sh +npm install --global choosekit-mcp +``` + ## Why Agents often need to choose from known options: @@ -18,13 +45,13 @@ Agents often need to choose from known options: `choosekit` scores choices using the model's conditional log probabilities at the token branches that distinguish them. -The project was inspired by [Jev and the System One model interface](https://typesafe.ai/blog/introducing-system-one-models-and-jev): application state in, typed probabilistic decisions out. Jev is a specialized hosted model. `choosekit` explores the same useful interface with a model you control. The llama.cpp backend keeps application state on infrastructure you choose; OpenRouter is available when a hosted model is more convenient. +The project was inspired by [Jev and the System One model interface](https://typesafe.ai/blog/introducing-system-one-models-and-jev): application state in, typed probabilistic decisions out. Jev is a specialized hosted model. `choosekit` brings the same typed decision interface to general-purpose language models. The llama.cpp backend runs on infrastructure you choose; OpenRouter provides hosted inference. `choosekit` is an independent project with no affiliation to TypeSafe or Jev. ## MCP server -[`choosekit-mcp`](packages/choosekit-mcp/README.md) exposes choosekit through llama.cpp or OpenRouter as a read-only stdio tool for Claude Code, Codex, and OpenCode. Select the backend and configure it with environment variables when starting the MCP server. Every `choose` call uses this configuration. +[`choosekit-mcp`](packages/choosekit-mcp/README.md) exposes choosekit through llama.cpp or OpenRouter as a read-only stdio tool for Claude Code, Codex, and OpenCode. Select the backend and configure it with environment variables when starting the MCP server. ## llama.cpp @@ -65,17 +92,11 @@ const choose = fromOpenRouter({ }); ``` -The OpenRouter backend supports models and providers that return first-token `top_logprobs`, with up to 20 choices. Unlike llama.cpp, this backend sends the prompt to OpenRouter. It requests reasoning to be disabled. Choices omitted from `top_logprobs` receive zero probability. Returned probabilities are normalized across the supplied choices and are not calibrated correctness estimates. +The OpenRouter backend supports models and providers that return first-token `top_logprobs`, with up to 20 choices. It sends the prompt to OpenRouter and requests reasoning to be disabled. -OpenRouter may route the same model through different providers. Set `provider` to an OpenRouter provider slug to use only that provider and disable fallback: +Choices omitted from `top_logprobs` receive zero probability. Returned probabilities are normalized across the supplied choices and are not calibrated correctness estimates. -```ts -const choose = fromOpenRouter({ - apiKey: process.env.OPENROUTER_API_KEY!, - model: "qwen/qwen3.8-27b", - provider: process.env.OPENROUTER_PROVIDER!, -}); -``` +OpenRouter may route the same model through different providers. Set `provider: "provider-slug"` to use only that provider and disable fallback. ## Scoring modes @@ -88,14 +109,53 @@ In `labels` mode, choices are shown to the model as `A`, `B`, `C` instead of the `minimal-prefix` walks the token tree until every key is distinguishable. For keys such as `watermelon` and `watermelon juice`, the shared token path is handled once and scoring stops when the paths separate. -## Benchmark +## Return value + +`choose()` resolves to: + +```ts +{ + choice, // selected caller key + distribution, // normalized probability for every supplied key + scores, // backend log-probability score for every key + margin, // largest probability minus the second largest + entropy, // Shannon entropy in nats + boundaryTokens, // prompt tokens rolled back at a tokenization boundary + usage, // backend work, when reported +} +``` + +The result and its nested records are immutable. Each call is stateless. The caller controls action execution, inference retries, and model selection. + +## Prompt formatting + +`context` is copied unchanged to the start of the scoring prompt. The default formatter then appends the question, choice descriptions, and an answer marker. + +Use `formatPrompt` only when you need custom prompt formatting. The result must preserve `context` as an unchanged prefix so an existing server-side prefix cache can still be reused. + +## Custom scorer + +Use `createChooser` with any backend that can return one comparable conditional log-probability score per candidate: + +```ts +import { createChooser, type Scorer } from "choosekit"; + +const scorer: Scorer = async ({ prompt, candidates, signal }) => ({ + logprobs: await scoreCandidateSequences(prompt, candidates, signal), +}); + +const choose = createChooser(scorer); +``` + +Scores use natural logarithms and must be at most zero. + +## SemIf comparison The local adapter was compared with `typesafe/jev-1.13` on SemIf's official 144-row `authored144` benchmark, which covers evidence interpretation, rule application, and candidate selection. The local model was **Qwen 3.8 27B Q4_XL** served by llama.cpp on an **NVIDIA RTX 4090**. The Qwen run used the default A/B/C mode. The model was already loaded, and requests were sent one at a time to a llama.cpp server on the same machine. | Metric | Qwen 3.8 27B Q4_XL + choosekit | Jev 1.13 | |---|---:|---:| | Accuracy | 96.53% (139/144) | 96.53% (139/144) | -| Average balanced accuracy across task families | 96.01% | 95.56% | | Median latency (p50) | 239 ms | 368 ms | | 95th percentile latency (p95) | 286 ms | 546 ms | | Throughput | 4.02 decisions/s | 2.43 decisions/s | @@ -106,8 +166,6 @@ These results are specific to this 144-case benchmark, and performance can diffe ### Probability examples -Examples from the same benchmark: - [`eafc22c8c40df3932a8e`](benchmarks/data/semif-authored144.jsonl#L112) asks whether the crate is currently in storage. The protocol gives the inventory priority; the current inventory and desk-log entries are missing. | Choice | Qwen probability | Jev probability | @@ -124,7 +182,7 @@ Examples from the same benchmark: | **Insufficient evidence (selected by both)** | **97.605%** | **99.000%** | | Contradicted | 0.029% | 1.000% | -The Qwen + llama.cpp probabilities shown here are [uncalibrated](https://proceedings.mlr.press/v70/guo17a.html). [Jev is trained for calibrated decisions](https://typesafe.ai/blog/introducing-system-one-models-and-jev). The distributions look similar in these examples. This benchmark measures accuracy, latency, and distribution similarity. +The Qwen + llama.cpp probabilities shown here are [uncalibrated](https://proceedings.mlr.press/v70/guo17a.html). [Jev is trained for calibrated decisions](https://typesafe.ai/blog/introducing-system-one-models-and-jev). The distributions look similar in these examples. ### Distribution comparison @@ -142,51 +200,9 @@ Total variation distance (TVD) compares two complete probability distributions. | Cases with TVD at or below 10% | 77.08% (111/144) | | Cases with TVD above 20% | 14.58% (21/144) | -Most distributions are similar. Some differ substantially: the systems select different choices in eight cases, and the largest TVD is 94.23%. - -## Prompt formatting - -`context` is copied unchanged to the start of the scoring prompt. The default formatter then appends the question, choice descriptions, and an answer marker. - -Use `formatPrompt` only when you need custom prompt formatting. The result must preserve `context` as an unchanged prefix so an existing server-side prefix cache can still be reused. - -## Custom scorer - -Use `createChooser` with any backend that can return one comparable conditional log-probability score per candidate: - -```ts -import { createChooser, type Scorer } from "choosekit"; - -const scorer: Scorer = async ({ prompt, candidates, signal }) => ({ - logprobs: await scoreCandidateSequences(prompt, candidates, signal), -}); - -const choose = createChooser(scorer); -``` - -Scores use natural logarithms and must be at most zero. - -## Result - -`choose()` resolves to: - -```ts -{ - choice, // selected caller key - distribution, // normalized probability for every supplied key - scores, // backend log-probability score for every key - margin, // largest probability minus the second largest - entropy, // Shannon entropy in nats - boundaryTokens, // prompt tokens rolled back at a tokenization boundary - usage, // backend work, when reported -} -``` - -The result and its nested records are immutable. Each call is stateless. The caller controls action execution, inference retries, and model selection. - ## Requirements - Node.js 20 or newer. -- No runtime dependencies, model downloads, installation hooks, or bundled inference servers. +- The `choosekit` package has no runtime dependencies, model downloads, installation hooks, or bundled inference servers. [Apache-2.0](LICENSE). Copyright 2026 NotXf1le. diff --git a/benchmarks/README.md b/benchmarks/README.md index c6073cb..e140136 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -1,41 +1,103 @@ # Benchmarks -`data/semif-authored144.jsonl` is an exact copy of SemIf's official [`benchmarks/data/authored144.jsonl`](https://github.com/TheoLeeCJ/SemIf/blob/b9cb32537e78be65f19abfcb1de8fc504b627d84/benchmarks/data/authored144.jsonl) at commit `b9cb32537e78be65f19abfcb1de8fc504b627d84`. Only the local filename differs. +## SuperGPQA -The 144 examples were authored by the SemIf project. The dataset is distributed under SemIf's MIT license, reproduced in `data/SEMIF-LICENSE.txt`. +The chart uses a deterministic 1,000-question sample stratified by discipline and +difficulty. The random-choice baseline is the mean of `1 / number of choices` across +the sample. OpenRouter models are included only when they return `top_logprobs`. +The evaluation sample excludes a deterministic 100-question pilot used to select +working model and provider pairs. -Build the package before running a benchmark: +### Prepare the dataset ```sh -npm ci +hf download m-a-p/SuperGPQA SuperGPQA-all.jsonl \ + --repo-type dataset \ + --revision 4430d4458112c7d4497fdcf94d7cc223313d6acf \ + --local-dir benchmarks/.data/supergpqa/source +node benchmarks/prepare-supergpqa.mjs npm run build ``` -Run the published llama.cpp adapter against a local server: +### Run the benchmark + +```sh +OPENROUTER_API_KEY=... node benchmarks/run-supergpqa.mjs \ + --backend openrouter \ + --model ibm-granite/granite-4.0-h-micro \ + --provider cloudflare \ + --sample-method proportional \ + --sample-size 1000 \ + --output benchmarks/results/supergpqa-granite-4.0-h-micro-cloudflare.json +``` + +The chart uses these OpenRouter model and provider pairs: + +| Model | Provider | Result | +|---|---|---| +| `ibm-granite/granite-4.0-h-micro` | `cloudflare` | `supergpqa-granite-4.0-h-micro-cloudflare.json` | +| `meta-llama/llama-3.1-8b-instruct` | `novita` | `supergpqa-llama-3.1-8b-novita.json` | +| `z-ai/glm-4.7-flash` | `cloudflare` | `supergpqa-glm-4.7-flash-cloudflare.json` | +| `z-ai/glm-5.2` | `cloudflare` | `supergpqa-glm-5.2-cloudflare.json` | +| `google/gemma-4-26b-a4b-it` | `dekallm` | `supergpqa-gemma-4-26b-dekallm.json` | +| `ibm-granite/granite-4.2-8b` | `coreweave` | `supergpqa-granite-4.2-8b-coreweave.json` | +| `deepseek/deepseek-v4.1-flash` | `wafer` | `supergpqa-deepseek-v4.1-flash-wafer.json` | +| `deepseek/deepseek-v4-pro-0813` | `cloudflare` | `supergpqa-deepseek-v4-pro-cloudflare.json` | +| `moonshotai/kimi-k3` | `morph` | `supergpqa-kimi-k3-morph.json` | + +Run Jev on the same sample: + +```sh +OPENROUTER_API_KEY=... node benchmarks/run-supergpqa.mjs \ + --backend jev \ + --sample-method proportional \ + --sample-size 1000 \ + --output benchmarks/results/supergpqa-jev-1.13.json +``` + +Run a local model through llama.cpp: + +```sh +node benchmarks/run-supergpqa.mjs \ + --backend llama-cpp \ + --base-url http://127.0.0.1:8080 \ + --model qwen3.8-27b-text-64k \ + --sample-method proportional \ + --sample-size 1000 \ + --output benchmarks/results/supergpqa-qwen3.8-27b-local.json +``` + +The X axis is average cost per decision multiplied by seconds per decision. The local +model is shown as a horizontal accuracy line. ```sh +node benchmarks/generate-supergpqa-chart.mjs +``` + +## SemIf + +`data/semif-authored144.jsonl` is SemIf's official +[`benchmarks/data/authored144.jsonl`](https://github.com/TheoLeeCJ/SemIf/blob/b9cb32537e78be65f19abfcb1de8fc504b627d84/benchmarks/data/authored144.jsonl) +at commit `b9cb32537e78be65f19abfcb1de8fc504b627d84`. The examples were authored by the +SemIf project and are distributed under its MIT license, reproduced in +`data/SEMIF-LICENSE.txt`. + +```sh +npm ci +npm run build + node benchmarks/run-semif.mjs \ --mode labels \ --base-url http://127.0.0.1:11434/ \ --model qwen3.8-27b-text-64k \ --output benchmarks/results/semif-qwen-labels.json -``` - -Run the Jev comparison with an OpenRouter API key: -```sh OPENROUTER_API_KEY=... node benchmarks/run-semif-openrouter-jev.mjs \ --model typesafe/jev-1.13 \ --output benchmarks/results/semif-jev-1.13.json -``` -Compare the complete distributions: - -```sh node benchmarks/compare-semif.mjs \ --qwen benchmarks/results/semif-qwen-labels.json \ --jev benchmarks/results/semif-jev-1.13.json \ --output benchmarks/results/semif-comparison.json ``` - -Benchmark result files are ignored because they can contain environment-specific timing and provider metadata. diff --git a/benchmarks/generate-supergpqa-chart.mjs b/benchmarks/generate-supergpqa-chart.mjs new file mode 100644 index 0000000..c151060 --- /dev/null +++ b/benchmarks/generate-supergpqa-chart.mjs @@ -0,0 +1,344 @@ +import { createHash } from "node:crypto"; +import { readFile, writeFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import { + SUPERGPQA_EVALUATION_SAMPLE_SEED, + SUPERGPQA_PILOT_ROWS, + SUPERGPQA_PREPARED_SHA256, + sampleSuperGpqaEvaluationRows, +} from "./prepare-supergpqa.mjs"; + +const EVALUATION_ROWS = 1000; +const benchmarksDirectory = new URL("./", import.meta.url); +const runInputs = [ + { + file: "supergpqa-granite-4.0-h-micro-cloudflare.json", + id: "granite-4.0-h-micro", + label: "Granite 4.0 H Micro", + series: "openrouter", + model: "ibm-granite/granite-4.0-h-micro", + provider: "cloudflare", + resolvedProvider: "Cloudflare", + }, + { + file: "supergpqa-llama-3.1-8b-novita.json", + id: "llama-3.1-8b", + label: "Llama 3.1 8B", + series: "openrouter", + model: "meta-llama/llama-3.1-8b-instruct", + provider: "novita", + resolvedProvider: "Novita", + labelDy: 27, + }, + { + file: "supergpqa-glm-4.7-flash-cloudflare.json", + id: "glm-4.7-flash", + label: "GLM 4.7 Flash", + series: "openrouter", + model: "z-ai/glm-4.7-flash", + provider: "cloudflare", + resolvedProvider: "Cloudflare", + }, + { + file: "supergpqa-glm-5.2-cloudflare.json", + id: "glm-5.2", + label: "GLM 5.2", + series: "openrouter", + model: "z-ai/glm-5.2", + provider: "cloudflare", + resolvedProvider: "Cloudflare", + }, + { + file: "supergpqa-gemma-4-26b-dekallm.json", + id: "gemma-4-26b", + label: "Gemma 4 26B", + series: "openrouter", + model: "google/gemma-4-26b-a4b-it", + provider: "dekallm", + resolvedProvider: "DekaLLM", + }, + { + file: "supergpqa-granite-4.2-8b-coreweave.json", + id: "granite-4.2-8b", + label: "Granite 4.2 8B", + series: "openrouter", + model: "ibm-granite/granite-4.2-8b", + provider: "coreweave", + resolvedProvider: "CoreWeave", + labelDy: 22, + }, + { + file: "supergpqa-deepseek-v4.1-flash-wafer.json", + id: "deepseek-v4.1-flash", + label: "DeepSeek V4.1 Flash", + series: "openrouter", + model: "deepseek/deepseek-v4.1-flash", + provider: "wafer", + resolvedProvider: "Wafer", + }, + { + file: "supergpqa-deepseek-v4-pro-cloudflare.json", + id: "deepseek-v4-pro", + label: "DeepSeek V4 Pro", + series: "openrouter", + model: "deepseek/deepseek-v4-pro-0813", + provider: "cloudflare", + resolvedProvider: "Cloudflare", + labelDy: 27, + }, + { + file: "supergpqa-kimi-k3-morph.json", + id: "kimi-k3", + label: "Kimi K3", + series: "openrouter", + model: "moonshotai/kimi-k3", + provider: "morph", + resolvedProvider: "Morph", + labelDx: -10, + labelAnchor: "end", + }, + { + file: "supergpqa-jev-1.13.json", + id: "jev-1.13", + label: "Jev 1.13", + series: "jev", + model: "typesafe/jev-1.13", + }, +]; +const localReferenceInput = { + file: "supergpqa-qwen3.8-27b-local.json", + id: "qwen3.8-27b-local", + label: "Local Qwen3.8 27B Q4_XL", + series: "llama-cpp", + model: "qwen3.8-27b-text-64k", +}; + +function invariant(condition, message) { + if (!condition) throw new Error(message); +} + +function formatNumber(value, digits = 2) { + return Number(value.toFixed(digits)).toString(); +} + +function wilson95(correct, total) { + const z = 1.959963984540054; + const p = correct / total; + const z2 = z ** 2; + const denominator = 1 + z2 / total; + const center = (p + z2 / (2 * total)) / denominator; + const margin = (z / denominator) + * Math.sqrt((p * (1 - p) + z2 / (4 * total)) / total); + return [center - margin, center + margin]; +} + +function median(values) { + const sorted = [...values].sort((a, b) => a - b); + const middle = Math.floor(sorted.length / 2); + return sorted.length % 2 === 0 + ? (sorted[middle - 1] + sorted[middle]) / 2 + : sorted[middle]; +} + +async function loadRun(input) { + const report = JSON.parse( + await readFile(new URL(`results/${input.file}`, benchmarksDirectory), "utf8"), + ); + invariant( + report.dataset?.selectedRows === EVALUATION_ROWS, + `${input.file}: expected ${EVALUATION_ROWS} selected rows`, + ); + invariant(report.dataset?.sample?.method === "proportional-discipline-difficulty", + `${input.file}: unexpected sample method`); + invariant(report.dataset?.sha256 === SUPERGPQA_PREPARED_SHA256, + `${input.file}: unexpected prepared dataset hash`); + invariant(report.dataset?.sample?.seed === SUPERGPQA_EVALUATION_SAMPLE_SEED, + `${input.file}: unexpected sample seed`); + invariant( + report.dataset?.sample?.excludedPilotSampleSize === SUPERGPQA_PILOT_ROWS, + `${input.file}: unexpected excluded pilot sample size`, + ); + invariant( + report.summary?.rowsAttempted === EVALUATION_ROWS, + `${input.file}: expected ${EVALUATION_ROWS} attempted rows`, + ); + invariant( + report.summary?.rowsSuccessful === EVALUATION_ROWS, + `${input.file}: expected ${EVALUATION_ROWS} successful rows`, + ); + invariant(report.summary?.errors === 0, `${input.file}: expected zero errors`); + invariant(report.summary?.costComplete === true, `${input.file}: incomplete cost data`); + invariant(Number.isFinite(report.summary?.costUsd) + && (input.series === "llama-cpp" ? report.summary.costUsd === 0 : report.summary.costUsd > 0), + `${input.file}: invalid cost`); + invariant(Number.isFinite(report.summary?.decisionsPerSecond) && report.summary.decisionsPerSecond > 0, + `${input.file}: invalid throughput`); + invariant(Array.isArray(report.results) && report.results.length === EVALUATION_ROWS, + `${input.file}: expected ${EVALUATION_ROWS} result rows`); + invariant(report.runtime?.model === input.model, `${input.file}: unexpected model`); + invariant(report.runtime?.backend === input.series, `${input.file}: unexpected backend`); + if (input.series === "openrouter") { + invariant(report.runtime?.provider === input.provider, `${input.file}: unexpected provider`); + invariant(report.results.every((row) => row.resolvedModel === input.model), + `${input.file}: OpenRouter resolved a different model`); + invariant(report.results.every((row) => row.resolvedProvider === input.resolvedProvider), + `${input.file}: OpenRouter resolved a different provider`); + } + invariant(report.results.every((row) => Number.isFinite(row.latencyMs) && row.latencyMs > 0), + `${input.file}: invalid latency`); + const correct = report.results.filter((row) => row.correct === true).length; + invariant(report.summary.endToEndAccuracy === correct / EVALUATION_ROWS, + `${input.file}: accuracy does not match result rows`); + return { + sha256: report.dataset.sha256, + ids: report.results.map((row) => row.id), + run: { + id: input.id, + label: input.label, + series: input.series, + model: input.model, + ...(input.provider === undefined ? {} : { provider: input.provider }), + ...(input.resolvedProvider === undefined ? {} : { resolvedProvider: input.resolvedProvider }), + ...(input.labelDx === undefined ? {} : { labelDx: input.labelDx }), + ...(input.labelDy === undefined ? {} : { labelDy: input.labelDy }), + ...(input.labelAnchor === undefined ? {} : { labelAnchor: input.labelAnchor }), + correct, + totalCostUsd: report.summary.costUsd, + medianLatencyMs: median(report.results.map((row) => row.latencyMs)), + decisionsPerSecond: report.summary.decisionsPerSecond, + }, + }; +} + +function renderSvg(aggregate) { + const width = 920; + const height = 540; + const plot = { left: 82, right: 882, top: 76, bottom: 430 }; + const points = aggregate.runs.map((run) => ({ + ...run, + xValue: (run.totalCostUsd / aggregate.dataset.rows) * (1 / run.decisionsPerSecond), + yValue: run.correct / aggregate.dataset.rows, + })); + const logMin = -6.2; + const logMax = -2.7; + const yMax = 0.7; + const scaleX = (value) => plot.left + ((Math.log10(value) - logMin) / (logMax - logMin)) + * (plot.right - plot.left); + const scaleY = (value) => plot.bottom - (value / yMax) * (plot.bottom - plot.top); + const xTicks = [0.000001, 0.00001, 0.0001, 0.001]; + const yTicks = [0, 0.2, 0.4, 0.6]; + const lines = []; + + lines.push(``); + lines.push(" SuperGPQA benchmark: ChooseKit models and Jev"); + lines.push(` Accuracy on the same ${aggregate.dataset.rows} proportionally sampled questions. Only OpenRouter models with top_logprobs are included. Lower cost per decision times seconds per decision and higher accuracy are better. Vertical bars show 95 percent Wilson intervals. A horizontal line shows local Qwen3.8 27B Q4_XL accuracy without assigning it a cloud cost coordinate.`); + lines.push(` `); + lines.push(" "); + lines.push(` SuperGPQA decision benchmark`); + lines.push(` proportional sample · ${aggregate.dataset.rows} questions · ChooseKit OpenRouter providers pinned · 2026-09-21`); + lines.push(` ChooseKit + llama.cpp`); + lines.push(` ChooseKit + OpenRouter`); + lines.push(` Jev`); + + for (const tick of yTicks) { + const y = scaleY(tick); + lines.push(` `); + lines.push(` ${Math.round(tick * 100)}%`); + } + for (const tick of xTicks) { + const x = scaleX(tick); + lines.push(` `); + lines.push(` ${tick.toExponential(0)}`); + } + const baselineY = scaleY(aggregate.dataset.randomBaseline); + lines.push(` `); + lines.push(` random choice ${formatNumber(aggregate.dataset.randomBaseline * 100, 1)}%`); + const localY = scaleY(aggregate.localReference.correct / aggregate.dataset.rows); + const [localLower, localUpper] = wilson95( + aggregate.localReference.correct, + aggregate.dataset.rows, + ); + const localUpperY = scaleY(localUpper); + const localLowerY = scaleY(localLower); + lines.push(` `); + lines.push(` `); + lines.push(` ${aggregate.localReference.label} · ${formatNumber((aggregate.localReference.correct / aggregate.dataset.rows) * 100, 1)}%`); + lines.push(` `); + lines.push(` Cost–latency product ↓ (log scale)`); + lines.push(` Only OpenRouter models with top_logprobs are included · measured end to end · bars and green band show 95% Wilson intervals`); + lines.push(` Accuracy →`); + + for (const point of points) { + const x = scaleX(point.xValue); + const y = scaleY(point.yValue); + const color = point.series === "jev" ? "#e85d3f" : "#356ae6"; + const [lower, upper] = wilson95(point.correct, aggregate.dataset.rows); + const upperY = scaleY(upper); + const lowerY = scaleY(lower); + lines.push(` `); + lines.push(` `); + lines.push(` `); + if (point.series === "jev") { + lines.push(` `); + } else { + lines.push(` `); + } + const labelX = x + (point.labelDx ?? 0); + const labelY = y + (point.labelDy ?? -16); + lines.push(` ${point.label} · ${formatNumber(point.yValue * 100, 1)}%`); + } + lines.push(" "); + lines.push(""); + return `${lines.join("\n")}\n`; +} + +const loadedRuns = await Promise.all(runInputs.map(loadRun)); +const localReference = await loadRun(localReferenceInput); +const allRuns = [...loadedRuns, localReference]; +invariant(new Set(allRuns.map(({ sha256 }) => sha256)).size === 1, + "all reports must use the same dataset hash"); +const expectedIds = JSON.stringify(loadedRuns[0].ids); +invariant(allRuns.every(({ ids }) => JSON.stringify(ids) === expectedIds), + "all reports must contain the same questions in the same order"); + +const preparedDataset = await readFile( + new URL(".data/supergpqa/supergpqa.jsonl", benchmarksDirectory), +); +const preparedSha256 = createHash("sha256").update(preparedDataset).digest("hex"); +invariant(preparedSha256 === SUPERGPQA_PREPARED_SHA256, + "prepared dataset has an unexpected hash"); +const preparedRows = preparedDataset.toString("utf8").trim() + .split(/\r?\n/).map((line) => JSON.parse(line)); +const rowsById = new Map(preparedRows.map((row) => [row.id, row])); +const evaluationRows = loadedRuns[0].ids.map((id) => rowsById.get(id)); +invariant(evaluationRows.every(Boolean), "result IDs must exist in the prepared dataset"); +const expectedEvaluationIds = sampleSuperGpqaEvaluationRows( + preparedRows, + EVALUATION_ROWS, + SUPERGPQA_PILOT_ROWS, +).map(({ id }) => id); +invariant(JSON.stringify(loadedRuns[0].ids) === JSON.stringify(expectedEvaluationIds), + "reports must contain the expected evaluation sample in order"); +const randomBaseline = evaluationRows.reduce( + (sum, row) => sum + 1 / Object.keys(row.choices).length, + 0, +) / evaluationRows.length; + +const aggregate = { + dataset: { + rows: EVALUATION_ROWS, + sha256: loadedRuns[0].sha256, + randomBaseline, + sample: "proportional-discipline-difficulty", + }, + runs: loadedRuns.map(({ run }) => run), + localReference: localReference.run, +}; + +await writeFile( + new URL("supergpqa-benchmark.json", benchmarksDirectory), + `${JSON.stringify(aggregate, null, 2)}\n`, +); +await writeFile(new URL("supergpqa-benchmark.svg", benchmarksDirectory), renderSvg(aggregate)); +console.log(`Wrote ${fileURLToPath(new URL("supergpqa-benchmark.json", benchmarksDirectory))}`); +console.log(`Wrote ${fileURLToPath(new URL("supergpqa-benchmark.svg", benchmarksDirectory))}`); diff --git a/benchmarks/prepare-supergpqa.mjs b/benchmarks/prepare-supergpqa.mjs new file mode 100644 index 0000000..44edd16 --- /dev/null +++ b/benchmarks/prepare-supergpqa.mjs @@ -0,0 +1,212 @@ +import { createHash } from "node:crypto"; +import { createReadStream, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { dirname } from "node:path"; +import { pathToFileURL } from "node:url"; + +export const SUPERGPQA_REVISION = "4430d4458112c7d4497fdcf94d7cc223313d6acf"; +export const SUPERGPQA_SHA256 = "28b998e70205ee95e540317b5adc06a06552a3961fb50b153df126b833f7a910"; +export const SUPERGPQA_PREPARED_SHA256 = "d17a48292cc8f081ca576afc7e8eaa6ca510de59fa9881282922409666bb3aff"; +export const SUPERGPQA_ROWS = 26529; +export const SUPERGPQA_PILOT_SAMPLE_SEED = "choosekit-supergpqa-v1"; +export const SUPERGPQA_EVALUATION_SAMPLE_SEED = "choosekit-supergpqa-final-v2"; +export const SUPERGPQA_PILOT_ROWS = 100; + +function option(name, fallback) { + const index = process.argv.indexOf(name); + if (index === -1) return fallback; + const value = process.argv[index + 1]; + if (value === undefined || value.startsWith("--")) throw new TypeError(`${name} requires a value.`); + return value; +} + +function requireText(value, name, id) { + if (typeof value !== "string" || value.trim().length === 0) { + throw new TypeError(`SuperGPQA row ${id} has an invalid ${name}.`); + } + return value.trim(); +} + +function stableRank(value, seed = SUPERGPQA_PILOT_SAMPLE_SEED) { + return createHash("sha256").update(`${seed}\0${value}`).digest("hex"); +} + +export function prepareSuperGpqaRows(rows) { + if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array."); + const ids = new Set(); + return Object.freeze(rows.map((row, index) => { + if (row === null || typeof row !== "object" || Array.isArray(row)) { + throw new TypeError(`SuperGPQA row ${index + 1} is not an object.`); + } + const id = requireText(row.uuid, "uuid", index + 1); + if (ids.has(id)) throw new Error(`SuperGPQA contains duplicate row ID ${id}.`); + ids.add(id); + const question = requireText(row.question, "question", id); + if (!Array.isArray(row.options) || row.options.length < 2 || row.options.length > 20) { + throw new TypeError(`SuperGPQA row ${id} has invalid options.`); + } + const answerLetter = requireText(row.answer_letter, "answer_letter", id); + const answerIndex = answerLetter.charCodeAt(0) - 65; + if (!/^[A-T]$/.test(answerLetter) || answerIndex >= row.options.length) { + throw new Error(`SuperGPQA row ${id} has an answer outside its options.`); + } + const choices = {}; + for (const [optionIndex, value] of row.options.entries()) { + choices[String.fromCharCode(65 + optionIndex)] = requireText(value, `option ${optionIndex + 1}`, id); + } + const answer = requireText(row.answer, "answer", id); + if (choices[answerLetter] !== answer) { + throw new Error(`SuperGPQA row ${id} has inconsistent answer text and letter.`); + } + return Object.freeze({ + id, + context: "", + question, + choices: Object.freeze(choices), + gold: answerLetter, + metadata: Object.freeze({ + discipline: requireText(row.discipline, "discipline", id), + field: requireText(row.field, "field", id), + subfield: requireText(row.subfield, "subfield", id), + difficulty: requireText(row.difficulty, "difficulty", id), + isCalculation: row.is_calculation === true, + }), + }); + })); +} + +export function sampleSuperGpqaPilotRows(rows, size) { + if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array."); + if (!Number.isSafeInteger(size) || size < 1 || size > rows.length) { + throw new TypeError("SuperGPQA sample size must be a positive integer no larger than the dataset."); + } + const groups = new Map(); + for (const row of rows) { + const key = `${row.metadata?.discipline}\0${row.metadata?.difficulty}`; + const group = groups.get(key) ?? []; + group.push(row); + groups.set(key, group); + } + const orderedGroups = [...groups.entries()] + .sort(([a], [b]) => stableRank(a).localeCompare(stableRank(b))) + .map(([, group]) => [...group].sort((a, b) => stableRank(a.id).localeCompare(stableRank(b.id)))); + const selected = []; + for (let round = 0; selected.length < size; round++) { + let added = 0; + for (const group of orderedGroups) { + if (selected.length === size) break; + if (round < group.length) { + selected.push(group[round]); + added++; + } + } + if (added === 0) break; + } + return selected; +} + +export function sampleSuperGpqaStratifiedRows( + rows, + size, + seed = SUPERGPQA_EVALUATION_SAMPLE_SEED, +) { + if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array."); + if (!Number.isSafeInteger(size) || size < 1 || size > rows.length) { + throw new TypeError("SuperGPQA sample size must be a positive integer no larger than the dataset."); + } + const groups = new Map(); + for (const row of rows) { + const key = `${row.metadata?.discipline}\0${row.metadata?.difficulty}`; + const group = groups.get(key) ?? []; + group.push(row); + groups.set(key, group); + } + const allocations = [...groups.entries()].map(([key, group]) => { + const exact = (size * group.length) / rows.length; + return { + key, + group: [...group].sort((a, b) => stableRank(a.id, seed).localeCompare(stableRank(b.id, seed))), + count: Math.floor(exact), + remainder: exact - Math.floor(exact), + }; + }); + let remaining = size - allocations.reduce((total, allocation) => total + allocation.count, 0); + allocations.sort((a, b) => b.remainder - a.remainder + || stableRank(a.key, seed).localeCompare(stableRank(b.key, seed))); + for (let index = 0; index < remaining; index++) allocations[index].count++; + return allocations + .flatMap(({ group, count }) => group.slice(0, count)) + .sort((a, b) => stableRank(a.id, seed).localeCompare(stableRank(b.id, seed))); +} + +export function sampleSuperGpqaEvaluationRows( + rows, + size, + pilotSampleSize = SUPERGPQA_PILOT_ROWS, +) { + if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array."); + if (!Number.isSafeInteger(pilotSampleSize) + || pilotSampleSize < 1 || pilotSampleSize >= rows.length) { + throw new TypeError( + "SuperGPQA pilot sample size must be a positive integer smaller than the dataset.", + ); + } + const pilotIds = new Set( + sampleSuperGpqaPilotRows(rows, pilotSampleSize).map(({ id }) => id), + ); + const evaluationPool = rows.filter(({ id }) => !pilotIds.has(id)); + return sampleSuperGpqaStratifiedRows( + evaluationPool, + size, + SUPERGPQA_EVALUATION_SAMPLE_SEED, + ); +} + +async function sha256(path) { + const hash = createHash("sha256"); + for await (const chunk of createReadStream(path)) hash.update(chunk); + return hash.digest("hex"); +} + +export async function prepareSuperGpqaFile({ input, output, audit }) { + if (existsSync(output)) throw new Error(`Refusing to overwrite existing output: ${output}`); + if (existsSync(audit)) throw new Error(`Refusing to overwrite existing audit: ${audit}`); + const inputSha256 = await sha256(input); + if (inputSha256 !== SUPERGPQA_SHA256) { + throw new Error(`Unexpected SuperGPQA SHA-256: ${inputSha256}. Expected ${SUPERGPQA_SHA256}.`); + } + const text = readFileSync(input, "utf8").trim(); + const sourceRows = text.length === 0 ? [] : text.split(/\r?\n/).map((line) => JSON.parse(line)); + const rows = prepareSuperGpqaRows(sourceRows); + if (rows.length !== SUPERGPQA_ROWS) { + throw new Error(`Expected ${SUPERGPQA_ROWS} SuperGPQA rows, received ${rows.length}.`); + } + const counts = {}; + for (const row of rows) { + const key = `${row.metadata.discipline} / ${row.metadata.difficulty}`; + counts[key] = (counts[key] ?? 0) + 1; + } + const report = { + dataset: { + source: "https://huggingface.co/datasets/m-a-p/SuperGPQA", + revision: SUPERGPQA_REVISION, + path: input, + sha256: inputSha256, + }, + summary: { total: rows.length, strata: counts }, + }; + mkdirSync(dirname(output), { recursive: true }); + mkdirSync(dirname(audit), { recursive: true }); + writeFileSync(output, `${rows.map((row) => JSON.stringify(row)).join("\n")}\n`); + writeFileSync(audit, `${JSON.stringify(report, null, 2)}\n`); + return Object.freeze(report); +} + +async function main() { + const input = option("--input", "benchmarks/.data/supergpqa/source/SuperGPQA-all.jsonl"); + const output = option("--output", "benchmarks/.data/supergpqa/supergpqa.jsonl"); + const audit = option("--audit", "benchmarks/.data/supergpqa/supergpqa.audit.json"); + console.log(JSON.stringify(await prepareSuperGpqaFile({ input, output, audit }), null, 2)); +} + +const entry = process.argv[1] === undefined ? undefined : pathToFileURL(process.argv[1]).href; +if (entry === import.meta.url) await main(); diff --git a/benchmarks/run-semif-openrouter-jev.mjs b/benchmarks/run-semif-openrouter-jev.mjs index 0688694..d97648a 100644 --- a/benchmarks/run-semif-openrouter-jev.mjs +++ b/benchmarks/run-semif-openrouter-jev.mjs @@ -130,7 +130,7 @@ function validateAnswer(answer, optionIds) { } const input = option("--input", "benchmarks/data/semif-authored144.jsonl"); -const output = option("--output", "benchmarks/results/semif-openrouter-jev-latest.json"); +const output = option("--output", "benchmarks/results/semif-jev-1.13.json"); const model = option("--model", "typesafe/jev-1.13"); const rawLimit = option("--limit", undefined); const limit = rawLimit === undefined ? undefined : Number.parseInt(rawLimit, 10); diff --git a/benchmarks/run-semif.mjs b/benchmarks/run-semif.mjs index 67fa26a..3e49a8e 100644 --- a/benchmarks/run-semif.mjs +++ b/benchmarks/run-semif.mjs @@ -94,7 +94,7 @@ const mode = option("--mode", "labels"); if (mode !== "labels" && mode !== "minimal-prefix") { throw new TypeError("--mode must be labels or minimal-prefix."); } -const output = option("--output", `benchmarks/results/semif-qwen3.8-27b-production-${mode}.json`); +const output = option("--output", `benchmarks/results/semif-qwen3.8-27b-${mode}.json`); const baseURL = option("--base-url", process.env.LLAMA_CPP_BASE_URL ?? "http://127.0.0.1:11434/"); const model = option("--model", process.env.LLAMA_CPP_MODEL ?? "qwen3.8-27b-text-64k"); diff --git a/benchmarks/run-supergpqa.mjs b/benchmarks/run-supergpqa.mjs new file mode 100644 index 0000000..c62c3d0 --- /dev/null +++ b/benchmarks/run-supergpqa.mjs @@ -0,0 +1,366 @@ +import { createHash } from "node:crypto"; +import { existsSync, mkdirSync, readFileSync, renameSync, writeFileSync } from "node:fs"; +import { dirname } from "node:path"; +import { performance } from "node:perf_hooks"; +import { setTimeout as delay } from "node:timers/promises"; +import { fromLlamaCpp } from "../dist/esm/llama-cpp.js"; +import { fromOpenRouter } from "../dist/esm/openrouter.js"; +import { + SUPERGPQA_REVISION, + SUPERGPQA_PILOT_SAMPLE_SEED, + SUPERGPQA_EVALUATION_SAMPLE_SEED, + SUPERGPQA_PILOT_ROWS, + SUPERGPQA_PREPARED_SHA256, + SUPERGPQA_ROWS, + sampleSuperGpqaPilotRows, + sampleSuperGpqaEvaluationRows, +} from "./prepare-supergpqa.mjs"; + +const MAX_429_ATTEMPTS = 4; + +function option(name, fallback) { + const index = process.argv.indexOf(name); + if (index === -1) return fallback; + const value = process.argv[index + 1]; + if (value === undefined || value.startsWith("--")) throw new TypeError(`${name} requires a value.`); + return value; +} + +function flag(name) { + return process.argv.includes(name); +} + +function requireText(value, name) { + if (typeof value !== "string" || value.trim().length === 0) { + throw new TypeError(`${name} must be a non-empty string.`); + } + return value; +} + +function readCost(value) { + const cost = value?.cost; + return typeof cost === "number" && Number.isFinite(cost) && cost >= 0 ? cost : null; +} + +function validateRow(row, index) { + const optionIds = row?.choices !== null && typeof row?.choices === "object" && !Array.isArray(row.choices) + ? Object.keys(row.choices) : []; + if (row === null || typeof row !== "object" || Array.isArray(row) + || typeof row.id !== "string" || row.id.length === 0 || row.context !== "" + || typeof row.question !== "string" || row.question.length === 0 + || optionIds.length < 2 || optionIds.length > 20 + || typeof row.gold !== "string" || !optionIds.includes(row.gold) + || typeof row.metadata?.discipline !== "string" || typeof row.metadata?.difficulty !== "string") { + throw new TypeError(`SuperGPQA row ${row?.id ?? index + 1} is invalid.`); + } +} + +function summarize(results, elapsedSeconds) { + const successful = results.filter((row) => row.error === undefined); + const correct = successful.filter((row) => row.correct).length; + const costComplete = results.every((row) => typeof row.costUsd === "number"); + return { + rowsAttempted: results.length, + rowsSuccessful: successful.length, + errors: results.length - successful.length, + endToEndAccuracy: results.length === 0 ? null : correct / results.length, + wallSeconds: elapsedSeconds, + decisionsPerSecond: elapsedSeconds === 0 ? null : results.length / elapsedSeconds, + costUsd: costComplete ? results.reduce((total, row) => total + row.costUsd, 0) : null, + costComplete, + retries: results.reduce((total, row) => total + (row.retryCount ?? 0), 0), + }; +} + +const backend = option("--backend", undefined); +if (backend !== "openrouter" && backend !== "llama-cpp" && backend !== "jev") { + throw new TypeError("--backend must be openrouter, llama-cpp, or jev."); +} +const inputPath = option("--input", "benchmarks/.data/supergpqa/supergpqa.jsonl"); +const sampleSizeArgument = option("--sample-size", "100"); +const sampleSize = Number(sampleSizeArgument); +const sampleMethod = option("--sample-method", "balanced"); +if (sampleMethod !== "balanced" && sampleMethod !== "proportional") { + throw new TypeError("--sample-method must be balanced or proportional."); +} +const model = backend === "jev" + ? option("--model", "typesafe/jev-1.13") + : requireText(option("--model", process.env.OPENROUTER_MODEL), "--model"); +const provider = backend === "openrouter" + ? requireText(option("--provider", process.env.OPENROUTER_PROVIDER), "--provider") + : undefined; +const baseURL = backend === "llama-cpp" + ? requireText(option("--base-url", process.env.LLAMA_CPP_BASE_URL), "--base-url") + : undefined; +const recordedProvider = provider ?? (backend === "llama-cpp" ? "local" : "openrouter"); +const modelSlug = model.replace(/[^a-zA-Z0-9._-]+/g, "-"); +const providerSlug = recordedProvider.replace(/[^a-zA-Z0-9._-]+/g, "-"); +const outputPath = option("--output", + `benchmarks/results/supergpqa-${backend}-${providerSlug}-${modelSlug}-${sampleMethod}-sample${sampleSize}.json`); +const resume = flag("--resume"); +const apiKey = process.env.OPENROUTER_API_KEY; + +if (backend !== "llama-cpp" && (typeof apiKey !== "string" || apiKey.length < 20)) { + throw new Error("OPENROUTER_API_KEY is missing or invalid."); +} +if (existsSync(outputPath) && !resume) { + throw new Error(`Refusing to overwrite existing output: ${outputPath}`); +} +if (!existsSync(outputPath) && resume) { + throw new Error(`Cannot resume missing output: ${outputPath}`); +} +if (!Number.isSafeInteger(sampleSize) || sampleSize < 1) { + throw new TypeError("--sample-size must be a positive integer."); +} +if (sampleMethod === "balanced" && sampleSize > SUPERGPQA_PILOT_ROWS) { + throw new TypeError(`--sample-size must not exceed ${SUPERGPQA_PILOT_ROWS} for balanced sampling.`); +} + +const preparedDataset = readFileSync(inputPath); +const preparedSha256 = createHash("sha256").update(preparedDataset).digest("hex"); +if (preparedSha256 !== SUPERGPQA_PREPARED_SHA256) { + throw new Error(`Unexpected prepared SuperGPQA SHA-256: ${preparedSha256}. Expected ${SUPERGPQA_PREPARED_SHA256}.`); +} +const preparedText = preparedDataset.toString("utf8").trim(); +const preparedRows = preparedText.length === 0 + ? [] + : preparedText.split(/\r?\n/).map((line) => JSON.parse(line)); +if (preparedRows.length !== SUPERGPQA_ROWS) { + throw new Error(`Expected ${SUPERGPQA_ROWS} SuperGPQA rows, received ${preparedRows.length}.`); +} +const rowIds = new Set(); +for (const [index, row] of preparedRows.entries()) { + validateRow(row, index); + if (rowIds.has(row.id)) throw new Error(`SuperGPQA contains duplicate row ID ${row.id}.`); + rowIds.add(row.id); +} +const selectedRows = sampleMethod === "proportional" + ? sampleSuperGpqaEvaluationRows(preparedRows, sampleSize) + : sampleSuperGpqaPilotRows(preparedRows, sampleSize); +mkdirSync(dirname(outputPath), { recursive: true }); + +const packageVersion = JSON.parse( + readFileSync(new URL("../package.json", import.meta.url), "utf8"), +).version; +const sampleMetadata = sampleMethod === "proportional" + ? { + method: "proportional-discipline-difficulty", + seed: SUPERGPQA_EVALUATION_SAMPLE_SEED, + excludedPilotSampleSize: SUPERGPQA_PILOT_ROWS, + } + : { + method: "balanced-discipline-difficulty", + seed: SUPERGPQA_PILOT_SAMPLE_SEED, + }; +const benchmarkName = sampleMethod === "proportional" + ? "SuperGPQA proportional discipline/difficulty sample" + : "SuperGPQA balanced discipline/difficulty sample"; +const datasetMetadata = { + source: "https://huggingface.co/datasets/m-a-p/SuperGPQA", + revision: SUPERGPQA_REVISION, + sha256: preparedSha256, + totalRows: preparedRows.length, + selectedRows: selectedRows.length, + sample: sampleMetadata, +}; +const runtimeMetadata = { + backend, + model, + provider: recordedProvider, + ...(baseURL === undefined ? {} : { baseURL }), + packageVersion, +}; + +let responseMetadata = { costUsd: null, retryCount: 0 }; +const fetchWithMetadata = async (...args) => { + let accumulatedCost = 0; + let sawCost = false; + for (let attempt = 1; attempt <= MAX_429_ATTEMPTS; attempt++) { + const response = await globalThis.fetch(...args); + const body = await response.text(); + let value; + try { value = JSON.parse(body); } catch { value = undefined; } + const attemptCost = readCost(value?.usage); + if (attemptCost !== null) { + accumulatedCost += attemptCost; + sawCost = true; + } + responseMetadata = { + costUsd: sawCost ? accumulatedCost : null, + retryCount: attempt - 1, + ...(typeof value?.model === "string" ? { resolvedModel: value.model } : {}), + ...(typeof value?.provider === "string" ? { resolvedProvider: value.provider } : {}), + }; + if (response.status === 429 && attempt < MAX_429_ATTEMPTS) { + const retryAfter = Number(response.headers.get("retry-after")); + await delay(Number.isFinite(retryAfter) ? retryAfter * 1000 : 1000, undefined, + { signal: args[1]?.signal }); + continue; + } + return new Response(body, { status: response.status, statusText: response.statusText, headers: response.headers }); + } + throw new Error("OpenRouter 429 retry limit reached."); +}; + +const choose = backend === "openrouter" + ? fromOpenRouter({ apiKey, model, provider, fetch: fetchWithMetadata }) + : backend === "llama-cpp" + ? fromLlamaCpp({ baseURL, model, mode: "labels" }) + : undefined; + +async function decide(row) { + if (choose !== undefined) { + const decision = await choose({ + context: row.context, + question: row.question, + choices: row.choices, + signal: AbortSignal.timeout(120_000), + }); + return { + predicted: decision.choice, + scores: decision.scores, + ...(backend === "llama-cpp" ? { costUsd: 0, retryCount: 0 } : responseMetadata), + }; + } + let response; + let body; + let accumulatedCost = 0; + let sawCost = false; + for (let attempt = 1; attempt <= MAX_429_ATTEMPTS; attempt++) { + response = await fetch("https://openrouter.ai/api/alpha/decisions", { + method: "POST", + headers: { + authorization: `Bearer ${apiKey}`, + "content-type": "application/json", + "x-openrouter-title": "choosekit SuperGPQA benchmark", + }, + body: JSON.stringify({ + model, + state: row.question, + questions: { + decision: { + type: "choice", + instructions: "Which answer choice is correct?", + criteria: row.choices, + }, + }, + }), + signal: AbortSignal.timeout(120_000), + }); + const responseText = await response.text(); + try { body = JSON.parse(responseText); } catch { body = undefined; } + const attemptCost = readCost(body?.usage); + if (attemptCost !== null) { + accumulatedCost += attemptCost; + sawCost = true; + } + responseMetadata = { + costUsd: sawCost ? accumulatedCost : null, + retryCount: attempt - 1, + }; + if (response.status !== 429 || attempt === MAX_429_ATTEMPTS) break; + const retryAfter = Number(response.headers.get("retry-after")); + await delay(Number.isFinite(retryAfter) ? retryAfter * 1000 : 1000); + } + if (!response.ok) throw new Error(`OpenRouter returned HTTP ${response.status}.`); + const answer = body?.answers?.decision; + if (answer === null || typeof answer !== "object" || !Object.hasOwn(row.choices, answer.choice)) { + throw new Error("Jev returned an invalid answer."); + } + return { predicted: answer.choice, ...responseMetadata }; +} + +let results = []; +let startedAt = new Date().toISOString(); +let elapsedBeforeSession = 0; + +function assertResumeMetadata(report) { + const expected = { + benchmark: benchmarkName, + dataset: datasetMetadata, + runtime: runtimeMetadata, + }; + for (const key of Object.keys(expected)) { + if (JSON.stringify(report[key]) !== JSON.stringify(expected[key])) { + throw new Error(`Cannot resume: ${key} metadata does not match the current run.`); + } + } + if (typeof report.startedAt !== "string" || !Number.isFinite(Date.parse(report.startedAt))) { + throw new Error("Cannot resume: checkpoint has an invalid startedAt."); + } + if (!Array.isArray(report.results) || report.results.length > selectedRows.length + || report.results.some((result, index) => result?.id !== selectedRows[index].id)) { + throw new Error("Cannot resume: checkpoint result IDs are not a prefix of the selected rows."); + } +} + +if (resume) { + const checkpoint = JSON.parse(readFileSync(outputPath, "utf8")); + assertResumeMetadata(checkpoint); + results = checkpoint.results; + startedAt = checkpoint.startedAt; + if (!Number.isFinite(checkpoint.summary?.wallSeconds) || checkpoint.summary.wallSeconds < 0) { + throw new Error("Cannot resume: checkpoint has invalid elapsed time."); + } + elapsedBeforeSession = checkpoint.summary.wallSeconds; +} + +const sessionStartedAt = performance.now(); + +function writeReport() { + const elapsedSeconds = elapsedBeforeSession + (performance.now() - sessionStartedAt) / 1000; + const report = { + benchmark: benchmarkName, + startedAt, + dataset: datasetMetadata, + runtime: runtimeMetadata, + summary: summarize(results, elapsedSeconds), + results, + }; + const temporaryOutput = `${outputPath}.tmp`; + writeFileSync(temporaryOutput, `${JSON.stringify(report, null, 2)}\n`); + renameSync(temporaryOutput, outputPath); + return report; +} + +for (let index = 0; index < selectedRows.length; index++) { + if (index < results.length && results[index].error === undefined) continue; + const row = selectedRows[index]; + const decisionStartedAt = performance.now(); + responseMetadata = backend === "llama-cpp" + ? { costUsd: 0, retryCount: 0 } + : { costUsd: null, retryCount: 0 }; + try { + const decision = await decide(row); + const result = { + id: row.id, + gold: row.gold, + predicted: decision.predicted, + correct: decision.predicted === row.gold, + latencyMs: performance.now() - decisionStartedAt, + discipline: row.metadata.discipline, + difficulty: row.metadata.difficulty, + ...decision, + }; + results[index] = result; + console.log(`[${index + 1}/${selectedRows.length}] ${row.id} ${result.correct ? "correct" : "wrong"}` + + ` ${result.latencyMs.toFixed(1)}ms`); + } catch (error) { + results[index] = { + id: row.id, + gold: row.gold, + latencyMs: performance.now() - decisionStartedAt, + discipline: row.metadata.discipline, + difficulty: row.metadata.difficulty, + ...responseMetadata, + error: error instanceof Error ? `${error.name}: ${error.message}` : String(error), + }; + console.error( + `[${index + 1}/${selectedRows.length}] ${row.id} ERROR ${results[index].error}`, + ); + } + writeReport(); +} + +const report = writeReport(); +console.log(JSON.stringify(report.summary, null, 2)); diff --git a/benchmarks/supergpqa-benchmark.json b/benchmarks/supergpqa-benchmark.json new file mode 100644 index 0000000..3b0fe71 --- /dev/null +++ b/benchmarks/supergpqa-benchmark.json @@ -0,0 +1,143 @@ +{ + "dataset": { + "rows": 1000, + "sha256": "d17a48292cc8f081ca576afc7e8eaa6ca510de59fa9881282922409666bb3aff", + "randomBaseline": 0.1051944444444431, + "sample": "proportional-discipline-difficulty" + }, + "runs": [ + { + "id": "granite-4.0-h-micro", + "label": "Granite 4.0 H Micro", + "series": "openrouter", + "model": "ibm-granite/granite-4.0-h-micro", + "provider": "cloudflare", + "resolvedProvider": "Cloudflare", + "correct": 193, + "totalCostUsd": 0.005256013000000002, + "medianLatencyMs": 316.5877500000206, + "decisionsPerSecond": 2.843996339426331 + }, + { + "id": "llama-3.1-8b", + "label": "Llama 3.1 8B", + "series": "openrouter", + "model": "meta-llama/llama-3.1-8b-instruct", + "provider": "novita", + "resolvedProvider": "Novita", + "labelDy": 27, + "correct": 190, + "totalCostUsd": 0.006144179999999994, + "medianLatencyMs": 625.0682000000379, + "decisionsPerSecond": 1.3287550260398036 + }, + { + "id": "glm-4.7-flash", + "label": "GLM 4.7 Flash", + "series": "openrouter", + "model": "z-ai/glm-4.7-flash", + "provider": "cloudflare", + "resolvedProvider": "Cloudflare", + "correct": 257, + "totalCostUsd": 0.017208291499999993, + "medianLatencyMs": 261.28954999995767, + "decisionsPerSecond": 1.1753601925513055 + }, + { + "id": "glm-5.2", + "label": "GLM 5.2", + "series": "openrouter", + "model": "z-ai/glm-5.2", + "provider": "cloudflare", + "resolvedProvider": "Cloudflare", + "correct": 446, + "totalCostUsd": 0.39322003999999994, + "medianLatencyMs": 1068.5187000000028, + "decisionsPerSecond": 0.69116943608323 + }, + { + "id": "gemma-4-26b", + "label": "Gemma 4 26B", + "series": "openrouter", + "model": "google/gemma-4-26b-a4b-it", + "provider": "dekallm", + "resolvedProvider": "DekaLLM", + "correct": 376, + "totalCostUsd": 0.0184062, + "medianLatencyMs": 384.1862999999921, + "decisionsPerSecond": 2.1881737925945197 + }, + { + "id": "granite-4.2-8b", + "label": "Granite 4.2 8B", + "series": "openrouter", + "model": "ibm-granite/granite-4.2-8b", + "provider": "coreweave", + "resolvedProvider": "CoreWeave", + "labelDy": 22, + "correct": 238, + "totalCostUsd": 0.029343299999999975, + "medianLatencyMs": 290.28174999999464, + "decisionsPerSecond": 3.1536941543271486 + }, + { + "id": "deepseek-v4.1-flash", + "label": "DeepSeek V4.1 Flash", + "series": "openrouter", + "model": "deepseek/deepseek-v4.1-flash", + "provider": "wafer", + "resolvedProvider": "Wafer", + "correct": 456, + "totalCostUsd": 0.054357200000000015, + "medianLatencyMs": 641.0851999999941, + "decisionsPerSecond": 1.3966703323131509 + }, + { + "id": "deepseek-v4-pro", + "label": "DeepSeek V4 Pro", + "series": "openrouter", + "model": "deepseek/deepseek-v4-pro-0813", + "provider": "cloudflare", + "resolvedProvider": "Cloudflare", + "labelDy": 27, + "correct": 432, + "totalCostUsd": 0.35875751999999944, + "medianLatencyMs": 787.4300499999663, + "decisionsPerSecond": 0.8806034365130117 + }, + { + "id": "kimi-k3", + "label": "Kimi K3", + "series": "openrouter", + "model": "moonshotai/kimi-k3", + "provider": "morph", + "resolvedProvider": "Morph", + "labelDx": -10, + "labelAnchor": "end", + "correct": 593, + "totalCostUsd": 0.6242506249999994, + "medianLatencyMs": 1296.1853999999585, + "decisionsPerSecond": 0.7251960952904043 + }, + { + "id": "jev-1.13", + "label": "Jev 1.13", + "series": "jev", + "model": "typesafe/jev-1.13", + "correct": 536, + "totalCostUsd": 0.024364368000000036, + "medianLatencyMs": 325.2531000000017, + "decisionsPerSecond": 2.8591025350335553 + } + ], + "localReference": { + "id": "qwen3.8-27b-local", + "label": "Local Qwen3.8 27B Q4_XL", + "series": "llama-cpp", + "model": "qwen3.8-27b-text-64k", + "correct": 400, + "totalCostUsd": 0, + "medianLatencyMs": 555.8330999999889, + "decisionsPerSecond": 1.342476004288219 + } +} diff --git a/benchmarks/supergpqa-benchmark.svg b/benchmarks/supergpqa-benchmark.svg new file mode 100644 index 0000000..d67780d --- /dev/null +++ b/benchmarks/supergpqa-benchmark.svg @@ -0,0 +1,87 @@ + + SuperGPQA benchmark: ChooseKit models and Jev + Accuracy on the same 1000 proportionally sampled questions. Only OpenRouter models with top_logprobs are included. Lower cost per decision times seconds per decision and higher accuracy are better. Vertical bars show 95 percent Wilson intervals. A horizontal line shows local Qwen3.8 27B Q4_XL accuracy without assigning it a cloud cost coordinate. + + + SuperGPQA decision benchmark + proportional sample · 1000 questions · ChooseKit OpenRouter providers pinned · 2026-09-21 + ChooseKit + llama.cpp + ChooseKit + OpenRouter + Jev + + 0% + + 20% + + 40% + + 60% + + 1e-6 + + 1e-5 + + 1e-4 + + 1e-3 + + random choice 10.5% + + + Local Qwen3.8 27B Q4_XL · 40% + + Cost–latency product ↓ (log scale) + Only OpenRouter models with top_logprobs are included · measured end to end · bars and green band show 95% Wilson intervals + Accuracy → + + + + + Granite 4.0 H Micro · 19.3% + + + + + Llama 3.1 8B · 19% + + + + + GLM 4.7 Flash · 25.7% + + + + + GLM 5.2 · 44.6% + + + + + Gemma 4 26B · 37.6% + + + + + Granite 4.2 8B · 23.8% + + + + + DeepSeek V4.1 Flash · 45.6% + + + + + DeepSeek V4 Pro · 43.2% + + + + + Kimi K3 · 59.3% + + + + + Jev 1.13 · 53.6% + + diff --git a/tests/benchmark.test.mjs b/tests/benchmark.test.mjs index bd7a1fa..f9e3f50 100644 --- a/tests/benchmark.test.mjs +++ b/tests/benchmark.test.mjs @@ -76,6 +76,19 @@ test("OpenRouter benchmark creates its output directory before inference", () => assert.match(report.results[0].error, /Intentional benchmark test failure/); }); +test("SuperGPQA pilot runs are limited to 100 questions", () => { + const result = run("run-supergpqa.mjs", [ + "--backend", "llama-cpp", + "--base-url", "http://127.0.0.1:8080", + "--model", "test-model", + "--sample-method", "balanced", + "--sample-size", "101", + ]); + + assert.notEqual(result.status, 0); + assert.match(result.stderr, /must not exceed 100 for balanced sampling/); +}); + test("comparator creates a nested output for valid reports", () => { const qwen = temporaryPath("qwen.json"); const jev = temporaryPath("jev.json"); diff --git a/tests/supergpqa.test.mjs b/tests/supergpqa.test.mjs new file mode 100644 index 0000000..d25ded9 --- /dev/null +++ b/tests/supergpqa.test.mjs @@ -0,0 +1,112 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { + prepareSuperGpqaRows, + sampleSuperGpqaPilotRows, + sampleSuperGpqaEvaluationRows, + sampleSuperGpqaStratifiedRows, +} from "../benchmarks/prepare-supergpqa.mjs"; + +function row(overrides = {}) { + return { + uuid: "case-1", + question: "Which answer is correct?", + options: ["First", "Second"], + answer: "Second", + answer_letter: "B", + discipline: "Science", + field: "Physics", + subfield: "Mechanics", + difficulty: "middle", + is_calculation: false, + ...overrides, + }; +} + +test("prepares choices and validates the answer letter against its text", () => { + const [prepared] = prepareSuperGpqaRows([row()]); + assert.deepEqual(prepared.choices, { A: "First", B: "Second" }); + assert.equal(prepared.gold, "B"); + assert.deepEqual(prepared.metadata, { + discipline: "Science", + field: "Physics", + subfield: "Mechanics", + difficulty: "middle", + isCalculation: false, + }); + assert.throws(() => prepareSuperGpqaRows([row({ answer: "First" })]), /inconsistent answer/); +}); + +test("samples discipline and difficulty strata deterministically", () => { + const source = []; + for (const discipline of ["Science", "Law"]) { + for (const difficulty of ["easy", "hard"]) { + for (let index = 0; index < 3; index++) { + source.push(row({ + uuid: `${discipline}-${difficulty}-${index}`, + discipline, + difficulty, + })); + } + } + } + const prepared = prepareSuperGpqaRows(source); + const sample = sampleSuperGpqaPilotRows(prepared, 8); + const counts = sample.reduce((map, item) => { + const key = `${item.metadata.discipline}/${item.metadata.difficulty}`; + map[key] = (map[key] ?? 0) + 1; + return map; + }, {}); + assert.deepEqual(Object.values(counts).sort(), [2, 2, 2, 2]); + assert.deepEqual( + sample.map(({ id }) => id), + sampleSuperGpqaPilotRows([...prepared].reverse(), 8).map(({ id }) => id), + ); +}); + +test("rejects malformed rows and sample sizes", () => { + assert.throws(() => prepareSuperGpqaRows([row(), row()]), /duplicate row ID/); + assert.throws(() => prepareSuperGpqaRows([row({ options: ["First"] })]), /invalid options/); + assert.throws(() => prepareSuperGpqaRows([row({ answer_letter: "C" })]), /outside its options/); + assert.throws( + () => sampleSuperGpqaPilotRows(prepareSuperGpqaRows([row()]), 2), + /sample size/, + ); +}); + +test("samples discipline and difficulty strata proportionally and deterministically", () => { + const source = []; + for (let index = 0; index < 2; index++) { + source.push(row({ uuid: `Law-hard-${index}`, discipline: "Law", difficulty: "hard" })); + } + for (let index = 0; index < 8; index++) { + source.push(row({ uuid: `Science-easy-${index}`, discipline: "Science", difficulty: "easy" })); + } + const prepared = prepareSuperGpqaRows(source); + const sample = sampleSuperGpqaStratifiedRows(prepared, 5); + assert.equal(sample.filter(({ metadata }) => metadata.discipline === "Law").length, 1); + assert.equal(sample.filter(({ metadata }) => metadata.discipline === "Science").length, 4); + assert.deepEqual( + sample.map(({ id }) => id), + sampleSuperGpqaStratifiedRows([...prepared].reverse(), 5).map(({ id }) => id), + ); +}); + +test("keeps the evaluation sample disjoint from the pilot sample", () => { + const source = []; + for (const discipline of ["Science", "Law"]) { + for (let index = 0; index < 10; index++) { + source.push(row({ uuid: `${discipline}-${index}`, discipline })); + } + } + const prepared = prepareSuperGpqaRows(source); + const pilotIds = new Set( + sampleSuperGpqaPilotRows(prepared, 4).map(({ id }) => id), + ); + const evaluation = sampleSuperGpqaEvaluationRows(prepared, 10, 4); + assert.equal(evaluation.some(({ id }) => pilotIds.has(id)), false); + assert.deepEqual( + evaluation.map(({ id }) => id), + sampleSuperGpqaEvaluationRows([...prepared].reverse(), 10, 4).map(({ id }) => id), + ); +});