diff --git a/.gitignore b/.gitignore
index 1533ed3..f01d2bf 100644
--- a/.gitignore
+++ b/.gitignore
@@ -5,4 +5,5 @@ dist/
!.env.example
coverage/
/docs/
+/benchmarks/.data/
/benchmarks/results/
diff --git a/README.md b/README.md
index 3cbe7f8..8d9b324 100644
--- a/README.md
+++ b/README.md
@@ -2,10 +2,37 @@
`choosekit` scores a finite set of choices with a language model and returns a typed decision with a probability distribution. It supports local llama.cpp models and an optional OpenRouter backend.
+
+
+The chart compares accuracy with a lower-is-better cost-latency product. The green line and confidence band show local Qwen3.8 27B Q4_XL accuracy; it has no cloud cost coordinate. [Method and reproduction](benchmarks/README.md#supergpqa)
+
+| Model | Accuracy | Cost / 1,000 decisions | Decisions/s |
+|---|---:|---:|---:|
+| Granite 4.0 H Micro | 19.3% | $0.0053 | 2.84 |
+| Llama 3.1 8B | 19.0% | $0.0061 | 1.33 |
+| GLM 4.7 Flash | 25.7% | $0.0172 | 1.18 |
+| Gemma 4 26B | 37.6% | $0.0184 | 2.19 |
+| Jev 1.13 | 53.6% | $0.0244 | 2.86 |
+| Granite 4.2 8B | 23.8% | $0.0293 | 3.15 |
+| DeepSeek V4.1 Flash | 45.6% | $0.0544 | 1.40 |
+| DeepSeek V4 Pro | 43.2% | $0.3588 | 0.88 |
+| GLM 5.2 | 44.6% | $0.3932 | 0.69 |
+| Kimi K3 | 59.3% | $0.6243 | 0.73 |
+
+## Install
+
+Library:
+
```sh
npm install choosekit
```
+MCP server:
+
+```sh
+npm install --global choosekit-mcp
+```
+
## Why
Agents often need to choose from known options:
@@ -18,13 +45,13 @@ Agents often need to choose from known options:
`choosekit` scores choices using the model's conditional log probabilities at the token branches that distinguish them.
-The project was inspired by [Jev and the System One model interface](https://typesafe.ai/blog/introducing-system-one-models-and-jev): application state in, typed probabilistic decisions out. Jev is a specialized hosted model. `choosekit` explores the same useful interface with a model you control. The llama.cpp backend keeps application state on infrastructure you choose; OpenRouter is available when a hosted model is more convenient.
+The project was inspired by [Jev and the System One model interface](https://typesafe.ai/blog/introducing-system-one-models-and-jev): application state in, typed probabilistic decisions out. Jev is a specialized hosted model. `choosekit` brings the same typed decision interface to general-purpose language models. The llama.cpp backend runs on infrastructure you choose; OpenRouter provides hosted inference.
`choosekit` is an independent project with no affiliation to TypeSafe or Jev.
## MCP server
-[`choosekit-mcp`](packages/choosekit-mcp/README.md) exposes choosekit through llama.cpp or OpenRouter as a read-only stdio tool for Claude Code, Codex, and OpenCode. Select the backend and configure it with environment variables when starting the MCP server. Every `choose` call uses this configuration.
+[`choosekit-mcp`](packages/choosekit-mcp/README.md) exposes choosekit through llama.cpp or OpenRouter as a read-only stdio tool for Claude Code, Codex, and OpenCode. Select the backend and configure it with environment variables when starting the MCP server.
## llama.cpp
@@ -65,17 +92,11 @@ const choose = fromOpenRouter({
});
```
-The OpenRouter backend supports models and providers that return first-token `top_logprobs`, with up to 20 choices. Unlike llama.cpp, this backend sends the prompt to OpenRouter. It requests reasoning to be disabled. Choices omitted from `top_logprobs` receive zero probability. Returned probabilities are normalized across the supplied choices and are not calibrated correctness estimates.
+The OpenRouter backend supports models and providers that return first-token `top_logprobs`, with up to 20 choices. It sends the prompt to OpenRouter and requests reasoning to be disabled.
-OpenRouter may route the same model through different providers. Set `provider` to an OpenRouter provider slug to use only that provider and disable fallback:
+Choices omitted from `top_logprobs` receive zero probability. Returned probabilities are normalized across the supplied choices and are not calibrated correctness estimates.
-```ts
-const choose = fromOpenRouter({
- apiKey: process.env.OPENROUTER_API_KEY!,
- model: "qwen/qwen3.8-27b",
- provider: process.env.OPENROUTER_PROVIDER!,
-});
-```
+OpenRouter may route the same model through different providers. Set `provider: "provider-slug"` to use only that provider and disable fallback.
## Scoring modes
@@ -88,14 +109,53 @@ In `labels` mode, choices are shown to the model as `A`, `B`, `C` instead of the
`minimal-prefix` walks the token tree until every key is distinguishable. For keys such as `watermelon` and `watermelon juice`, the shared token path is handled once and scoring stops when the paths separate.
-## Benchmark
+## Return value
+
+`choose()` resolves to:
+
+```ts
+{
+ choice, // selected caller key
+ distribution, // normalized probability for every supplied key
+ scores, // backend log-probability score for every key
+ margin, // largest probability minus the second largest
+ entropy, // Shannon entropy in nats
+ boundaryTokens, // prompt tokens rolled back at a tokenization boundary
+ usage, // backend work, when reported
+}
+```
+
+The result and its nested records are immutable. Each call is stateless. The caller controls action execution, inference retries, and model selection.
+
+## Prompt formatting
+
+`context` is copied unchanged to the start of the scoring prompt. The default formatter then appends the question, choice descriptions, and an answer marker.
+
+Use `formatPrompt` only when you need custom prompt formatting. The result must preserve `context` as an unchanged prefix so an existing server-side prefix cache can still be reused.
+
+## Custom scorer
+
+Use `createChooser` with any backend that can return one comparable conditional log-probability score per candidate:
+
+```ts
+import { createChooser, type Scorer } from "choosekit";
+
+const scorer: Scorer = async ({ prompt, candidates, signal }) => ({
+ logprobs: await scoreCandidateSequences(prompt, candidates, signal),
+});
+
+const choose = createChooser(scorer);
+```
+
+Scores use natural logarithms and must be at most zero.
+
+## SemIf comparison
The local adapter was compared with `typesafe/jev-1.13` on SemIf's official 144-row `authored144` benchmark, which covers evidence interpretation, rule application, and candidate selection. The local model was **Qwen 3.8 27B Q4_XL** served by llama.cpp on an **NVIDIA RTX 4090**. The Qwen run used the default A/B/C mode. The model was already loaded, and requests were sent one at a time to a llama.cpp server on the same machine.
| Metric | Qwen 3.8 27B Q4_XL + choosekit | Jev 1.13 |
|---|---:|---:|
| Accuracy | 96.53% (139/144) | 96.53% (139/144) |
-| Average balanced accuracy across task families | 96.01% | 95.56% |
| Median latency (p50) | 239 ms | 368 ms |
| 95th percentile latency (p95) | 286 ms | 546 ms |
| Throughput | 4.02 decisions/s | 2.43 decisions/s |
@@ -106,8 +166,6 @@ These results are specific to this 144-case benchmark, and performance can diffe
### Probability examples
-Examples from the same benchmark:
-
[`eafc22c8c40df3932a8e`](benchmarks/data/semif-authored144.jsonl#L112) asks whether the crate is currently in storage. The protocol gives the inventory priority; the current inventory and desk-log entries are missing.
| Choice | Qwen probability | Jev probability |
@@ -124,7 +182,7 @@ Examples from the same benchmark:
| **Insufficient evidence (selected by both)** | **97.605%** | **99.000%** |
| Contradicted | 0.029% | 1.000% |
-The Qwen + llama.cpp probabilities shown here are [uncalibrated](https://proceedings.mlr.press/v70/guo17a.html). [Jev is trained for calibrated decisions](https://typesafe.ai/blog/introducing-system-one-models-and-jev). The distributions look similar in these examples. This benchmark measures accuracy, latency, and distribution similarity.
+The Qwen + llama.cpp probabilities shown here are [uncalibrated](https://proceedings.mlr.press/v70/guo17a.html). [Jev is trained for calibrated decisions](https://typesafe.ai/blog/introducing-system-one-models-and-jev). The distributions look similar in these examples.
### Distribution comparison
@@ -142,51 +200,9 @@ Total variation distance (TVD) compares two complete probability distributions.
| Cases with TVD at or below 10% | 77.08% (111/144) |
| Cases with TVD above 20% | 14.58% (21/144) |
-Most distributions are similar. Some differ substantially: the systems select different choices in eight cases, and the largest TVD is 94.23%.
-
-## Prompt formatting
-
-`context` is copied unchanged to the start of the scoring prompt. The default formatter then appends the question, choice descriptions, and an answer marker.
-
-Use `formatPrompt` only when you need custom prompt formatting. The result must preserve `context` as an unchanged prefix so an existing server-side prefix cache can still be reused.
-
-## Custom scorer
-
-Use `createChooser` with any backend that can return one comparable conditional log-probability score per candidate:
-
-```ts
-import { createChooser, type Scorer } from "choosekit";
-
-const scorer: Scorer = async ({ prompt, candidates, signal }) => ({
- logprobs: await scoreCandidateSequences(prompt, candidates, signal),
-});
-
-const choose = createChooser(scorer);
-```
-
-Scores use natural logarithms and must be at most zero.
-
-## Result
-
-`choose()` resolves to:
-
-```ts
-{
- choice, // selected caller key
- distribution, // normalized probability for every supplied key
- scores, // backend log-probability score for every key
- margin, // largest probability minus the second largest
- entropy, // Shannon entropy in nats
- boundaryTokens, // prompt tokens rolled back at a tokenization boundary
- usage, // backend work, when reported
-}
-```
-
-The result and its nested records are immutable. Each call is stateless. The caller controls action execution, inference retries, and model selection.
-
## Requirements
- Node.js 20 or newer.
-- No runtime dependencies, model downloads, installation hooks, or bundled inference servers.
+- The `choosekit` package has no runtime dependencies, model downloads, installation hooks, or bundled inference servers.
[Apache-2.0](LICENSE). Copyright 2026 NotXf1le.
diff --git a/benchmarks/README.md b/benchmarks/README.md
index c6073cb..e140136 100644
--- a/benchmarks/README.md
+++ b/benchmarks/README.md
@@ -1,41 +1,103 @@
# Benchmarks
-`data/semif-authored144.jsonl` is an exact copy of SemIf's official [`benchmarks/data/authored144.jsonl`](https://github.com/TheoLeeCJ/SemIf/blob/b9cb32537e78be65f19abfcb1de8fc504b627d84/benchmarks/data/authored144.jsonl) at commit `b9cb32537e78be65f19abfcb1de8fc504b627d84`. Only the local filename differs.
+## SuperGPQA
-The 144 examples were authored by the SemIf project. The dataset is distributed under SemIf's MIT license, reproduced in `data/SEMIF-LICENSE.txt`.
+The chart uses a deterministic 1,000-question sample stratified by discipline and
+difficulty. The random-choice baseline is the mean of `1 / number of choices` across
+the sample. OpenRouter models are included only when they return `top_logprobs`.
+The evaluation sample excludes a deterministic 100-question pilot used to select
+working model and provider pairs.
-Build the package before running a benchmark:
+### Prepare the dataset
```sh
-npm ci
+hf download m-a-p/SuperGPQA SuperGPQA-all.jsonl \
+ --repo-type dataset \
+ --revision 4430d4458112c7d4497fdcf94d7cc223313d6acf \
+ --local-dir benchmarks/.data/supergpqa/source
+node benchmarks/prepare-supergpqa.mjs
npm run build
```
-Run the published llama.cpp adapter against a local server:
+### Run the benchmark
+
+```sh
+OPENROUTER_API_KEY=... node benchmarks/run-supergpqa.mjs \
+ --backend openrouter \
+ --model ibm-granite/granite-4.0-h-micro \
+ --provider cloudflare \
+ --sample-method proportional \
+ --sample-size 1000 \
+ --output benchmarks/results/supergpqa-granite-4.0-h-micro-cloudflare.json
+```
+
+The chart uses these OpenRouter model and provider pairs:
+
+| Model | Provider | Result |
+|---|---|---|
+| `ibm-granite/granite-4.0-h-micro` | `cloudflare` | `supergpqa-granite-4.0-h-micro-cloudflare.json` |
+| `meta-llama/llama-3.1-8b-instruct` | `novita` | `supergpqa-llama-3.1-8b-novita.json` |
+| `z-ai/glm-4.7-flash` | `cloudflare` | `supergpqa-glm-4.7-flash-cloudflare.json` |
+| `z-ai/glm-5.2` | `cloudflare` | `supergpqa-glm-5.2-cloudflare.json` |
+| `google/gemma-4-26b-a4b-it` | `dekallm` | `supergpqa-gemma-4-26b-dekallm.json` |
+| `ibm-granite/granite-4.2-8b` | `coreweave` | `supergpqa-granite-4.2-8b-coreweave.json` |
+| `deepseek/deepseek-v4.1-flash` | `wafer` | `supergpqa-deepseek-v4.1-flash-wafer.json` |
+| `deepseek/deepseek-v4-pro-0813` | `cloudflare` | `supergpqa-deepseek-v4-pro-cloudflare.json` |
+| `moonshotai/kimi-k3` | `morph` | `supergpqa-kimi-k3-morph.json` |
+
+Run Jev on the same sample:
+
+```sh
+OPENROUTER_API_KEY=... node benchmarks/run-supergpqa.mjs \
+ --backend jev \
+ --sample-method proportional \
+ --sample-size 1000 \
+ --output benchmarks/results/supergpqa-jev-1.13.json
+```
+
+Run a local model through llama.cpp:
+
+```sh
+node benchmarks/run-supergpqa.mjs \
+ --backend llama-cpp \
+ --base-url http://127.0.0.1:8080 \
+ --model qwen3.8-27b-text-64k \
+ --sample-method proportional \
+ --sample-size 1000 \
+ --output benchmarks/results/supergpqa-qwen3.8-27b-local.json
+```
+
+The X axis is average cost per decision multiplied by seconds per decision. The local
+model is shown as a horizontal accuracy line.
```sh
+node benchmarks/generate-supergpqa-chart.mjs
+```
+
+## SemIf
+
+`data/semif-authored144.jsonl` is SemIf's official
+[`benchmarks/data/authored144.jsonl`](https://github.com/TheoLeeCJ/SemIf/blob/b9cb32537e78be65f19abfcb1de8fc504b627d84/benchmarks/data/authored144.jsonl)
+at commit `b9cb32537e78be65f19abfcb1de8fc504b627d84`. The examples were authored by the
+SemIf project and are distributed under its MIT license, reproduced in
+`data/SEMIF-LICENSE.txt`.
+
+```sh
+npm ci
+npm run build
+
node benchmarks/run-semif.mjs \
--mode labels \
--base-url http://127.0.0.1:11434/ \
--model qwen3.8-27b-text-64k \
--output benchmarks/results/semif-qwen-labels.json
-```
-
-Run the Jev comparison with an OpenRouter API key:
-```sh
OPENROUTER_API_KEY=... node benchmarks/run-semif-openrouter-jev.mjs \
--model typesafe/jev-1.13 \
--output benchmarks/results/semif-jev-1.13.json
-```
-Compare the complete distributions:
-
-```sh
node benchmarks/compare-semif.mjs \
--qwen benchmarks/results/semif-qwen-labels.json \
--jev benchmarks/results/semif-jev-1.13.json \
--output benchmarks/results/semif-comparison.json
```
-
-Benchmark result files are ignored because they can contain environment-specific timing and provider metadata.
diff --git a/benchmarks/generate-supergpqa-chart.mjs b/benchmarks/generate-supergpqa-chart.mjs
new file mode 100644
index 0000000..c151060
--- /dev/null
+++ b/benchmarks/generate-supergpqa-chart.mjs
@@ -0,0 +1,344 @@
+import { createHash } from "node:crypto";
+import { readFile, writeFile } from "node:fs/promises";
+import { fileURLToPath } from "node:url";
+import {
+ SUPERGPQA_EVALUATION_SAMPLE_SEED,
+ SUPERGPQA_PILOT_ROWS,
+ SUPERGPQA_PREPARED_SHA256,
+ sampleSuperGpqaEvaluationRows,
+} from "./prepare-supergpqa.mjs";
+
+const EVALUATION_ROWS = 1000;
+const benchmarksDirectory = new URL("./", import.meta.url);
+const runInputs = [
+ {
+ file: "supergpqa-granite-4.0-h-micro-cloudflare.json",
+ id: "granite-4.0-h-micro",
+ label: "Granite 4.0 H Micro",
+ series: "openrouter",
+ model: "ibm-granite/granite-4.0-h-micro",
+ provider: "cloudflare",
+ resolvedProvider: "Cloudflare",
+ },
+ {
+ file: "supergpqa-llama-3.1-8b-novita.json",
+ id: "llama-3.1-8b",
+ label: "Llama 3.1 8B",
+ series: "openrouter",
+ model: "meta-llama/llama-3.1-8b-instruct",
+ provider: "novita",
+ resolvedProvider: "Novita",
+ labelDy: 27,
+ },
+ {
+ file: "supergpqa-glm-4.7-flash-cloudflare.json",
+ id: "glm-4.7-flash",
+ label: "GLM 4.7 Flash",
+ series: "openrouter",
+ model: "z-ai/glm-4.7-flash",
+ provider: "cloudflare",
+ resolvedProvider: "Cloudflare",
+ },
+ {
+ file: "supergpqa-glm-5.2-cloudflare.json",
+ id: "glm-5.2",
+ label: "GLM 5.2",
+ series: "openrouter",
+ model: "z-ai/glm-5.2",
+ provider: "cloudflare",
+ resolvedProvider: "Cloudflare",
+ },
+ {
+ file: "supergpqa-gemma-4-26b-dekallm.json",
+ id: "gemma-4-26b",
+ label: "Gemma 4 26B",
+ series: "openrouter",
+ model: "google/gemma-4-26b-a4b-it",
+ provider: "dekallm",
+ resolvedProvider: "DekaLLM",
+ },
+ {
+ file: "supergpqa-granite-4.2-8b-coreweave.json",
+ id: "granite-4.2-8b",
+ label: "Granite 4.2 8B",
+ series: "openrouter",
+ model: "ibm-granite/granite-4.2-8b",
+ provider: "coreweave",
+ resolvedProvider: "CoreWeave",
+ labelDy: 22,
+ },
+ {
+ file: "supergpqa-deepseek-v4.1-flash-wafer.json",
+ id: "deepseek-v4.1-flash",
+ label: "DeepSeek V4.1 Flash",
+ series: "openrouter",
+ model: "deepseek/deepseek-v4.1-flash",
+ provider: "wafer",
+ resolvedProvider: "Wafer",
+ },
+ {
+ file: "supergpqa-deepseek-v4-pro-cloudflare.json",
+ id: "deepseek-v4-pro",
+ label: "DeepSeek V4 Pro",
+ series: "openrouter",
+ model: "deepseek/deepseek-v4-pro-0813",
+ provider: "cloudflare",
+ resolvedProvider: "Cloudflare",
+ labelDy: 27,
+ },
+ {
+ file: "supergpqa-kimi-k3-morph.json",
+ id: "kimi-k3",
+ label: "Kimi K3",
+ series: "openrouter",
+ model: "moonshotai/kimi-k3",
+ provider: "morph",
+ resolvedProvider: "Morph",
+ labelDx: -10,
+ labelAnchor: "end",
+ },
+ {
+ file: "supergpqa-jev-1.13.json",
+ id: "jev-1.13",
+ label: "Jev 1.13",
+ series: "jev",
+ model: "typesafe/jev-1.13",
+ },
+];
+const localReferenceInput = {
+ file: "supergpqa-qwen3.8-27b-local.json",
+ id: "qwen3.8-27b-local",
+ label: "Local Qwen3.8 27B Q4_XL",
+ series: "llama-cpp",
+ model: "qwen3.8-27b-text-64k",
+};
+
+function invariant(condition, message) {
+ if (!condition) throw new Error(message);
+}
+
+function formatNumber(value, digits = 2) {
+ return Number(value.toFixed(digits)).toString();
+}
+
+function wilson95(correct, total) {
+ const z = 1.959963984540054;
+ const p = correct / total;
+ const z2 = z ** 2;
+ const denominator = 1 + z2 / total;
+ const center = (p + z2 / (2 * total)) / denominator;
+ const margin = (z / denominator)
+ * Math.sqrt((p * (1 - p) + z2 / (4 * total)) / total);
+ return [center - margin, center + margin];
+}
+
+function median(values) {
+ const sorted = [...values].sort((a, b) => a - b);
+ const middle = Math.floor(sorted.length / 2);
+ return sorted.length % 2 === 0
+ ? (sorted[middle - 1] + sorted[middle]) / 2
+ : sorted[middle];
+}
+
+async function loadRun(input) {
+ const report = JSON.parse(
+ await readFile(new URL(`results/${input.file}`, benchmarksDirectory), "utf8"),
+ );
+ invariant(
+ report.dataset?.selectedRows === EVALUATION_ROWS,
+ `${input.file}: expected ${EVALUATION_ROWS} selected rows`,
+ );
+ invariant(report.dataset?.sample?.method === "proportional-discipline-difficulty",
+ `${input.file}: unexpected sample method`);
+ invariant(report.dataset?.sha256 === SUPERGPQA_PREPARED_SHA256,
+ `${input.file}: unexpected prepared dataset hash`);
+ invariant(report.dataset?.sample?.seed === SUPERGPQA_EVALUATION_SAMPLE_SEED,
+ `${input.file}: unexpected sample seed`);
+ invariant(
+ report.dataset?.sample?.excludedPilotSampleSize === SUPERGPQA_PILOT_ROWS,
+ `${input.file}: unexpected excluded pilot sample size`,
+ );
+ invariant(
+ report.summary?.rowsAttempted === EVALUATION_ROWS,
+ `${input.file}: expected ${EVALUATION_ROWS} attempted rows`,
+ );
+ invariant(
+ report.summary?.rowsSuccessful === EVALUATION_ROWS,
+ `${input.file}: expected ${EVALUATION_ROWS} successful rows`,
+ );
+ invariant(report.summary?.errors === 0, `${input.file}: expected zero errors`);
+ invariant(report.summary?.costComplete === true, `${input.file}: incomplete cost data`);
+ invariant(Number.isFinite(report.summary?.costUsd)
+ && (input.series === "llama-cpp" ? report.summary.costUsd === 0 : report.summary.costUsd > 0),
+ `${input.file}: invalid cost`);
+ invariant(Number.isFinite(report.summary?.decisionsPerSecond) && report.summary.decisionsPerSecond > 0,
+ `${input.file}: invalid throughput`);
+ invariant(Array.isArray(report.results) && report.results.length === EVALUATION_ROWS,
+ `${input.file}: expected ${EVALUATION_ROWS} result rows`);
+ invariant(report.runtime?.model === input.model, `${input.file}: unexpected model`);
+ invariant(report.runtime?.backend === input.series, `${input.file}: unexpected backend`);
+ if (input.series === "openrouter") {
+ invariant(report.runtime?.provider === input.provider, `${input.file}: unexpected provider`);
+ invariant(report.results.every((row) => row.resolvedModel === input.model),
+ `${input.file}: OpenRouter resolved a different model`);
+ invariant(report.results.every((row) => row.resolvedProvider === input.resolvedProvider),
+ `${input.file}: OpenRouter resolved a different provider`);
+ }
+ invariant(report.results.every((row) => Number.isFinite(row.latencyMs) && row.latencyMs > 0),
+ `${input.file}: invalid latency`);
+ const correct = report.results.filter((row) => row.correct === true).length;
+ invariant(report.summary.endToEndAccuracy === correct / EVALUATION_ROWS,
+ `${input.file}: accuracy does not match result rows`);
+ return {
+ sha256: report.dataset.sha256,
+ ids: report.results.map((row) => row.id),
+ run: {
+ id: input.id,
+ label: input.label,
+ series: input.series,
+ model: input.model,
+ ...(input.provider === undefined ? {} : { provider: input.provider }),
+ ...(input.resolvedProvider === undefined ? {} : { resolvedProvider: input.resolvedProvider }),
+ ...(input.labelDx === undefined ? {} : { labelDx: input.labelDx }),
+ ...(input.labelDy === undefined ? {} : { labelDy: input.labelDy }),
+ ...(input.labelAnchor === undefined ? {} : { labelAnchor: input.labelAnchor }),
+ correct,
+ totalCostUsd: report.summary.costUsd,
+ medianLatencyMs: median(report.results.map((row) => row.latencyMs)),
+ decisionsPerSecond: report.summary.decisionsPerSecond,
+ },
+ };
+}
+
+function renderSvg(aggregate) {
+ const width = 920;
+ const height = 540;
+ const plot = { left: 82, right: 882, top: 76, bottom: 430 };
+ const points = aggregate.runs.map((run) => ({
+ ...run,
+ xValue: (run.totalCostUsd / aggregate.dataset.rows) * (1 / run.decisionsPerSecond),
+ yValue: run.correct / aggregate.dataset.rows,
+ }));
+ const logMin = -6.2;
+ const logMax = -2.7;
+ const yMax = 0.7;
+ const scaleX = (value) => plot.left + ((Math.log10(value) - logMin) / (logMax - logMin))
+ * (plot.right - plot.left);
+ const scaleY = (value) => plot.bottom - (value / yMax) * (plot.bottom - plot.top);
+ const xTicks = [0.000001, 0.00001, 0.0001, 0.001];
+ const yTicks = [0, 0.2, 0.4, 0.6];
+ const lines = [];
+
+ lines.push(`");
+ return `${lines.join("\n")}\n`;
+}
+
+const loadedRuns = await Promise.all(runInputs.map(loadRun));
+const localReference = await loadRun(localReferenceInput);
+const allRuns = [...loadedRuns, localReference];
+invariant(new Set(allRuns.map(({ sha256 }) => sha256)).size === 1,
+ "all reports must use the same dataset hash");
+const expectedIds = JSON.stringify(loadedRuns[0].ids);
+invariant(allRuns.every(({ ids }) => JSON.stringify(ids) === expectedIds),
+ "all reports must contain the same questions in the same order");
+
+const preparedDataset = await readFile(
+ new URL(".data/supergpqa/supergpqa.jsonl", benchmarksDirectory),
+);
+const preparedSha256 = createHash("sha256").update(preparedDataset).digest("hex");
+invariant(preparedSha256 === SUPERGPQA_PREPARED_SHA256,
+ "prepared dataset has an unexpected hash");
+const preparedRows = preparedDataset.toString("utf8").trim()
+ .split(/\r?\n/).map((line) => JSON.parse(line));
+const rowsById = new Map(preparedRows.map((row) => [row.id, row]));
+const evaluationRows = loadedRuns[0].ids.map((id) => rowsById.get(id));
+invariant(evaluationRows.every(Boolean), "result IDs must exist in the prepared dataset");
+const expectedEvaluationIds = sampleSuperGpqaEvaluationRows(
+ preparedRows,
+ EVALUATION_ROWS,
+ SUPERGPQA_PILOT_ROWS,
+).map(({ id }) => id);
+invariant(JSON.stringify(loadedRuns[0].ids) === JSON.stringify(expectedEvaluationIds),
+ "reports must contain the expected evaluation sample in order");
+const randomBaseline = evaluationRows.reduce(
+ (sum, row) => sum + 1 / Object.keys(row.choices).length,
+ 0,
+) / evaluationRows.length;
+
+const aggregate = {
+ dataset: {
+ rows: EVALUATION_ROWS,
+ sha256: loadedRuns[0].sha256,
+ randomBaseline,
+ sample: "proportional-discipline-difficulty",
+ },
+ runs: loadedRuns.map(({ run }) => run),
+ localReference: localReference.run,
+};
+
+await writeFile(
+ new URL("supergpqa-benchmark.json", benchmarksDirectory),
+ `${JSON.stringify(aggregate, null, 2)}\n`,
+);
+await writeFile(new URL("supergpqa-benchmark.svg", benchmarksDirectory), renderSvg(aggregate));
+console.log(`Wrote ${fileURLToPath(new URL("supergpqa-benchmark.json", benchmarksDirectory))}`);
+console.log(`Wrote ${fileURLToPath(new URL("supergpqa-benchmark.svg", benchmarksDirectory))}`);
diff --git a/benchmarks/prepare-supergpqa.mjs b/benchmarks/prepare-supergpqa.mjs
new file mode 100644
index 0000000..44edd16
--- /dev/null
+++ b/benchmarks/prepare-supergpqa.mjs
@@ -0,0 +1,212 @@
+import { createHash } from "node:crypto";
+import { createReadStream, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
+import { dirname } from "node:path";
+import { pathToFileURL } from "node:url";
+
+export const SUPERGPQA_REVISION = "4430d4458112c7d4497fdcf94d7cc223313d6acf";
+export const SUPERGPQA_SHA256 = "28b998e70205ee95e540317b5adc06a06552a3961fb50b153df126b833f7a910";
+export const SUPERGPQA_PREPARED_SHA256 = "d17a48292cc8f081ca576afc7e8eaa6ca510de59fa9881282922409666bb3aff";
+export const SUPERGPQA_ROWS = 26529;
+export const SUPERGPQA_PILOT_SAMPLE_SEED = "choosekit-supergpqa-v1";
+export const SUPERGPQA_EVALUATION_SAMPLE_SEED = "choosekit-supergpqa-final-v2";
+export const SUPERGPQA_PILOT_ROWS = 100;
+
+function option(name, fallback) {
+ const index = process.argv.indexOf(name);
+ if (index === -1) return fallback;
+ const value = process.argv[index + 1];
+ if (value === undefined || value.startsWith("--")) throw new TypeError(`${name} requires a value.`);
+ return value;
+}
+
+function requireText(value, name, id) {
+ if (typeof value !== "string" || value.trim().length === 0) {
+ throw new TypeError(`SuperGPQA row ${id} has an invalid ${name}.`);
+ }
+ return value.trim();
+}
+
+function stableRank(value, seed = SUPERGPQA_PILOT_SAMPLE_SEED) {
+ return createHash("sha256").update(`${seed}\0${value}`).digest("hex");
+}
+
+export function prepareSuperGpqaRows(rows) {
+ if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array.");
+ const ids = new Set();
+ return Object.freeze(rows.map((row, index) => {
+ if (row === null || typeof row !== "object" || Array.isArray(row)) {
+ throw new TypeError(`SuperGPQA row ${index + 1} is not an object.`);
+ }
+ const id = requireText(row.uuid, "uuid", index + 1);
+ if (ids.has(id)) throw new Error(`SuperGPQA contains duplicate row ID ${id}.`);
+ ids.add(id);
+ const question = requireText(row.question, "question", id);
+ if (!Array.isArray(row.options) || row.options.length < 2 || row.options.length > 20) {
+ throw new TypeError(`SuperGPQA row ${id} has invalid options.`);
+ }
+ const answerLetter = requireText(row.answer_letter, "answer_letter", id);
+ const answerIndex = answerLetter.charCodeAt(0) - 65;
+ if (!/^[A-T]$/.test(answerLetter) || answerIndex >= row.options.length) {
+ throw new Error(`SuperGPQA row ${id} has an answer outside its options.`);
+ }
+ const choices = {};
+ for (const [optionIndex, value] of row.options.entries()) {
+ choices[String.fromCharCode(65 + optionIndex)] = requireText(value, `option ${optionIndex + 1}`, id);
+ }
+ const answer = requireText(row.answer, "answer", id);
+ if (choices[answerLetter] !== answer) {
+ throw new Error(`SuperGPQA row ${id} has inconsistent answer text and letter.`);
+ }
+ return Object.freeze({
+ id,
+ context: "",
+ question,
+ choices: Object.freeze(choices),
+ gold: answerLetter,
+ metadata: Object.freeze({
+ discipline: requireText(row.discipline, "discipline", id),
+ field: requireText(row.field, "field", id),
+ subfield: requireText(row.subfield, "subfield", id),
+ difficulty: requireText(row.difficulty, "difficulty", id),
+ isCalculation: row.is_calculation === true,
+ }),
+ });
+ }));
+}
+
+export function sampleSuperGpqaPilotRows(rows, size) {
+ if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array.");
+ if (!Number.isSafeInteger(size) || size < 1 || size > rows.length) {
+ throw new TypeError("SuperGPQA sample size must be a positive integer no larger than the dataset.");
+ }
+ const groups = new Map();
+ for (const row of rows) {
+ const key = `${row.metadata?.discipline}\0${row.metadata?.difficulty}`;
+ const group = groups.get(key) ?? [];
+ group.push(row);
+ groups.set(key, group);
+ }
+ const orderedGroups = [...groups.entries()]
+ .sort(([a], [b]) => stableRank(a).localeCompare(stableRank(b)))
+ .map(([, group]) => [...group].sort((a, b) => stableRank(a.id).localeCompare(stableRank(b.id))));
+ const selected = [];
+ for (let round = 0; selected.length < size; round++) {
+ let added = 0;
+ for (const group of orderedGroups) {
+ if (selected.length === size) break;
+ if (round < group.length) {
+ selected.push(group[round]);
+ added++;
+ }
+ }
+ if (added === 0) break;
+ }
+ return selected;
+}
+
+export function sampleSuperGpqaStratifiedRows(
+ rows,
+ size,
+ seed = SUPERGPQA_EVALUATION_SAMPLE_SEED,
+) {
+ if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array.");
+ if (!Number.isSafeInteger(size) || size < 1 || size > rows.length) {
+ throw new TypeError("SuperGPQA sample size must be a positive integer no larger than the dataset.");
+ }
+ const groups = new Map();
+ for (const row of rows) {
+ const key = `${row.metadata?.discipline}\0${row.metadata?.difficulty}`;
+ const group = groups.get(key) ?? [];
+ group.push(row);
+ groups.set(key, group);
+ }
+ const allocations = [...groups.entries()].map(([key, group]) => {
+ const exact = (size * group.length) / rows.length;
+ return {
+ key,
+ group: [...group].sort((a, b) => stableRank(a.id, seed).localeCompare(stableRank(b.id, seed))),
+ count: Math.floor(exact),
+ remainder: exact - Math.floor(exact),
+ };
+ });
+ let remaining = size - allocations.reduce((total, allocation) => total + allocation.count, 0);
+ allocations.sort((a, b) => b.remainder - a.remainder
+ || stableRank(a.key, seed).localeCompare(stableRank(b.key, seed)));
+ for (let index = 0; index < remaining; index++) allocations[index].count++;
+ return allocations
+ .flatMap(({ group, count }) => group.slice(0, count))
+ .sort((a, b) => stableRank(a.id, seed).localeCompare(stableRank(b.id, seed)));
+}
+
+export function sampleSuperGpqaEvaluationRows(
+ rows,
+ size,
+ pilotSampleSize = SUPERGPQA_PILOT_ROWS,
+) {
+ if (!Array.isArray(rows)) throw new TypeError("SuperGPQA rows must be an array.");
+ if (!Number.isSafeInteger(pilotSampleSize)
+ || pilotSampleSize < 1 || pilotSampleSize >= rows.length) {
+ throw new TypeError(
+ "SuperGPQA pilot sample size must be a positive integer smaller than the dataset.",
+ );
+ }
+ const pilotIds = new Set(
+ sampleSuperGpqaPilotRows(rows, pilotSampleSize).map(({ id }) => id),
+ );
+ const evaluationPool = rows.filter(({ id }) => !pilotIds.has(id));
+ return sampleSuperGpqaStratifiedRows(
+ evaluationPool,
+ size,
+ SUPERGPQA_EVALUATION_SAMPLE_SEED,
+ );
+}
+
+async function sha256(path) {
+ const hash = createHash("sha256");
+ for await (const chunk of createReadStream(path)) hash.update(chunk);
+ return hash.digest("hex");
+}
+
+export async function prepareSuperGpqaFile({ input, output, audit }) {
+ if (existsSync(output)) throw new Error(`Refusing to overwrite existing output: ${output}`);
+ if (existsSync(audit)) throw new Error(`Refusing to overwrite existing audit: ${audit}`);
+ const inputSha256 = await sha256(input);
+ if (inputSha256 !== SUPERGPQA_SHA256) {
+ throw new Error(`Unexpected SuperGPQA SHA-256: ${inputSha256}. Expected ${SUPERGPQA_SHA256}.`);
+ }
+ const text = readFileSync(input, "utf8").trim();
+ const sourceRows = text.length === 0 ? [] : text.split(/\r?\n/).map((line) => JSON.parse(line));
+ const rows = prepareSuperGpqaRows(sourceRows);
+ if (rows.length !== SUPERGPQA_ROWS) {
+ throw new Error(`Expected ${SUPERGPQA_ROWS} SuperGPQA rows, received ${rows.length}.`);
+ }
+ const counts = {};
+ for (const row of rows) {
+ const key = `${row.metadata.discipline} / ${row.metadata.difficulty}`;
+ counts[key] = (counts[key] ?? 0) + 1;
+ }
+ const report = {
+ dataset: {
+ source: "https://huggingface.co/datasets/m-a-p/SuperGPQA",
+ revision: SUPERGPQA_REVISION,
+ path: input,
+ sha256: inputSha256,
+ },
+ summary: { total: rows.length, strata: counts },
+ };
+ mkdirSync(dirname(output), { recursive: true });
+ mkdirSync(dirname(audit), { recursive: true });
+ writeFileSync(output, `${rows.map((row) => JSON.stringify(row)).join("\n")}\n`);
+ writeFileSync(audit, `${JSON.stringify(report, null, 2)}\n`);
+ return Object.freeze(report);
+}
+
+async function main() {
+ const input = option("--input", "benchmarks/.data/supergpqa/source/SuperGPQA-all.jsonl");
+ const output = option("--output", "benchmarks/.data/supergpqa/supergpqa.jsonl");
+ const audit = option("--audit", "benchmarks/.data/supergpqa/supergpqa.audit.json");
+ console.log(JSON.stringify(await prepareSuperGpqaFile({ input, output, audit }), null, 2));
+}
+
+const entry = process.argv[1] === undefined ? undefined : pathToFileURL(process.argv[1]).href;
+if (entry === import.meta.url) await main();
diff --git a/benchmarks/run-semif-openrouter-jev.mjs b/benchmarks/run-semif-openrouter-jev.mjs
index 0688694..d97648a 100644
--- a/benchmarks/run-semif-openrouter-jev.mjs
+++ b/benchmarks/run-semif-openrouter-jev.mjs
@@ -130,7 +130,7 @@ function validateAnswer(answer, optionIds) {
}
const input = option("--input", "benchmarks/data/semif-authored144.jsonl");
-const output = option("--output", "benchmarks/results/semif-openrouter-jev-latest.json");
+const output = option("--output", "benchmarks/results/semif-jev-1.13.json");
const model = option("--model", "typesafe/jev-1.13");
const rawLimit = option("--limit", undefined);
const limit = rawLimit === undefined ? undefined : Number.parseInt(rawLimit, 10);
diff --git a/benchmarks/run-semif.mjs b/benchmarks/run-semif.mjs
index 67fa26a..3e49a8e 100644
--- a/benchmarks/run-semif.mjs
+++ b/benchmarks/run-semif.mjs
@@ -94,7 +94,7 @@ const mode = option("--mode", "labels");
if (mode !== "labels" && mode !== "minimal-prefix") {
throw new TypeError("--mode must be labels or minimal-prefix.");
}
-const output = option("--output", `benchmarks/results/semif-qwen3.8-27b-production-${mode}.json`);
+const output = option("--output", `benchmarks/results/semif-qwen3.8-27b-${mode}.json`);
const baseURL = option("--base-url", process.env.LLAMA_CPP_BASE_URL
?? "http://127.0.0.1:11434/");
const model = option("--model", process.env.LLAMA_CPP_MODEL ?? "qwen3.8-27b-text-64k");
diff --git a/benchmarks/run-supergpqa.mjs b/benchmarks/run-supergpqa.mjs
new file mode 100644
index 0000000..c62c3d0
--- /dev/null
+++ b/benchmarks/run-supergpqa.mjs
@@ -0,0 +1,366 @@
+import { createHash } from "node:crypto";
+import { existsSync, mkdirSync, readFileSync, renameSync, writeFileSync } from "node:fs";
+import { dirname } from "node:path";
+import { performance } from "node:perf_hooks";
+import { setTimeout as delay } from "node:timers/promises";
+import { fromLlamaCpp } from "../dist/esm/llama-cpp.js";
+import { fromOpenRouter } from "../dist/esm/openrouter.js";
+import {
+ SUPERGPQA_REVISION,
+ SUPERGPQA_PILOT_SAMPLE_SEED,
+ SUPERGPQA_EVALUATION_SAMPLE_SEED,
+ SUPERGPQA_PILOT_ROWS,
+ SUPERGPQA_PREPARED_SHA256,
+ SUPERGPQA_ROWS,
+ sampleSuperGpqaPilotRows,
+ sampleSuperGpqaEvaluationRows,
+} from "./prepare-supergpqa.mjs";
+
+const MAX_429_ATTEMPTS = 4;
+
+function option(name, fallback) {
+ const index = process.argv.indexOf(name);
+ if (index === -1) return fallback;
+ const value = process.argv[index + 1];
+ if (value === undefined || value.startsWith("--")) throw new TypeError(`${name} requires a value.`);
+ return value;
+}
+
+function flag(name) {
+ return process.argv.includes(name);
+}
+
+function requireText(value, name) {
+ if (typeof value !== "string" || value.trim().length === 0) {
+ throw new TypeError(`${name} must be a non-empty string.`);
+ }
+ return value;
+}
+
+function readCost(value) {
+ const cost = value?.cost;
+ return typeof cost === "number" && Number.isFinite(cost) && cost >= 0 ? cost : null;
+}
+
+function validateRow(row, index) {
+ const optionIds = row?.choices !== null && typeof row?.choices === "object" && !Array.isArray(row.choices)
+ ? Object.keys(row.choices) : [];
+ if (row === null || typeof row !== "object" || Array.isArray(row)
+ || typeof row.id !== "string" || row.id.length === 0 || row.context !== ""
+ || typeof row.question !== "string" || row.question.length === 0
+ || optionIds.length < 2 || optionIds.length > 20
+ || typeof row.gold !== "string" || !optionIds.includes(row.gold)
+ || typeof row.metadata?.discipline !== "string" || typeof row.metadata?.difficulty !== "string") {
+ throw new TypeError(`SuperGPQA row ${row?.id ?? index + 1} is invalid.`);
+ }
+}
+
+function summarize(results, elapsedSeconds) {
+ const successful = results.filter((row) => row.error === undefined);
+ const correct = successful.filter((row) => row.correct).length;
+ const costComplete = results.every((row) => typeof row.costUsd === "number");
+ return {
+ rowsAttempted: results.length,
+ rowsSuccessful: successful.length,
+ errors: results.length - successful.length,
+ endToEndAccuracy: results.length === 0 ? null : correct / results.length,
+ wallSeconds: elapsedSeconds,
+ decisionsPerSecond: elapsedSeconds === 0 ? null : results.length / elapsedSeconds,
+ costUsd: costComplete ? results.reduce((total, row) => total + row.costUsd, 0) : null,
+ costComplete,
+ retries: results.reduce((total, row) => total + (row.retryCount ?? 0), 0),
+ };
+}
+
+const backend = option("--backend", undefined);
+if (backend !== "openrouter" && backend !== "llama-cpp" && backend !== "jev") {
+ throw new TypeError("--backend must be openrouter, llama-cpp, or jev.");
+}
+const inputPath = option("--input", "benchmarks/.data/supergpqa/supergpqa.jsonl");
+const sampleSizeArgument = option("--sample-size", "100");
+const sampleSize = Number(sampleSizeArgument);
+const sampleMethod = option("--sample-method", "balanced");
+if (sampleMethod !== "balanced" && sampleMethod !== "proportional") {
+ throw new TypeError("--sample-method must be balanced or proportional.");
+}
+const model = backend === "jev"
+ ? option("--model", "typesafe/jev-1.13")
+ : requireText(option("--model", process.env.OPENROUTER_MODEL), "--model");
+const provider = backend === "openrouter"
+ ? requireText(option("--provider", process.env.OPENROUTER_PROVIDER), "--provider")
+ : undefined;
+const baseURL = backend === "llama-cpp"
+ ? requireText(option("--base-url", process.env.LLAMA_CPP_BASE_URL), "--base-url")
+ : undefined;
+const recordedProvider = provider ?? (backend === "llama-cpp" ? "local" : "openrouter");
+const modelSlug = model.replace(/[^a-zA-Z0-9._-]+/g, "-");
+const providerSlug = recordedProvider.replace(/[^a-zA-Z0-9._-]+/g, "-");
+const outputPath = option("--output",
+ `benchmarks/results/supergpqa-${backend}-${providerSlug}-${modelSlug}-${sampleMethod}-sample${sampleSize}.json`);
+const resume = flag("--resume");
+const apiKey = process.env.OPENROUTER_API_KEY;
+
+if (backend !== "llama-cpp" && (typeof apiKey !== "string" || apiKey.length < 20)) {
+ throw new Error("OPENROUTER_API_KEY is missing or invalid.");
+}
+if (existsSync(outputPath) && !resume) {
+ throw new Error(`Refusing to overwrite existing output: ${outputPath}`);
+}
+if (!existsSync(outputPath) && resume) {
+ throw new Error(`Cannot resume missing output: ${outputPath}`);
+}
+if (!Number.isSafeInteger(sampleSize) || sampleSize < 1) {
+ throw new TypeError("--sample-size must be a positive integer.");
+}
+if (sampleMethod === "balanced" && sampleSize > SUPERGPQA_PILOT_ROWS) {
+ throw new TypeError(`--sample-size must not exceed ${SUPERGPQA_PILOT_ROWS} for balanced sampling.`);
+}
+
+const preparedDataset = readFileSync(inputPath);
+const preparedSha256 = createHash("sha256").update(preparedDataset).digest("hex");
+if (preparedSha256 !== SUPERGPQA_PREPARED_SHA256) {
+ throw new Error(`Unexpected prepared SuperGPQA SHA-256: ${preparedSha256}. Expected ${SUPERGPQA_PREPARED_SHA256}.`);
+}
+const preparedText = preparedDataset.toString("utf8").trim();
+const preparedRows = preparedText.length === 0
+ ? []
+ : preparedText.split(/\r?\n/).map((line) => JSON.parse(line));
+if (preparedRows.length !== SUPERGPQA_ROWS) {
+ throw new Error(`Expected ${SUPERGPQA_ROWS} SuperGPQA rows, received ${preparedRows.length}.`);
+}
+const rowIds = new Set();
+for (const [index, row] of preparedRows.entries()) {
+ validateRow(row, index);
+ if (rowIds.has(row.id)) throw new Error(`SuperGPQA contains duplicate row ID ${row.id}.`);
+ rowIds.add(row.id);
+}
+const selectedRows = sampleMethod === "proportional"
+ ? sampleSuperGpqaEvaluationRows(preparedRows, sampleSize)
+ : sampleSuperGpqaPilotRows(preparedRows, sampleSize);
+mkdirSync(dirname(outputPath), { recursive: true });
+
+const packageVersion = JSON.parse(
+ readFileSync(new URL("../package.json", import.meta.url), "utf8"),
+).version;
+const sampleMetadata = sampleMethod === "proportional"
+ ? {
+ method: "proportional-discipline-difficulty",
+ seed: SUPERGPQA_EVALUATION_SAMPLE_SEED,
+ excludedPilotSampleSize: SUPERGPQA_PILOT_ROWS,
+ }
+ : {
+ method: "balanced-discipline-difficulty",
+ seed: SUPERGPQA_PILOT_SAMPLE_SEED,
+ };
+const benchmarkName = sampleMethod === "proportional"
+ ? "SuperGPQA proportional discipline/difficulty sample"
+ : "SuperGPQA balanced discipline/difficulty sample";
+const datasetMetadata = {
+ source: "https://huggingface.co/datasets/m-a-p/SuperGPQA",
+ revision: SUPERGPQA_REVISION,
+ sha256: preparedSha256,
+ totalRows: preparedRows.length,
+ selectedRows: selectedRows.length,
+ sample: sampleMetadata,
+};
+const runtimeMetadata = {
+ backend,
+ model,
+ provider: recordedProvider,
+ ...(baseURL === undefined ? {} : { baseURL }),
+ packageVersion,
+};
+
+let responseMetadata = { costUsd: null, retryCount: 0 };
+const fetchWithMetadata = async (...args) => {
+ let accumulatedCost = 0;
+ let sawCost = false;
+ for (let attempt = 1; attempt <= MAX_429_ATTEMPTS; attempt++) {
+ const response = await globalThis.fetch(...args);
+ const body = await response.text();
+ let value;
+ try { value = JSON.parse(body); } catch { value = undefined; }
+ const attemptCost = readCost(value?.usage);
+ if (attemptCost !== null) {
+ accumulatedCost += attemptCost;
+ sawCost = true;
+ }
+ responseMetadata = {
+ costUsd: sawCost ? accumulatedCost : null,
+ retryCount: attempt - 1,
+ ...(typeof value?.model === "string" ? { resolvedModel: value.model } : {}),
+ ...(typeof value?.provider === "string" ? { resolvedProvider: value.provider } : {}),
+ };
+ if (response.status === 429 && attempt < MAX_429_ATTEMPTS) {
+ const retryAfter = Number(response.headers.get("retry-after"));
+ await delay(Number.isFinite(retryAfter) ? retryAfter * 1000 : 1000, undefined,
+ { signal: args[1]?.signal });
+ continue;
+ }
+ return new Response(body, { status: response.status, statusText: response.statusText, headers: response.headers });
+ }
+ throw new Error("OpenRouter 429 retry limit reached.");
+};
+
+const choose = backend === "openrouter"
+ ? fromOpenRouter({ apiKey, model, provider, fetch: fetchWithMetadata })
+ : backend === "llama-cpp"
+ ? fromLlamaCpp({ baseURL, model, mode: "labels" })
+ : undefined;
+
+async function decide(row) {
+ if (choose !== undefined) {
+ const decision = await choose({
+ context: row.context,
+ question: row.question,
+ choices: row.choices,
+ signal: AbortSignal.timeout(120_000),
+ });
+ return {
+ predicted: decision.choice,
+ scores: decision.scores,
+ ...(backend === "llama-cpp" ? { costUsd: 0, retryCount: 0 } : responseMetadata),
+ };
+ }
+ let response;
+ let body;
+ let accumulatedCost = 0;
+ let sawCost = false;
+ for (let attempt = 1; attempt <= MAX_429_ATTEMPTS; attempt++) {
+ response = await fetch("https://openrouter.ai/api/alpha/decisions", {
+ method: "POST",
+ headers: {
+ authorization: `Bearer ${apiKey}`,
+ "content-type": "application/json",
+ "x-openrouter-title": "choosekit SuperGPQA benchmark",
+ },
+ body: JSON.stringify({
+ model,
+ state: row.question,
+ questions: {
+ decision: {
+ type: "choice",
+ instructions: "Which answer choice is correct?",
+ criteria: row.choices,
+ },
+ },
+ }),
+ signal: AbortSignal.timeout(120_000),
+ });
+ const responseText = await response.text();
+ try { body = JSON.parse(responseText); } catch { body = undefined; }
+ const attemptCost = readCost(body?.usage);
+ if (attemptCost !== null) {
+ accumulatedCost += attemptCost;
+ sawCost = true;
+ }
+ responseMetadata = {
+ costUsd: sawCost ? accumulatedCost : null,
+ retryCount: attempt - 1,
+ };
+ if (response.status !== 429 || attempt === MAX_429_ATTEMPTS) break;
+ const retryAfter = Number(response.headers.get("retry-after"));
+ await delay(Number.isFinite(retryAfter) ? retryAfter * 1000 : 1000);
+ }
+ if (!response.ok) throw new Error(`OpenRouter returned HTTP ${response.status}.`);
+ const answer = body?.answers?.decision;
+ if (answer === null || typeof answer !== "object" || !Object.hasOwn(row.choices, answer.choice)) {
+ throw new Error("Jev returned an invalid answer.");
+ }
+ return { predicted: answer.choice, ...responseMetadata };
+}
+
+let results = [];
+let startedAt = new Date().toISOString();
+let elapsedBeforeSession = 0;
+
+function assertResumeMetadata(report) {
+ const expected = {
+ benchmark: benchmarkName,
+ dataset: datasetMetadata,
+ runtime: runtimeMetadata,
+ };
+ for (const key of Object.keys(expected)) {
+ if (JSON.stringify(report[key]) !== JSON.stringify(expected[key])) {
+ throw new Error(`Cannot resume: ${key} metadata does not match the current run.`);
+ }
+ }
+ if (typeof report.startedAt !== "string" || !Number.isFinite(Date.parse(report.startedAt))) {
+ throw new Error("Cannot resume: checkpoint has an invalid startedAt.");
+ }
+ if (!Array.isArray(report.results) || report.results.length > selectedRows.length
+ || report.results.some((result, index) => result?.id !== selectedRows[index].id)) {
+ throw new Error("Cannot resume: checkpoint result IDs are not a prefix of the selected rows.");
+ }
+}
+
+if (resume) {
+ const checkpoint = JSON.parse(readFileSync(outputPath, "utf8"));
+ assertResumeMetadata(checkpoint);
+ results = checkpoint.results;
+ startedAt = checkpoint.startedAt;
+ if (!Number.isFinite(checkpoint.summary?.wallSeconds) || checkpoint.summary.wallSeconds < 0) {
+ throw new Error("Cannot resume: checkpoint has invalid elapsed time.");
+ }
+ elapsedBeforeSession = checkpoint.summary.wallSeconds;
+}
+
+const sessionStartedAt = performance.now();
+
+function writeReport() {
+ const elapsedSeconds = elapsedBeforeSession + (performance.now() - sessionStartedAt) / 1000;
+ const report = {
+ benchmark: benchmarkName,
+ startedAt,
+ dataset: datasetMetadata,
+ runtime: runtimeMetadata,
+ summary: summarize(results, elapsedSeconds),
+ results,
+ };
+ const temporaryOutput = `${outputPath}.tmp`;
+ writeFileSync(temporaryOutput, `${JSON.stringify(report, null, 2)}\n`);
+ renameSync(temporaryOutput, outputPath);
+ return report;
+}
+
+for (let index = 0; index < selectedRows.length; index++) {
+ if (index < results.length && results[index].error === undefined) continue;
+ const row = selectedRows[index];
+ const decisionStartedAt = performance.now();
+ responseMetadata = backend === "llama-cpp"
+ ? { costUsd: 0, retryCount: 0 }
+ : { costUsd: null, retryCount: 0 };
+ try {
+ const decision = await decide(row);
+ const result = {
+ id: row.id,
+ gold: row.gold,
+ predicted: decision.predicted,
+ correct: decision.predicted === row.gold,
+ latencyMs: performance.now() - decisionStartedAt,
+ discipline: row.metadata.discipline,
+ difficulty: row.metadata.difficulty,
+ ...decision,
+ };
+ results[index] = result;
+ console.log(`[${index + 1}/${selectedRows.length}] ${row.id} ${result.correct ? "correct" : "wrong"}`
+ + ` ${result.latencyMs.toFixed(1)}ms`);
+ } catch (error) {
+ results[index] = {
+ id: row.id,
+ gold: row.gold,
+ latencyMs: performance.now() - decisionStartedAt,
+ discipline: row.metadata.discipline,
+ difficulty: row.metadata.difficulty,
+ ...responseMetadata,
+ error: error instanceof Error ? `${error.name}: ${error.message}` : String(error),
+ };
+ console.error(
+ `[${index + 1}/${selectedRows.length}] ${row.id} ERROR ${results[index].error}`,
+ );
+ }
+ writeReport();
+}
+
+const report = writeReport();
+console.log(JSON.stringify(report.summary, null, 2));
diff --git a/benchmarks/supergpqa-benchmark.json b/benchmarks/supergpqa-benchmark.json
new file mode 100644
index 0000000..3b0fe71
--- /dev/null
+++ b/benchmarks/supergpqa-benchmark.json
@@ -0,0 +1,143 @@
+{
+ "dataset": {
+ "rows": 1000,
+ "sha256": "d17a48292cc8f081ca576afc7e8eaa6ca510de59fa9881282922409666bb3aff",
+ "randomBaseline": 0.1051944444444431,
+ "sample": "proportional-discipline-difficulty"
+ },
+ "runs": [
+ {
+ "id": "granite-4.0-h-micro",
+ "label": "Granite 4.0 H Micro",
+ "series": "openrouter",
+ "model": "ibm-granite/granite-4.0-h-micro",
+ "provider": "cloudflare",
+ "resolvedProvider": "Cloudflare",
+ "correct": 193,
+ "totalCostUsd": 0.005256013000000002,
+ "medianLatencyMs": 316.5877500000206,
+ "decisionsPerSecond": 2.843996339426331
+ },
+ {
+ "id": "llama-3.1-8b",
+ "label": "Llama 3.1 8B",
+ "series": "openrouter",
+ "model": "meta-llama/llama-3.1-8b-instruct",
+ "provider": "novita",
+ "resolvedProvider": "Novita",
+ "labelDy": 27,
+ "correct": 190,
+ "totalCostUsd": 0.006144179999999994,
+ "medianLatencyMs": 625.0682000000379,
+ "decisionsPerSecond": 1.3287550260398036
+ },
+ {
+ "id": "glm-4.7-flash",
+ "label": "GLM 4.7 Flash",
+ "series": "openrouter",
+ "model": "z-ai/glm-4.7-flash",
+ "provider": "cloudflare",
+ "resolvedProvider": "Cloudflare",
+ "correct": 257,
+ "totalCostUsd": 0.017208291499999993,
+ "medianLatencyMs": 261.28954999995767,
+ "decisionsPerSecond": 1.1753601925513055
+ },
+ {
+ "id": "glm-5.2",
+ "label": "GLM 5.2",
+ "series": "openrouter",
+ "model": "z-ai/glm-5.2",
+ "provider": "cloudflare",
+ "resolvedProvider": "Cloudflare",
+ "correct": 446,
+ "totalCostUsd": 0.39322003999999994,
+ "medianLatencyMs": 1068.5187000000028,
+ "decisionsPerSecond": 0.69116943608323
+ },
+ {
+ "id": "gemma-4-26b",
+ "label": "Gemma 4 26B",
+ "series": "openrouter",
+ "model": "google/gemma-4-26b-a4b-it",
+ "provider": "dekallm",
+ "resolvedProvider": "DekaLLM",
+ "correct": 376,
+ "totalCostUsd": 0.0184062,
+ "medianLatencyMs": 384.1862999999921,
+ "decisionsPerSecond": 2.1881737925945197
+ },
+ {
+ "id": "granite-4.2-8b",
+ "label": "Granite 4.2 8B",
+ "series": "openrouter",
+ "model": "ibm-granite/granite-4.2-8b",
+ "provider": "coreweave",
+ "resolvedProvider": "CoreWeave",
+ "labelDy": 22,
+ "correct": 238,
+ "totalCostUsd": 0.029343299999999975,
+ "medianLatencyMs": 290.28174999999464,
+ "decisionsPerSecond": 3.1536941543271486
+ },
+ {
+ "id": "deepseek-v4.1-flash",
+ "label": "DeepSeek V4.1 Flash",
+ "series": "openrouter",
+ "model": "deepseek/deepseek-v4.1-flash",
+ "provider": "wafer",
+ "resolvedProvider": "Wafer",
+ "correct": 456,
+ "totalCostUsd": 0.054357200000000015,
+ "medianLatencyMs": 641.0851999999941,
+ "decisionsPerSecond": 1.3966703323131509
+ },
+ {
+ "id": "deepseek-v4-pro",
+ "label": "DeepSeek V4 Pro",
+ "series": "openrouter",
+ "model": "deepseek/deepseek-v4-pro-0813",
+ "provider": "cloudflare",
+ "resolvedProvider": "Cloudflare",
+ "labelDy": 27,
+ "correct": 432,
+ "totalCostUsd": 0.35875751999999944,
+ "medianLatencyMs": 787.4300499999663,
+ "decisionsPerSecond": 0.8806034365130117
+ },
+ {
+ "id": "kimi-k3",
+ "label": "Kimi K3",
+ "series": "openrouter",
+ "model": "moonshotai/kimi-k3",
+ "provider": "morph",
+ "resolvedProvider": "Morph",
+ "labelDx": -10,
+ "labelAnchor": "end",
+ "correct": 593,
+ "totalCostUsd": 0.6242506249999994,
+ "medianLatencyMs": 1296.1853999999585,
+ "decisionsPerSecond": 0.7251960952904043
+ },
+ {
+ "id": "jev-1.13",
+ "label": "Jev 1.13",
+ "series": "jev",
+ "model": "typesafe/jev-1.13",
+ "correct": 536,
+ "totalCostUsd": 0.024364368000000036,
+ "medianLatencyMs": 325.2531000000017,
+ "decisionsPerSecond": 2.8591025350335553
+ }
+ ],
+ "localReference": {
+ "id": "qwen3.8-27b-local",
+ "label": "Local Qwen3.8 27B Q4_XL",
+ "series": "llama-cpp",
+ "model": "qwen3.8-27b-text-64k",
+ "correct": 400,
+ "totalCostUsd": 0,
+ "medianLatencyMs": 555.8330999999889,
+ "decisionsPerSecond": 1.342476004288219
+ }
+}
diff --git a/benchmarks/supergpqa-benchmark.svg b/benchmarks/supergpqa-benchmark.svg
new file mode 100644
index 0000000..d67780d
--- /dev/null
+++ b/benchmarks/supergpqa-benchmark.svg
@@ -0,0 +1,87 @@
+
diff --git a/tests/benchmark.test.mjs b/tests/benchmark.test.mjs
index bd7a1fa..f9e3f50 100644
--- a/tests/benchmark.test.mjs
+++ b/tests/benchmark.test.mjs
@@ -76,6 +76,19 @@ test("OpenRouter benchmark creates its output directory before inference", () =>
assert.match(report.results[0].error, /Intentional benchmark test failure/);
});
+test("SuperGPQA pilot runs are limited to 100 questions", () => {
+ const result = run("run-supergpqa.mjs", [
+ "--backend", "llama-cpp",
+ "--base-url", "http://127.0.0.1:8080",
+ "--model", "test-model",
+ "--sample-method", "balanced",
+ "--sample-size", "101",
+ ]);
+
+ assert.notEqual(result.status, 0);
+ assert.match(result.stderr, /must not exceed 100 for balanced sampling/);
+});
+
test("comparator creates a nested output for valid reports", () => {
const qwen = temporaryPath("qwen.json");
const jev = temporaryPath("jev.json");
diff --git a/tests/supergpqa.test.mjs b/tests/supergpqa.test.mjs
new file mode 100644
index 0000000..d25ded9
--- /dev/null
+++ b/tests/supergpqa.test.mjs
@@ -0,0 +1,112 @@
+import assert from "node:assert/strict";
+import test from "node:test";
+import {
+ prepareSuperGpqaRows,
+ sampleSuperGpqaPilotRows,
+ sampleSuperGpqaEvaluationRows,
+ sampleSuperGpqaStratifiedRows,
+} from "../benchmarks/prepare-supergpqa.mjs";
+
+function row(overrides = {}) {
+ return {
+ uuid: "case-1",
+ question: "Which answer is correct?",
+ options: ["First", "Second"],
+ answer: "Second",
+ answer_letter: "B",
+ discipline: "Science",
+ field: "Physics",
+ subfield: "Mechanics",
+ difficulty: "middle",
+ is_calculation: false,
+ ...overrides,
+ };
+}
+
+test("prepares choices and validates the answer letter against its text", () => {
+ const [prepared] = prepareSuperGpqaRows([row()]);
+ assert.deepEqual(prepared.choices, { A: "First", B: "Second" });
+ assert.equal(prepared.gold, "B");
+ assert.deepEqual(prepared.metadata, {
+ discipline: "Science",
+ field: "Physics",
+ subfield: "Mechanics",
+ difficulty: "middle",
+ isCalculation: false,
+ });
+ assert.throws(() => prepareSuperGpqaRows([row({ answer: "First" })]), /inconsistent answer/);
+});
+
+test("samples discipline and difficulty strata deterministically", () => {
+ const source = [];
+ for (const discipline of ["Science", "Law"]) {
+ for (const difficulty of ["easy", "hard"]) {
+ for (let index = 0; index < 3; index++) {
+ source.push(row({
+ uuid: `${discipline}-${difficulty}-${index}`,
+ discipline,
+ difficulty,
+ }));
+ }
+ }
+ }
+ const prepared = prepareSuperGpqaRows(source);
+ const sample = sampleSuperGpqaPilotRows(prepared, 8);
+ const counts = sample.reduce((map, item) => {
+ const key = `${item.metadata.discipline}/${item.metadata.difficulty}`;
+ map[key] = (map[key] ?? 0) + 1;
+ return map;
+ }, {});
+ assert.deepEqual(Object.values(counts).sort(), [2, 2, 2, 2]);
+ assert.deepEqual(
+ sample.map(({ id }) => id),
+ sampleSuperGpqaPilotRows([...prepared].reverse(), 8).map(({ id }) => id),
+ );
+});
+
+test("rejects malformed rows and sample sizes", () => {
+ assert.throws(() => prepareSuperGpqaRows([row(), row()]), /duplicate row ID/);
+ assert.throws(() => prepareSuperGpqaRows([row({ options: ["First"] })]), /invalid options/);
+ assert.throws(() => prepareSuperGpqaRows([row({ answer_letter: "C" })]), /outside its options/);
+ assert.throws(
+ () => sampleSuperGpqaPilotRows(prepareSuperGpqaRows([row()]), 2),
+ /sample size/,
+ );
+});
+
+test("samples discipline and difficulty strata proportionally and deterministically", () => {
+ const source = [];
+ for (let index = 0; index < 2; index++) {
+ source.push(row({ uuid: `Law-hard-${index}`, discipline: "Law", difficulty: "hard" }));
+ }
+ for (let index = 0; index < 8; index++) {
+ source.push(row({ uuid: `Science-easy-${index}`, discipline: "Science", difficulty: "easy" }));
+ }
+ const prepared = prepareSuperGpqaRows(source);
+ const sample = sampleSuperGpqaStratifiedRows(prepared, 5);
+ assert.equal(sample.filter(({ metadata }) => metadata.discipline === "Law").length, 1);
+ assert.equal(sample.filter(({ metadata }) => metadata.discipline === "Science").length, 4);
+ assert.deepEqual(
+ sample.map(({ id }) => id),
+ sampleSuperGpqaStratifiedRows([...prepared].reverse(), 5).map(({ id }) => id),
+ );
+});
+
+test("keeps the evaluation sample disjoint from the pilot sample", () => {
+ const source = [];
+ for (const discipline of ["Science", "Law"]) {
+ for (let index = 0; index < 10; index++) {
+ source.push(row({ uuid: `${discipline}-${index}`, discipline }));
+ }
+ }
+ const prepared = prepareSuperGpqaRows(source);
+ const pilotIds = new Set(
+ sampleSuperGpqaPilotRows(prepared, 4).map(({ id }) => id),
+ );
+ const evaluation = sampleSuperGpqaEvaluationRows(prepared, 10, 4);
+ assert.equal(evaluation.some(({ id }) => pilotIds.has(id)), false);
+ assert.deepEqual(
+ evaluation.map(({ id }) => id),
+ sampleSuperGpqaEvaluationRows([...prepared].reverse(), 10, 4).map(({ id }) => id),
+ );
+});