diff --git a/optimized_solutions/challenge-01/factor-ablation.svg b/optimized_solutions/challenge-01/factor-ablation.svg
new file mode 100644
index 0000000..7d12a1c
--- /dev/null
+++ b/optimized_solutions/challenge-01/factor-ablation.svg
@@ -0,0 +1,691 @@
+
+
+
diff --git a/optimized_solutions/challenge-02/factor-ablation.svg b/optimized_solutions/challenge-02/factor-ablation.svg
new file mode 100644
index 0000000..5434b25
--- /dev/null
+++ b/optimized_solutions/challenge-02/factor-ablation.svg
@@ -0,0 +1,571 @@
+
+
+
diff --git a/optimized_solutions/challenge-03/factor-ablation.svg b/optimized_solutions/challenge-03/factor-ablation.svg
new file mode 100644
index 0000000..dbc801b
--- /dev/null
+++ b/optimized_solutions/challenge-03/factor-ablation.svg
@@ -0,0 +1,601 @@
+
+
+
diff --git a/optimized_solutions/challenge-04/factor-ablation.svg b/optimized_solutions/challenge-04/factor-ablation.svg
new file mode 100644
index 0000000..5886950
--- /dev/null
+++ b/optimized_solutions/challenge-04/factor-ablation.svg
@@ -0,0 +1,571 @@
+
+
+
diff --git a/optimized_solutions/challenge-05/factor-ablation.svg b/optimized_solutions/challenge-05/factor-ablation.svg
new file mode 100644
index 0000000..27b438b
--- /dev/null
+++ b/optimized_solutions/challenge-05/factor-ablation.svg
@@ -0,0 +1,591 @@
+
+
+
diff --git a/optimized_solutions/challenge-06/factor-ablation.svg b/optimized_solutions/challenge-06/factor-ablation.svg
new file mode 100644
index 0000000..5738920
--- /dev/null
+++ b/optimized_solutions/challenge-06/factor-ablation.svg
@@ -0,0 +1,663 @@
+
+
+
diff --git a/optimized_solutions/challenge-07/factor-ablation.svg b/optimized_solutions/challenge-07/factor-ablation.svg
new file mode 100644
index 0000000..7670022
--- /dev/null
+++ b/optimized_solutions/challenge-07/factor-ablation.svg
@@ -0,0 +1,558 @@
+
+
+
diff --git a/optimized_solutions/challenge-08/factor-ablation.svg b/optimized_solutions/challenge-08/factor-ablation.svg
new file mode 100644
index 0000000..00dcbfc
--- /dev/null
+++ b/optimized_solutions/challenge-08/factor-ablation.svg
@@ -0,0 +1,607 @@
+
+
+
diff --git a/optimized_solutions/challenge-09/factor-ablation.svg b/optimized_solutions/challenge-09/factor-ablation.svg
new file mode 100644
index 0000000..677aa87
--- /dev/null
+++ b/optimized_solutions/challenge-09/factor-ablation.svg
@@ -0,0 +1,639 @@
+
+
+
diff --git a/optimized_solutions/challenge-10/factor-ablation.svg b/optimized_solutions/challenge-10/factor-ablation.svg
new file mode 100644
index 0000000..520cfd7
--- /dev/null
+++ b/optimized_solutions/challenge-10/factor-ablation.svg
@@ -0,0 +1,300 @@
+
+
+
diff --git a/optimized_solutions/challenge-11/factor-ablation.svg b/optimized_solutions/challenge-11/factor-ablation.svg
new file mode 100644
index 0000000..ac7b22a
--- /dev/null
+++ b/optimized_solutions/challenge-11/factor-ablation.svg
@@ -0,0 +1,2527 @@
+
+
+
diff --git a/optimized_solutions/challenge-12/factor-ablation.svg b/optimized_solutions/challenge-12/factor-ablation.svg
new file mode 100644
index 0000000..8fad390
--- /dev/null
+++ b/optimized_solutions/challenge-12/factor-ablation.svg
@@ -0,0 +1,2499 @@
+
+
+
diff --git a/paper/FIGURE_CONTRACT.md b/paper/FIGURE_CONTRACT.md
new file mode 100644
index 0000000..e0a9ad4
--- /dev/null
+++ b/paper/FIGURE_CONTRACT.md
@@ -0,0 +1,39 @@
+# Figure contract
+
+The paper update follows the visual and metric contracts of
+[arXiv:2607.03105](https://arxiv.org/abs/2607.03105).
+
+1. Updated Fig. 1c preserves the agent-by-framework matrix, places all
+ configurations in one continuous layout without an old/new separator, and
+ includes nine added TensorCircuit-NG campaigns through PR #29.
+2. Updated Fig. 2b retains all thirteen model points and the previous non-Astra
+ coordinates. Astra uses eleven paired measurements plus the approved
+ 50 s estimated Task 08 expert reference. Both axes remain linear
+ with extra padding left of zero failure and below unit runtime. No footer,
+ reference-reuse marker, or numeric runtime annotation is added to the figure.
+3. Updated Fig. 4 preserves the paper's 2 × 3 wall-time, token, and efficiency
+ layout. All token bars use the same total-token encoding because the source
+ tables do not publish components for every configuration.
+4. The expert-optimization overview reports paired end-to-end measurements;
+ exact task reductions are distinguished visually; Task 08 remains a valid
+ optimized implementation with a small measured point estimate.
+5. Fable 5 and DeepSeek max remain excluded. Published outcomes are preserved
+ per campaign: Task 08 is failed for the eight pre-Astra configurations and
+ passed for Astra. Astra retains its automatic result; no blanket failure
+ override or retroactive adjudication is applied.
+6. Figure labels omit the repeated `high` qualifier. The accompanying text
+ states that Sol, Terra, Luna, DeepSeek V4 Flash/Pro, Grok 4.5/4.6 and Astra use high
+ thinking effort; the separately named Sol ultra configuration uses ultra.
+7. Astra uses purple in model-specific encodings. Passed/failed and total-token
+ bars retain their common colors; the matrix retains its count scale.
+8. The Astra paired runtime figure supplies the Astra ratio in Fig. 2b: it uses
+ same-host remeasurements for eleven tasks and the estimated Task 08 reference.
+ The raw Task 08 OOM record is preserved separately; the replacement is
+ not reported as a completed measurement. Its source is given in the data
+ and accompanying text, not as an added figure footer.
+ Its right panel reports expert/Astra speedup, not Astra/expert runtime.
+
+Vector PDF is the publication deliverable; PNG copies are non-publication
+previews for inline GitHub/PR display only.
+Agent wall time, token use, artifact-runtime ratio, and expert-implementation
+speedup remain separate quantities.
diff --git a/paper/FIGURE_QA.md b/paper/FIGURE_QA.md
new file mode 100644
index 0000000..df0857b
--- /dev/null
+++ b/paper/FIGURE_QA.md
@@ -0,0 +1,57 @@
+# Figure QA record
+
+Date: 2026-09-05
+
+- Regenerated Fig. 1c, Fig. 2b and Fig. 4; added a separate Astra/expert
+ paired-runtime figure. Each has a vector PDF master and matched 300-dpi PNG.
+- Rendered all four changed PDFs with Poppler and visually inspected them.
+ Adjusted scatter labels to remove overlap; retained direct labels without
+ colored leader lines. Astra uses purple; the count matrix, passed/failed
+ bars, and total-token bars retain their shared encodings.
+- All five PDFs, including the unchanged expert-optimization overview, have
+ one page, embedded TrueType fonts, and no raster image objects. SVG/TIFF
+ are intentionally omitted; PDF is the publication master.
+- Validated 14 agent rows and 108 unique task outcomes. Added-campaign pass
+ counts are **10, 10, 9, 9, 5, 8, 5, 9, 12**. Task 08 source outcomes are
+ preserved: failed for the first eight campaigns, passed for Astra.
+- Cross-checked the 36 new task records and resource totals against pinned
+ PR #27–29 evidence; verified the three original source-file SHA-256 hashes
+ against local Git objects.
+- Fig. 2b retains all thirteen model points and all previous non-Astra
+ coordinates. Astra alone uses the updated artifact comparison. Both axes are linear,
+ with limits (-10, 72) and (-0.6, 9.2) for visual padding, not negative data.
+- Inspected the revised PDF after Poppler rendering: Astra is below the
+ reference anchor; no footer, dagger, numeric Astra runtime annotation,
+ colored leader lines, or label overlap is present.
+- Independently checked the updated ratios and reciprocal speedups: **12**
+ comparisons, consisting of **11 measured pairs and one estimated reference**.
+ Task 08 uses **50.00 s / 16.31 s**, giving **3.065604×** expert/Astra speedup.
+ Astra is faster on **8/12** tasks; geometric mean Astra/expert is
+ **0.539698×**, with totals **440.81 / 640.11 s** (ratio **0.688647×**).
+ The raw Task 08 OOM record and original paired-source hash are unchanged.
+- Re-rendered both runtime figures after the replacement and inspected the
+ actual PDF pages. The per-task plot now contains all 24 runtime bars.
+ Programmatically checked all thirteen scatter coordinates, including
+ Astra at (0, 0.53969753); every non-Astra coordinate remains unchanged.
+ The matrix and solver-resource PNG hashes are also unchanged.
+- Twelve regression tests pass, including rejection of missing tasks, changed
+ outcomes, zero Grok cost, mixed runtime baselines, filled OOM times, and
+ inverted speedups, an incorrect replacement, an estimate labeled as measured,
+ changes to unrelated pairs, stale aggregates, and inflated measured-pair
+ counts. `git diff --check` passes.
+- The earlier 72 task records and all original numeric agent rows are retained.
+ The original framework and expert-reference tables, expert optimizations,
+ insights, and expert-overview PDF/PNG are unchanged from the PR head.
+- Fable 5 and DeepSeek max remain excluded. No benchmark was rerun for this
+ paper-summary refresh, and unrelated worktree changes were not included.
+
+Reproduce from the repository root:
+
+```bash
+python3 paper/verify_updated_results.py
+python3 -m unittest discover -s paper -p test_updated_results.py
+python3 paper/make_updated_paper_figures.py
+```
+
+With the pinned campaign commits available locally, additionally run
+`python3 paper/verify_updated_results.py --check-git-sources`.
diff --git a/paper/README.md b/paper/README.md
new file mode 100644
index 0000000..ad0fb08
--- /dev/null
+++ b/paper/README.md
@@ -0,0 +1,129 @@
+# ORBIT-Q paper-update figures
+
+This directory provides paper-ready updates to Fig. 1c, Fig. 2b, and Fig. 4
+of [arXiv:2607.03105](https://arxiv.org/abs/2607.03105), plus one new summary
+figure for AI-assisted optimization of the 12 human-expert implementations
+and a separate paired Astra/expert runtime comparison. Updated 2026-09-05
+with the complete campaigns in PRs #27, #28 and #29.
+The manuscript masters are vector PDF; matched PNG files are included only for
+inline GitHub/PR previews.
+Fable 5 and DeepSeek max are intentionally excluded. Sol, Terra, Luna,
+DeepSeek V4 Flash/Pro, Grok 4.5/4.6, and Astra use high thinking effort;
+the explicitly named Sol ultra configuration uses ultra.
+
+## Campaign results
+
+| Model | Passed | Solve time (min) | Tokens (M) | Cost (USD) | Runtime / expert (Fig. 2b) | Source |
+|---|---:|---:|---:|---:|---:|---|
+| GPT-5.6 Sol | 10/12 | 197.70 | 26.071 | 25.527 | 2.541× | [#5](https://github.com/sxzgroup/ORBIT-Q/pull/5) |
+| GPT-5.6 Sol ultra | 10/12 | 182.80 | 33.048 | 30.019 | 2.072× | [#6](https://github.com/sxzgroup/ORBIT-Q/pull/6) |
+| GPT-5.6 Terra | 9/12 | 206.43 | 36.804 | 12.037 | 2.702× | [#20](https://github.com/sxzgroup/ORBIT-Q/pull/20) |
+| GPT-5.6 Luna | 9/12 | 239.37 | 74.853 | 2.237 | 4.692× | [#21](https://github.com/sxzgroup/ORBIT-Q/pull/21) |
+| DeepSeek V4 Flash | 5/12 | 290.41 | 81.585 | 0.573 | 3.476× | [#23](https://github.com/sxzgroup/ORBIT-Q/pull/23) |
+| Grok 4.5 | 8/12 | 163.09 | 8.462 | 5.090 | 3.844× | [#26](https://github.com/sxzgroup/ORBIT-Q/pull/26) |
+| DeepSeek V4 Pro | 5/12 | 324.22 | 67.180 | 1.101 | 2.588× | [#27](https://github.com/sxzgroup/ORBIT-Q/pull/27) |
+| Grok 4.6 | 9/12 | 199.17 | 29.767 | 20.130 | 2.191× | [#28](https://github.com/sxzgroup/ORBIT-Q/pull/28) |
+| GPT-6 Astra | 12/12 | 44.72 | 6.835 | 13.967 | 0.540× | [#29](https://github.com/sxzgroup/ORBIT-Q/pull/29) |
+
+Reported outcomes retain each campaign's published audit/adjudication basis.
+Astra's 12/12 is the original automatic verifier result, not a newly applied
+blanket human adjudication. The maintainer's subsequent source review accepts
+its Task 08 reduction ([review](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5548928438)).
+Earlier campaigns' Task 08 failures are not retroactively changed.
+
+## Updated Fig. 1c — agent–framework benchmark matrix
+
+[PDF](updated_figures/fig1c_updated_agent_framework_matrix.pdf)
+
+The matrix presents all reported configurations in one continuous layout,
+without visually separating earlier and later evaluations. The added
+TensorCircuit-NG results are Sol **10/12**, Sol ultra **10/12**, Terra **9/12**,
+Luna **9/12**, DeepSeek V4 Flash **5/12**, Grok 4.5 **8/12**, DeepSeek V4 Pro
+**5/12**, Grok 4.6 **9/12**, and Astra **12/12**. The count-based matrix palette
+is unchanged; model-specific scatter markers and the Astra label use purple.
+
+## Updated Fig. 2b — failure rate versus artifact runtime
+
+[PDF](updated_figures/fig2b_updated_agent_axis.pdf)
+
+All thirteen model points are retained. Astra uses the runtime comparison
+below; other points retain their previous fixed-paper-reference values.
+The linear axes add space left of zero failure and below unit runtime.
+
+## Updated Fig. 4 — benchmark resource use
+
+[PDF](updated_figures/fig4_updated_agent_framework_resources.pdf)
+
+The original 2 × 3 layout is preserved. Panels (a–c) combine all fourteen agent
+configurations; panels (d–f) retain the paper's original framework comparison.
+Panels (b) and (e) use one consistent total-token encoding for every
+configuration because the paper tables do not publish token components for
+every legacy configuration. Bubble area represents total solver cost. All
+three added models appear in wall-time, token, and cost panels. Their costs
+per valid solution are **$0.220223** (DeepSeek Pro), **$2.236615** (Grok 4.6),
+and **$1.163920** (Astra). Cost values retain the source campaigns' accounting.
+
+## Astra / expert artifact runtime
+
+[PDF](updated_figures/astra_expert_paired_runtime.pdf)
+
+The [PR #29 runtime reply](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804)
+reports fresh runs on the same Apple M2 host, pinned TensorCircuit image and
+6 CPU / 10 GiB limits. Following the [reviewer's suggestion](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776),
+Task 08 uses an **estimated 50 s** expert reference (25.05 s × 2, rounded),
+compared with Astra's measured 16.31 s. The other eleven pairs are unchanged.
+With this replacement, Astra is faster on **8/12** tasks. The geometric mean
+Astra/expert ratio is **0.540×** (equivalently **1.85×** speedup), and totals
+are **440.81 s / 640.11 s = 0.689×**. This includes 11 measured pairs and
+one estimated expert reference, not 12 completed paired measurements.
+The plot's right panel uses the reciprocal expert/Astra speedup, so >1 means
+Astra is faster. Only the two exported Task 04 field names were restored in
+its execution copy; the algorithms and original archive were unchanged.
+The [updated table](astra_runtime_update.md) and
+[plot data](source_data/astra_expert_runtime.csv) record the replacement.
+The original [paired measurements](source_data/astra_expert_paired.csv),
+including the missing expert Task 08 runtime, remain unchanged.
+
+## New figure — human expert + AI co-optimization
+
+[PDF](updated_figures/expert_optimization_updated_overview.pdf)
+
+Panel (a) compares end-to-end runtime before and after optimization on a log
+scale. Panel (b) reports the paired end-to-end speedup and the dominant insight
+retained after ablation. The largest result is the exact Task 07 reduction
+(**45.76×**); Task 01 reaches **9.64×**, while Tasks 03, 09, 10, and 12 reach
+**3.82–4.90×**. Task 08 is included as a valid optimized implementation with a
+small **1.04×** paired point estimate.
+This concerns the optimized human-expert implementation and is separate from
+the Task 08 benchmark outcomes of the solver campaigns above.
+The legal TensorCircuit-native Task 05 result is used (**1.939× mean paired**).
+This is the arithmetic mean of five paired public-expert/final-candidate runtime
+ratios, not the ratio of the two reported mean runtimes. The candidate won all
+five pairs; the median is **1.668×**. One `180.22 s` expert-reference long tail
+raises the mean, while the four-pair sensitivity mean is **1.639×**. The direct
+ablation attributes **1.420×** to the retained OMECo `4×4` path-search budget;
+gate fusion alone remains unresolved.
+
+Individual factor-removal plots remain available as supplementary evidence:
+
+| Task | End-to-end result | Upstream PR |
+|---:|---:|---:|
+| 01 | **9.636×** | [#8](https://github.com/sxzgroup/ORBIT-Q/pull/8) |
+| 02 | **1.116×** | [#9](https://github.com/sxzgroup/ORBIT-Q/pull/9) |
+| 03 | **4.894×** | [#10](https://github.com/sxzgroup/ORBIT-Q/pull/10) |
+| 04 | **2.602×** | [#11](https://github.com/sxzgroup/ORBIT-Q/pull/11) |
+| 05 | **1.939× mean paired** | [#19](https://github.com/sxzgroup/ORBIT-Q/pull/19) |
+| 06 | **1.504×** | [#13](https://github.com/sxzgroup/ORBIT-Q/pull/13) |
+| 07 | **45.758×** | [#7](https://github.com/sxzgroup/ORBIT-Q/pull/7) |
+| 08 | **1.045×** | [#18](https://github.com/sxzgroup/ORBIT-Q/pull/18) |
+| 09 | **3.822×** | [#14](https://github.com/sxzgroup/ORBIT-Q/pull/14) |
+| 10 | **4.898×** | [#15](https://github.com/sxzgroup/ORBIT-Q/pull/15) |
+| 11 | **1.464×** | [#16](https://github.com/sxzgroup/ORBIT-Q/pull/16) |
+| 12 | **3.914×** | [#17](https://github.com/sxzgroup/ORBIT-Q/pull/17) |
+
+Source tables, adjudication notes, and the reproducible plotting script are in
+[`source_data/`](source_data/) and
+[`make_updated_paper_figures.py`](make_updated_paper_figures.py).
+Run `python3 paper/verify_updated_results.py` from the repository root to check
+the tables without plotting dependencies; see [FIGURE_QA.md](FIGURE_QA.md)
+for regression tests and rendering checks.
diff --git a/paper/astra_runtime_update.md b/paper/astra_runtime_update.md
new file mode 100644
index 0000000..0e764a4
--- /dev/null
+++ b/paper/astra_runtime_update.md
@@ -0,0 +1,28 @@
+# Astra artifact runtime update
+
+Following the [reviewer's suggestion](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776), Task 08 uses an estimated expert reference of **50 s** (25.05 s × 2, rounded). Its Astra runtime remains the measured **16.31 s**. The other eleven pairs are unchanged same-host measurements. Original measurement records are preserved separately.
+
+Across all 12 tasks, the geometric mean Astra/expert runtime ratio is **0.540×**, equivalent to **1.85×** geometric-mean speedup. Astra is faster on **8/12** tasks. Summed times are **440.81 s / 640.11 s = 0.689×**. This comparison contains 11 measured pairs and one estimated expert reference.
+
+| Task | Expert reference (s) | Astra (s) | Astra / expert |
+|---|---:|---:|---:|
+| 01 | 58.59 | 13.68 | 0.233× |
+| 02 | 4.15 | 9.84 | 2.371× |
+| 03 | 3.82 | 5.10 | 1.335× |
+| 04 | 13.13 | 2.33 | 0.177× |
+| 05 | 91.87 | 5.26 | 0.057× |
+| 06 | 44.26 | 104.66 | 2.365× |
+| 07 | 147.46 | 21.22 | 0.144× |
+| 08 | 50.00 | 16.31 | 0.326× |
+| 09 | 23.64 | 8.63 | 0.365× |
+| 10 | 21.51 | 105.57 | 4.908× |
+| 11 | 171.11 | 143.04 | 0.836× |
+| 12 | 10.57 | 5.17 | 0.489× |
+
+
+
+Panel (b) shows expert/Astra speedup, the reciprocal of the table's runtime ratio.
+
+
+
+Only Astra's artifact-runtime coordinate changes; all other model points and solver-cost figures are unchanged.
diff --git a/paper/make_updated_paper_figures.py b/paper/make_updated_paper_figures.py
new file mode 100644
index 0000000..c594450
--- /dev/null
+++ b/paper/make_updated_paper_figures.py
@@ -0,0 +1,510 @@
+#!/usr/bin/env python3
+"""Generate vector-ready ORBIT-Q paper-update figures.
+
+The layouts deliberately follow arXiv:2607.03105: Fig. 1c is an
+agent-by-framework matrix, Fig. 2b is the TC agent-axis failure/runtime
+scatter, and Fig. 4 keeps the original 2 x 3 resource-use structure.
+"""
+
+from __future__ import annotations
+
+import csv
+import math
+from pathlib import Path
+
+import matplotlib as mpl
+mpl.use("Agg")
+import matplotlib.pyplot as plt
+import numpy as np
+from matplotlib.colors import LinearSegmentedColormap, Normalize
+from matplotlib.patches import Patch, Rectangle
+
+
+ROOT = Path(__file__).resolve().parent
+DATA = ROOT / "source_data"
+OUT = ROOT / "updated_figures"
+
+mpl.rcParams.update(
+ {
+ "font.family": "sans-serif",
+ "font.sans-serif": ["Arial", "Helvetica", "DejaVu Sans"],
+ "font.size": 8.5,
+ "axes.labelsize": 9,
+ "axes.titlesize": 10,
+ "xtick.labelsize": 8,
+ "ytick.labelsize": 8,
+ "legend.fontsize": 7.5,
+ "axes.linewidth": 0.8,
+ "grid.color": "#D6D6D6",
+ "grid.linewidth": 0.6,
+ "grid.alpha": 0.70,
+ "svg.fonttype": "none",
+ "pdf.fonttype": 42,
+ "ps.fonttype": 42,
+ "savefig.facecolor": "white",
+ }
+)
+
+COLORS = {
+ "gpt55": "#0072B2",
+ "gpt55_checklist": "#56B4E9",
+ "opus48": "#D55E00",
+ "glm52": "#A05195",
+ "sonnet46": "#E69F00",
+ "sol56_high": "#2563EB",
+ "sol56_ultra": "#F97316",
+ "terra56_high": "#009E73",
+ "luna56_high": "#CC79A7",
+ "deepseek_v4_flash_high": "#4D4D4D",
+ "grok45_high": "#111111",
+ "deepseek_v4_pro_high": "#007F87",
+ "grok46_high": "#566573",
+ "astra_high": "#7B2CBF",
+ "tensorcircuit": "#0072B2",
+ "pennylane": "#CC79A7",
+ "torchquantum": "#E69F00",
+ "mindquantum": "#009E73",
+}
+
+PASS = "#009E73"
+FAIL = "#FFFFFF"
+CACHE = "#56B4E9"
+
+
+def read_csv(name: str) -> list[dict[str, str]]:
+ with (DATA / name).open(newline="", encoding="utf-8") as f:
+ return list(csv.DictReader(f))
+
+
+def num(row: dict[str, str], key: str) -> float:
+ value = row.get(key, "")
+ return float(value) if value not in (None, "") else np.nan
+
+
+def clean_axes(ax: mpl.axes.Axes, *, grid_axis: str = "both") -> None:
+ ax.grid(True, axis=grid_axis, zorder=0)
+ ax.spines["top"].set_visible(False)
+ ax.spines["right"].set_visible(False)
+ ax.tick_params(direction="out", length=3, width=0.8)
+
+
+def panel_label(ax: mpl.axes.Axes, label: str) -> None:
+ ax.text(
+ -0.11,
+ 1.04,
+ label,
+ transform=ax.transAxes,
+ fontsize=12,
+ fontweight="bold",
+ va="bottom",
+ ha="left",
+ )
+
+
+def save_all(fig: mpl.figure.Figure, stem: str) -> None:
+ OUT.mkdir(parents=True, exist_ok=True)
+ fig.savefig(OUT / f"{stem}.pdf", bbox_inches="tight")
+ fig.savefig(OUT / f"{stem}.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+
+def validate_new_slowdowns(agent_rows: list[dict[str, str]]) -> None:
+ """Check archived fixed-paper ratios; these no longer drive Fig. 2b."""
+ refs = {int(r["task"]): num(r, "expert_tc_runtime_sec") for r in read_csv("paper_expert_runtimes.csv")}
+ outcomes = read_csv("task_outcomes.csv")
+ model_to_agent = {
+ "gpt56_sol_high": "sol56_high",
+ "gpt56_sol_ultra": "sol56_ultra",
+ "gpt56_terra_high": "terra56_high",
+ "gpt56_luna_high": "luna56_high",
+ "deepseek_v4_flash_high": "deepseek_v4_flash_high",
+ "grok45_high": "grok45_high",
+ "deepseek_v4_pro_high": "deepseek_v4_pro_high",
+ "grok46_high": "grok46_high",
+ "astra_high": "astra_high",
+ }
+ expected = {r["key"]: num(r, "gm_slowdown") for r in agent_rows if r["series"] == "new"}
+ for model, agent_key in model_to_agent.items():
+ ratios = [
+ num(r, "runtime_sec") / refs[int(r["task"])]
+ for r in outcomes
+ if r["model"] == model and r["final_pass"] == "1"
+ ]
+ recomputed = math.exp(sum(math.log(v) for v in ratios) / len(ratios))
+ if not math.isclose(recomputed, expected[agent_key], rel_tol=0.0, abs_tol=5e-7):
+ raise ValueError(f"{model}: stored GM {expected[agent_key]} != recomputed {recomputed}")
+
+
+def make_fig1c(agent_rows: list[dict[str, str]], framework_rows: list[dict[str, str]]) -> None:
+ agents = [r for r in agent_rows if r["include_fig1c"] == "yes"]
+ order = ["gpt55", "opus48", "glm52", "sonnet46", "sol56_high", "sol56_ultra", "terra56_high", "luna56_high", "deepseek_v4_flash_high", "grok45_high", "deepseek_v4_pro_high", "grok46_high", "astra_high"]
+ agents = sorted(agents, key=lambda r: order.index(r["key"]))
+ frameworks = {r["key"]: r for r in framework_rows}
+
+ values = np.full((4, len(agents)), np.nan)
+ values[0, :] = [int(r["passes"]) for r in agents]
+ values[1, 0] = int(frameworks["pennylane"]["passes"])
+ values[2, 0] = int(frameworks["torchquantum"]["passes"])
+ values[3, 0] = int(frameworks["mindquantum"]["passes"])
+
+ cmap = LinearSegmentedColormap.from_list(
+ "orbit_passes", ["#E9B5AB", "#F3D7A2", "#CFE1C9", "#B8D8D8"]
+ )
+ norm = Normalize(3, 12)
+ fig, ax = plt.subplots(figsize=(10.2, 2.75))
+ ax.set_xlim(-0.85, len(agents))
+ ax.set_ylim(4.25, -1.1)
+ ax.axis("off")
+
+ for i in range(4):
+ for j in range(len(agents)):
+ value = values[i, j]
+ color = "#F7F7F7" if np.isnan(value) else cmap(norm(value))
+ edge = "#D8D8D8" if np.isnan(value) else mpl.colors.to_hex(np.array(mpl.colors.to_rgb(color)) * 0.82)
+ ax.add_patch(Rectangle((j + 0.06, i + 0.06), 0.88, 0.88, facecolor=color, edgecolor=edge, linewidth=1.0))
+ text = "–" if np.isnan(value) else f"{int(value)}/12"
+ ax.text(j + 0.50, i + 0.50, text, ha="center", va="center", fontweight="bold" if not np.isnan(value) else "normal", color="#333333")
+
+ row_labels = ["TC", "PL", "TQ", "MQ"]
+ for i, label in enumerate(row_labels):
+ ax.text(-0.08, i + 0.50, label, ha="right", va="center", fontsize=9.5)
+
+ labels = [
+ "GPT-5.5",
+ "Opus-4.8",
+ "GLM-5.2",
+ "Sonnet-4.6",
+ "5.6 Sol",
+ "5.6 Sol\nultra",
+ "5.6 Terra",
+ "5.6 Luna",
+ "DeepSeek\nV4 Flash",
+ "Grok 4.5",
+ "DeepSeek\nV4 Pro",
+ "Grok 4.6",
+ "GPT-6\nAstra",
+ ]
+ for j, label in enumerate(labels):
+ ax.text(j + 0.50, -0.02, label, ha="center", va="bottom", fontsize=7.7,
+ color=COLORS["astra_high"] if agents[j]["key"] == "astra_high" else "#222222")
+
+ ax.text(-0.68, 2.0, "Framework", rotation=90, ha="center", va="center", fontsize=9.5, fontweight="bold", color="#666666")
+ ax.set_title("Updated agent–framework benchmarking matrix", loc="left", fontweight="bold", pad=18)
+ save_all(fig, "fig1c_updated_agent_framework_matrix")
+
+
+def make_fig2b(agent_rows: list[dict[str, str]]) -> None:
+ rows = [r for r in agent_rows if r["include_fig2b"] == "yes"]
+ paired = read_csv("astra_expert_runtime.csv")
+ paired_ratios = [num(r, "astra_sec") / num(r, "expert_reference_sec")
+ for r in paired]
+ astra_ratio = math.exp(sum(math.log(v) for v in paired_ratios) / len(paired_ratios))
+ fig, ax = plt.subplots(figsize=(5.3, 4.4))
+ clean_axes(ax)
+ ax.axhline(1.0, color="#888888", ls=(0, (3, 3)), lw=0.9, zorder=1)
+ ax.axvline(0.0, color="#888888", ls=(0, (3, 3)), lw=0.9, zorder=1)
+ # The diamond is the normalization anchor, not a measured failure rate
+ # for the expert (Astra's Task 08 expert run did not complete).
+ ax.scatter([0], [1], marker="D", s=82, facecolor="none", edgecolor="#888888", zorder=2)
+ ax.text(1.3, 1.40, "Expert TC\nreference", fontsize=7.5, color="#777777", ha="left", va="center")
+
+ # Preserve the original paper's four agent colors and label positions.
+ # Additional configurations use distinct colors in the same axis.
+ fig2_colors = {
+ "gpt55": "#0072B2",
+ "opus48": "#E69F00",
+ "sonnet46": "#CC79A7",
+ "glm52": "#D55E00",
+ "sol56_high": "#A65628",
+ "sol56_ultra": "#009E73",
+ "terra56_high": "#56B4E9",
+ "luna56_high": "#8C6D31",
+ "deepseek_v4_flash_high": "#4D4D4D",
+ "grok45_high": "#111111",
+ "deepseek_v4_pro_high": COLORS["deepseek_v4_pro_high"],
+ "grok46_high": COLORS["grok46_high"],
+ "astra_high": COLORS["astra_high"],
+ }
+ label_positions = {
+ "gpt55": (18.2, 1.30, "left"),
+ "opus48": (24.0, 3.70, "right"),
+ "sonnet46": (47.2, 8.15, "left"),
+ "glm52": (57.0, 7.10, "left"),
+ "sol56_high": (14.6, 3.00, "right"),
+ "sol56_ultra": (14.2, 2.15, "right"),
+ "terra56_high": (29.5, 2.78, "left"),
+ "luna56_high": (28.2, 5.00, "left"),
+ "deepseek_v4_flash_high": (58.3, 4.08, "center"),
+ "grok45_high": (36.0, 4.12, "left"),
+ "deepseek_v4_pro_high": (58.3, 2.02, "center"),
+ "grok46_high": (30.0, 2.00, "left"),
+ "astra_high": (3.5, 0.08, "left"),
+ }
+ for row in rows:
+ x = 100.0 * int(row["failures"]) / 12.0
+ y = astra_ratio if row["key"] == "astra_high" else num(row, "gm_slowdown")
+ color = fig2_colors[row["key"]]
+ ax.scatter(x, y, marker="o", s=34, facecolor=color, edgecolor="#222222", linewidth=0.8, alpha=0.9, zorder=3)
+ label_x, label_y, align = label_positions[row["key"]]
+ label = f"{row['short_label']}\n({row['passes']}/12)"
+ ax.text(
+ label_x,
+ label_y,
+ label,
+ fontsize=7.2,
+ fontweight="bold",
+ color=color,
+ ha=align,
+ va="center",
+ )
+
+ # Negative limits provide visual padding only; data coordinates are unchanged.
+ ax.set_xlim(-10, 72)
+ ax.set_ylim(-0.6, 9.2)
+ ax.set_xticks(np.arange(0, 71, 10))
+ ax.set_yticks([0, 1, 3, 5, 7, 9])
+ ax.set_xlabel("Failure rate (%)")
+ ax.set_ylabel("Runtime / expert TC reference")
+ panel_label(ax, "(b)")
+ save_all(fig, "fig2b_updated_agent_axis")
+
+
+def make_fig4(agent_rows: list[dict[str, str]], framework_rows: list[dict[str, str]]) -> None:
+ fig = plt.figure(figsize=(16.0, 9.4))
+ gs = fig.add_gridspec(2, 3, width_ratios=[1.15, 1.15, 1.15],
+ height_ratios=[1.25, 1], wspace=0.50, hspace=0.42)
+ axes = [fig.add_subplot(gs[i, j]) for i in range(2) for j in range(3)]
+ axa, axb, axc, axd, axe, axf = axes
+
+ # Agent axis: resource totals for every reported configuration.
+ y = np.arange(len(agent_rows))
+ passed_h = np.array([num(r, "wall_passed_sec") / 3600 for r in agent_rows])
+ failed_h = np.array([num(r, "wall_failed_sec") / 3600 for r in agent_rows])
+ axa.barh(y, passed_h, color=PASS, edgecolor="#222222", linewidth=0.7, label="Passed tasks", zorder=2)
+ axa.barh(y, failed_h, left=passed_h, color=FAIL, edgecolor="#222222", linewidth=0.7, label="Failed tasks", zorder=2)
+ axa.set_yticks(y, [r["short_label"] for r in agent_rows])
+ axa.invert_yaxis()
+ axa.set_xlabel("Agent solve wall time (h)")
+ axa.set_title("Agent axis", fontweight="bold")
+ axa.legend(loc="lower right", frameon=False)
+ clean_axes(axa, grid_axis="x")
+ panel_label(axa, "(a)")
+
+ axb.barh(
+ y,
+ [num(row, "total_tokens_m") for row in agent_rows],
+ color=CACHE,
+ edgecolor="#222222",
+ linewidth=0.7,
+ zorder=2,
+ )
+ axb.set_yticks(y, [r["short_label"] for r in agent_rows])
+ axb.invert_yaxis()
+ axb.set_xlabel("Total solving-side tokens (million)")
+ clean_axes(axb, grid_axis="x")
+ panel_label(axb, "(b)")
+
+ agent_label_positions = {
+ "gpt55": (6.2, 1.98),
+ "gpt55_checklist": (9.0, 1.10),
+ "opus48": (22.7, 1.82),
+ "glm52": (34.0, 2.50),
+ "sonnet46": (32.0, 1.72),
+ "sol56_high": (11.5, 2.74),
+ "sol56_ultra": (6.0, 3.10),
+ "terra56_high": (26.5, 1.47),
+ "luna56_high": (25.5, 0.43),
+ "deepseek_v4_flash_high": (45.0, 0.35),
+ "grok45_high": (17.0, 0.79),
+ "deepseek_v4_pro_high": (60.5, 0.60),
+ "grok46_high": (25.0, 2.30),
+ "astra_high": (0.8, 0.80),
+ }
+ for row in agent_rows:
+ x = num(row, "wall_total_sec") / 60 / int(row["passes"])
+ yy = num(row, "cost_usd") / int(row["passes"])
+ if np.isnan(yy):
+ continue
+ size = 30 + 11 * num(row, "cost_usd")
+ color = COLORS[row["key"]]
+ axc.scatter(x, yy, s=size, marker="o", facecolor=color, edgecolor="#222222", linewidth=0.8, alpha=0.95, zorder=3)
+ label_x, label_y = agent_label_positions[row["key"]]
+ axc.annotate(
+ f"{row['short_label']}\n({row['passes']}/12)",
+ (x, yy),
+ xytext=(label_x, label_y),
+ textcoords="data",
+ fontsize=6.4,
+ fontweight="bold",
+ color=color,
+ va="center",
+ )
+ axc.set_xlim(0, 76)
+ axc.set_ylim(0, 3.35)
+ axc.set_xlabel("Solve time per valid solution (min)")
+ axc.set_ylabel("Solver cost per valid solution (USD)")
+ axc.text(0.98, 0.97, "Marker area scales with total solver cost", transform=axc.transAxes, ha="right", va="top", fontsize=7.2, color="#444444")
+ clean_axes(axc)
+ panel_label(axc, "(c)")
+
+ # Framework axis: unchanged original-paper comparison.
+ fy = np.arange(len(framework_rows))
+ fpass = np.array([num(r, "wall_passed_sec") / 3600 for r in framework_rows])
+ ffail = np.array([num(r, "wall_failed_sec") / 3600 for r in framework_rows])
+ axd.barh(fy, fpass, color=PASS, edgecolor="#222222", linewidth=0.7, zorder=2)
+ axd.barh(fy, ffail, left=fpass, color=FAIL, edgecolor="#222222", linewidth=0.7, zorder=2)
+ axd.set_yticks(fy, [r["short_label"] for r in framework_rows])
+ axd.invert_yaxis()
+ axd.set_xlabel("Agent solve wall time (h)")
+ axd.set_title("Framework axis", fontweight="bold")
+ clean_axes(axd, grid_axis="x")
+ panel_label(axd, "(d)")
+
+ axe.barh(fy, [num(r, "total_tokens_m") for r in framework_rows], color=CACHE, edgecolor="#222222", linewidth=0.7, zorder=2)
+ axe.set_yticks(fy, [r["short_label"] for r in framework_rows])
+ axe.invert_yaxis()
+ axe.set_xlabel("Total solving-side tokens (million)")
+ clean_axes(axe, grid_axis="x")
+ panel_label(axe, "(e)")
+
+ foffsets = {"tensorcircuit": (8, 6), "pennylane": (8, 5), "torchquantum": (-48, -13), "mindquantum": (-48, 8)}
+ for row in framework_rows:
+ x = num(row, "wall_total_sec") / 60 / int(row["passes"])
+ yy = num(row, "cost_usd") / int(row["passes"])
+ size = 30 + 11 * num(row, "cost_usd")
+ color = COLORS[row["key"]]
+ axf.scatter(x, yy, s=size, facecolor=color, edgecolor="#222222", linewidth=0.8, zorder=3)
+ dx, dy = foffsets[row["key"]]
+ axf.annotate(f"{row['short_label']}\n({row['passes']}/12)", (x, yy), xytext=(dx, dy), textcoords="offset points", fontsize=7.2, fontweight="bold", color=color, va="center")
+ axf.set_xlim(7, 45)
+ axf.set_ylim(0.8, 8.1)
+ axf.set_xlabel("Solve time per valid solution (min)")
+ axf.set_ylabel("Recorded cost per valid solution (USD)")
+ clean_axes(axf)
+ panel_label(axf, "(f)")
+
+ fig.suptitle("Updated ORBIT-Q benchmark resource use", fontsize=13, fontweight="bold", y=0.995)
+ save_all(fig, "fig4_updated_agent_framework_resources")
+
+
+def make_expert_optimization() -> None:
+ rows = read_csv("expert_optimization.csv")
+ tasks = [r["task"] for r in rows]
+ baseline = np.array([num(r, "baseline_sec") for r in rows])
+ optimized = np.array([num(r, "optimized_sec") for r in rows])
+ speedup = np.array([num(r, "speedup") for r in rows])
+ short = {
+ "01": "batched gate construction",
+ "02": "batched exact purity",
+ "03": "exact product-state reduction",
+ "04": "batched probe networks",
+ "05": "tuned OMECo path search",
+ "06": "TC-native jaxode",
+ "07": "exact ancilla/branch reduction",
+ "08": "bounded mapped sampling",
+ "09": "causal-cone pruning",
+ "10": "fixed contraction program",
+ "11": "layer and onsite fusion",
+ "12": "batched Padé SU4",
+ }
+
+ fig, (axa, axb) = plt.subplots(1, 2, figsize=(12.2, 5.25), gridspec_kw={"width_ratios": [1.08, 1.35], "wspace": 0.35})
+ x = np.arange(len(tasks))
+ width = 0.38
+ axa.bar(x - width / 2, baseline, width, color="#7A7A7A", edgecolor="#222222", linewidth=0.6, label="Original expert", zorder=2)
+ axa.bar(x + width / 2, optimized, width, color=PASS, edgecolor="#222222", linewidth=0.6, label="Human + AI", zorder=2)
+ axa.set_yscale("log")
+ axa.set_xticks(x, tasks)
+ axa.set_xlabel("Challenge")
+ axa.set_ylabel("End-to-end runtime (s, log scale)")
+ axa.legend(frameon=False, loc="upper right")
+ clean_axes(axa, grid_axis="y")
+ panel_label(axa, "(a)")
+
+ y = np.arange(len(tasks))
+ bar_colors = ["#D55E00" if t in {"03", "07"} else PASS for t in tasks]
+ bars = axb.barh(y, speedup, color=bar_colors, edgecolor="#222222", linewidth=0.75, zorder=2)
+ axb.axvline(1.0, color="#777777", ls=(0, (3, 3)), lw=0.9)
+ axb.set_xscale("log")
+ axb.set_xlim(0.9, 78)
+ axb.set_yticks(y, [f"Task {t}" for t in tasks])
+ axb.invert_yaxis()
+ axb.set_xlabel("End-to-end speedup (×, log scale)")
+ clean_axes(axb, grid_axis="x")
+ panel_label(axb, "(b)")
+ for i, (t, value) in enumerate(zip(tasks, speedup)):
+ axb.text(value * 1.06, i, f"{value:.2f}× {short[t]}", va="center", ha="left", fontsize=7.2, color="#333333")
+
+ axb.legend(
+ handles=[
+ Patch(facecolor=PASS, edgecolor="#222222", label="Framework-native optimization"),
+ Patch(facecolor="#D55E00", edgecolor="#222222", label="Exact task reduction"),
+ ],
+ loc="lower right",
+ frameon=False,
+ )
+ fig.suptitle("Human-expert implementations after AI-assisted optimization", fontsize=12.5, fontweight="bold", y=0.995)
+ save_all(fig, "expert_optimization_updated_overview")
+
+
+def make_astra_paired() -> None:
+ """Use eleven measured pairs plus the reviewer-approved Task 08 estimate."""
+ rows = read_csv("astra_expert_runtime.csv")
+ tasks = [r["task"] for r in rows]
+ expert = np.array([num(r, "expert_reference_sec") for r in rows])
+ astra = np.array([num(r, "astra_sec") for r in rows])
+ speedups = np.array([num(r, "expert_over_astra") for r in rows])
+ fig, (ax, bx) = plt.subplots(1, 2, figsize=(12.2, 5.25),
+ gridspec_kw={"width_ratios": [1.08, 1.35], "wspace": 0.35})
+ x = np.arange(len(rows))
+ width = 0.38
+ purple = COLORS["astra_high"]
+ ax.bar(x-width/2, expert, width, color="#7A7A7A", edgecolor="#222222",
+ linewidth=0.6, label="Expert reference", zorder=2)
+ ax.bar(x+width/2, astra, width, color=purple, edgecolor="#222222",
+ linewidth=0.6, label="Astra high", zorder=2)
+ ax.set_yscale("log")
+ ax.set_ylim(1, 400)
+ ax.set_xticks(x, tasks)
+ ax.set_xlabel("Challenge")
+ ax.set_ylabel("End-to-end runtime (s, log scale)")
+ ax.legend(frameon=False, loc="upper right")
+ for i, r in enumerate(rows):
+ if np.isnan(expert[i]):
+ ax.text(i-width/2, 1.16, r["expert_status"], color="#7A7A7A",
+ fontsize=6.8, ha="center", va="bottom", rotation=90)
+ bx.barh(x, speedups, color=purple, edgecolor="#222222", linewidth=0.75, zorder=2)
+ bx.axvline(1, color="#777777", ls=(0,(3,3)), lw=0.9)
+ bx.set_xscale("log")
+ bx.set_xlim(0.1, 35)
+ bx.set_yticks(x, [f"Task {t}" for t in tasks])
+ bx.invert_yaxis()
+ bx.set_xlabel("End-to-end speedup (expert / Astra, ×, log scale)")
+ for i, value in enumerate(speedups):
+ if np.isnan(value):
+ bx.text(1.08, i, f"Expert {rows[i]['expert_status']}", color="#555555", va="center", fontsize=7.2)
+ else:
+ bx.text(value*1.08, i, f"{value:.2f}×", va="center", ha="left", fontsize=7.2, color="#333333")
+ for panel, grid_axis, label in [(ax,"y","(a)"),(bx,"x","(b)")]:
+ clean_axes(panel, grid_axis=grid_axis)
+ panel.set_axisbelow(True)
+ panel_label(panel, label)
+ fig.suptitle("Astra artifact runtime relative to the expert reference",
+ fontsize=12.5, fontweight="bold", y=0.995)
+ save_all(fig, "astra_expert_paired_runtime")
+
+
+def main() -> None:
+ agents = read_csv("paper_agent_axis.csv")
+ frameworks = read_csv("paper_framework_axis.csv")
+ validate_new_slowdowns(agents)
+ make_fig1c(agents, frameworks)
+ make_fig2b(agents)
+ make_fig4(agents, frameworks)
+ make_astra_paired()
+ # The earlier expert-optimization measurements and their figure are unchanged.
+ # Call make_expert_optimization() explicitly to reproduce that archived figure.
+
+
+if __name__ == "__main__":
+ main()
diff --git a/paper/source_data/README.md b/paper/source_data/README.md
new file mode 100644
index 0000000..fcf39d0
--- /dev/null
+++ b/paper/source_data/README.md
@@ -0,0 +1,59 @@
+# Source-data notes
+
+- `paper_agent_axis.csv`: original paper agent-axis totals plus Sol, Sol ultra,
+ Terra, Luna, DeepSeek V4 Flash/Pro, Grok 4.5/4.6, and Astra (14 rows). It drives updated Fig. 1c,
+ and the top row of Fig. 4. Its `gm_slowdown` and `include_fig2b` fields
+ drive Fig. 2b except that Astra uses `astra_expert_runtime.csv`.
+ Unqualified added configurations use
+ high thinking effort; Sol ultra uses ultra.
+- `paper_framework_axis.csv`: original paper framework-axis totals for the
+ unchanged bottom row of Fig. 4.
+- `paper_expert_runtimes.csv`: original public per-task TensorCircuit expert
+ references retained to validate the archived fixed-denominator calculations.
+- `benchmark_models.csv`: compact new-campaign aggregate table.
+- `task_outcomes.csv`: 9 configurations × 12 tasks. `raw_reward` preserves the
+ verifier result; `final_pass` retains each source campaign's reported outcome,
+ including its published adjudication where applicable.
+- `recent_pr_evidence.json`: pinned commits, source-file SHA-256 hashes, and
+ normalized per-task evidence for PRs #27–29, plus paired-run provenance.
+- `astra_expert_paired.csv`: 12 fresh paired-runtime records from the
+ [PR #29 reply](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804).
+ Task 08 has a missing expert time (OOM) and is excluded from aggregates.
+- `astra_expert_runtime.csv`: derived plot data for all 12 Astra tasks.
+ Eleven pairs retain the fresh measurements; Task 08 uses the estimated
+ **50 s** expert reference requested in the [PR #29 discussion](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776)
+ (25.05 s × 2, rounded). Its measured Astra time is 16.31 s.
+ This table supplies the Astra Fig. 2b point (**0.539698×**) and runtime bars.
+ `expert_reference_basis` distinguishes the estimate from measurements.
+- `astra_expert_adjustment.json`: replacement provenance, original-source
+ hash, and recomputed aggregate values. Neither the original paired table
+ nor the historical `paper_agent_axis.csv` measurements are overwritten.
+- `expert_optimization.csv`: original and optimized expert runtimes plus the
+ dominant factor retained after ablation.
+- `insights.csv`: short take-home messages for the expert optimizations.
+
+The original values come from the main and supplementary tables of
+[arXiv:2607.03105](https://arxiv.org/abs/2607.03105). The new campaign totals
+come from ORBIT-Q PRs #5, #6, #20, #21, #23, #26, #27, #28, and #29.
+Fig. 2b retains the fixed-paper-reference coordinates for all non-Astra models;
+Astra uses its paired remeasurement with the Task 08 replacement described above.
+The original CSV measurements are unchanged.
+
+The original paper tables expose total token use for the legacy configurations
+but not the full cache/non-cache/output decomposition. Updated Fig. 4b/e
+therefore compare total tokens with one consistent color for every bar.
+
+Task 08 is final `F` for the first eight added solver configurations. Luna and Sol ultra
+retain raw reward `1` in `task_outcomes.csv`; `final_pass=0` records the
+paper-facing decision. Astra retains its original 12/12 automatic result;
+the maintainer subsequently accepted its Task 08 implementation in
+[review](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5548928438).
+The newly added campaigns are not described as uniformly human-adjudicated.
+This solver-campaign adjudication is separate from the
+valid Task 08 human-expert optimization recorded in `expert_optimization.csv`.
+Task 05 uses the legal TensorCircuit-native result from
+[PR #19](https://github.com/sxzgroup/ORBIT-Q/pull/19): **1.939×** mean paired
+end-to-end speedup (median **1.668×**). This is the mean of the five paired
+expert/candidate ratios; one `180.22 s` expert-reference long tail raises it,
+and the four-pair sensitivity mean is **1.639×**. The candidate won 5/5 pairs.
+The earlier custom no-QR MPS result is excluded.
diff --git a/paper/source_data/astra_expert_adjustment.json b/paper/source_data/astra_expert_adjustment.json
new file mode 100644
index 0000000..af75a5e
--- /dev/null
+++ b/paper/source_data/astra_expert_adjustment.json
@@ -0,0 +1,38 @@
+{
+ "schema_version": 1,
+ "date": "2026-09-05",
+ "measured_source": "astra_expert_paired.csv",
+ "measured_source_sha256": "9d151a187c4b73e08c95ced4cfcbaae29afcd34bffdddecb376537710cd34e48",
+ "runtime_table": "astra_expert_runtime.csv",
+ "task08": {
+ "expert_reference_sec": 50,
+ "basis": "estimated",
+ "source": "https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776",
+ "published_reference_sec": 25.05,
+ "reviewer_scale_factor": 2,
+ "reviewer_proposed_sec": 50.1,
+ "rounding": "Nearest whole second (50 s), as approved for this figure update.",
+ "original_expert_status": "OOM",
+ "astra_sec": 16.31
+ },
+ "summary": {
+ "task_count": 12,
+ "measured_pair_count": 11,
+ "estimated_reference_count": 1,
+ "geometric_mean_ratio": 0.539697529957958,
+ "geometric_mean_speedup": 1.8528897104233535,
+ "expert_sum_sec": 640.11,
+ "astra_sum_sec": 440.81,
+ "ratio_of_sums": 0.6886472637515427,
+ "astra_faster_tasks": [
+ 1,
+ 4,
+ 5,
+ 7,
+ 8,
+ 9,
+ 11,
+ 12
+ ]
+ }
+}
diff --git a/paper/source_data/astra_expert_paired.csv b/paper/source_data/astra_expert_paired.csv
new file mode 100644
index 0000000..9ab4c98
--- /dev/null
+++ b/paper/source_data/astra_expert_paired.csv
@@ -0,0 +1,13 @@
+task,expert_sec,astra_sec,astra_over_expert,expert_over_astra,expert_status,astra_status,included_in_aggregate,source
+01,58.59,13.68,0.233487,4.282895,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+02,4.15,9.84,2.371084,0.421748,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+03,3.82,5.1,1.335079,0.74902,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+04,13.13,2.33,0.177456,5.635193,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+05,91.87,5.26,0.057255,17.465779,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+06,44.26,104.66,2.364663,0.422893,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+07,147.46,21.22,0.143903,6.949105,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+08,,16.31,,,OOM,PASS,0,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+09,23.64,8.63,0.365059,2.739282,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+10,21.51,105.57,4.90795,0.203751,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+11,171.11,143.04,0.835953,1.196239,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+12,10.57,5.17,0.48912,2.044487,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
diff --git a/paper/source_data/astra_expert_runtime.csv b/paper/source_data/astra_expert_runtime.csv
new file mode 100644
index 0000000..54654c6
--- /dev/null
+++ b/paper/source_data/astra_expert_runtime.csv
@@ -0,0 +1,13 @@
+task,expert_reference_sec,astra_sec,astra_over_expert,expert_over_astra,expert_reference_basis,source
+01,58.59,13.68,0.2334869431643625,4.282894736842105,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+02,4.15,9.84,2.3710843373493975,0.42174796747967486,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+03,3.82,5.1,1.3350785340314135,0.7490196078431373,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+04,13.13,2.33,0.17745620715917745,5.635193133047211,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+05,91.87,5.26,0.05725481658865788,17.46577946768061,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+06,44.26,104.66,2.3646633529145955,0.422893177909421,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+07,147.46,21.22,0.1439034314390343,6.949104618284638,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+08,50,16.31,0.3262,3.065603923973023,estimated,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776
+09,23.64,8.63,0.3650592216582065,2.73928157589803,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+10,21.51,105.57,4.907949790794978,0.20375106564364878,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+11,171.11,143.04,0.8359534802174039,1.1962388143176736,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
+12,10.57,5.17,0.489120151371807,2.044487427466151,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804
diff --git a/paper/source_data/benchmark_models.csv b/paper/source_data/benchmark_models.csv
new file mode 100644
index 0000000..5715d60
--- /dev/null
+++ b/paper/source_data/benchmark_models.csv
@@ -0,0 +1,10 @@
+model,label,effort,passes,failures,agent_wall_min,total_tokens_m,input_tokens_m,cache_tokens_m,output_tokens_m,cost_usd,cost_per_valid_usd,solve_time_per_valid_min,gm_slowdown,resource_comparable,notes
+gpt56_sol_high,GPT-5.6 Sol,high,10,2,197.70,26.071,25.908,24.199,0.163,25.527,2.553,19.770,2.541,yes,final Task 05 API adjudication; Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references
+gpt56_sol_ultra,GPT-5.6 Sol,ultra,10,2,182.80,33.048,32.790,31.478,0.257,30.019,3.002,18.280,2.072,yes,Task 07 source adjudication pass; Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references
+gpt56_terra_high,GPT-5.6 Terra,high,9,3,206.43,36.804,36.597,35.455,0.207,12.037,1.337,22.937,2.702,yes,Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references
+gpt56_luna_high,GPT-5.6 Luna,high,9,3,239.37,74.853,74.496,72.727,0.357,2.237,0.249,26.597,4.692,yes,Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references
+deepseek_v4_flash_high,DeepSeek V4 Flash,high,5,7,290.41,81.585,80.672,80.008,0.913,0.573,0.115,58.082,3.476,yes,official DeepSeek integration; five final passes; slowdown recomputed against the original paper expert-TC references
+grok45_high,Grok 4.5,high,8,4,163.09,8.462,8.255,7.449,0.207,5.090,0.636,20.386,3.844,yes,normalized repaired compatibility protocol; cost reconstructed from recorded token classes at xAI public list prices; slowdown recomputed against the original paper expert-TC references
+deepseek_v4_pro_high,DeepSeek V4 Pro,high,5,7,324.22412,67.18005,66.51614,65.861504,0.66391,1.101116,0.220223,64.844824,2.587924,yes,https://github.com/sxzgroup/ORBIT-Q/pull/27; published campaign outcome; runtime uses original paper expert references
+grok46_high,Grok 4.6,high,9,3,199.165271,29.767008,29.277393,27.575296,0.489615,20.129532,2.236615,22.129475,2.190531,yes,https://github.com/sxzgroup/ORBIT-Q/pull/28; published campaign outcome; runtime uses original paper expert references
+astra_high,GPT-6 Astra,high,12,0,44.71885,6.834954,6.768398,6.338304,0.066556,13.967044,1.16392,3.726571,0.974148,yes,https://github.com/sxzgroup/ORBIT-Q/pull/29; automatic verifier outcome; runtime uses original paper expert references
diff --git a/paper/source_data/expert_optimization.csv b/paper/source_data/expert_optimization.csv
new file mode 100644
index 0000000..3531e2c
--- /dev/null
+++ b/paper/source_data/expert_optimization.csv
@@ -0,0 +1,13 @@
+task,baseline_sec,optimized_sec,speedup,dominant_factor,dominant_factor_speedup,dominant_factor_context,source_branch,measurement_note
+01,60.651144,6.359807,9.636410,batched closed-form gate construction,3.575,end-to-end factor screen,codex/task-01-final-expert-optimization,six matched local pairs
+02,4.495463,4.031039,1.115649,batched exact purity,1.074,end-to-end factor screen,codex/task-02-final-expert-optimization,six matched local pairs
+03,4.259896,0.870991,4.893978,TensorCircuit K.vmap local conditional maps,2.063,incremental factor screen,codex/task-03-final-expert-optimization,original workload; exact reduction
+04,14.742286,5.672258,2.602133,probe-network K.vmap,2.165,incremental factor screen,codex/task-04-final-expert-optimization,six matched local pairs
+05,115.708000,60.114000,1.939000,OMECo 4x4 path-search budget,1.420,direct fused-circuit ablation,codex/task-05-tc-native-fused,five cold fresh-process triples; mean paired speedup; median 1.668x
+06,41.425900,27.536600,1.504460,TensorCircuit native jaxode,1.529,single-screen dominant factor,codex/task-06-final-expert-optimization,five matched local pairs
+07,140.076441,3.070839,45.757921,classical-ancilla and duplicate-trajectory reduction,45.758,end-to-end factor screen,codex/task-07-classical-ancilla-design-report,six matched local pairs; exact design loophole
+08,126.675000,123.188000,1.045000,bounded mapped sampling batch,1.045,small paired end-to-end gain,codex/task-08-final-expert-optimization,five matched pairs; valid optimized implementation
+09,33.503727,8.766516,3.821725,preconstructed causal cones plus compiled trajectory,3.8217,bundled end-to-end factor,codex/task-09-final-expert-optimization,six matched local pairs
+10,18.931000,3.869000,4.898000,native hyperedge with fixed contraction path,1.626,largest isolated factor,codex/task-10-final-expert-optimization,five paired cold-specialization runs
+11,168.361539,114.968325,1.464430,layer fusion plus diagonal onsite reduction,81.215,isolated onsite-term factor; not end-to-end,codex/task-11-final-expert-optimization,six matched local pairs
+12,9.082742,2.320613,3.914003,batched fixed-order Pade SU4 construction,3.109,isolated gate-build/gradient kernel,codex/task-12-final-expert-optimization,six matched local pairs
diff --git a/paper/source_data/insights.csv b/paper/source_data/insights.csv
new file mode 100644
index 0000000..596aa02
--- /dev/null
+++ b/paper/source_data/insights.csv
@@ -0,0 +1,7 @@
+task,insight,loophole_or_design_finding,primary_factor,measured_effect
+03,Post-selection after every brickwork layer leaves six independent one-qubit survivors,exact product-state reduction in original task,product reduction,4.894x end to end
+07,Measured ancillas form an analytically sampled classical controller with only two unique branches,exact classical-ancilla/trajectory reduction in published task,branch and trajectory merging,45.758x end to end
+10,The advantage is cold specialization and fixed contraction planning rather than MPO gate arithmetic,hyperedge path search and graph preprocessing dominate; fixed-path hyperedge versus MPO is 1.011x,native hyperedge with fixed path,4.898x end to end
+05,Keep the quantum process TensorCircuit-native and give OMECo enough search budget to avoid bad contraction paths,gate fusion alone is unresolved; OMECo 4x4 is the dominant measured factor,OMECo path-search budget,1.939x mean paired end to end
+09,Explicit causal-cone pruning plus compact TensorCircuit graphs removes irrelevant global work,not a loophole; framework-native local contraction,causal cones and compiled trajectory,3.822x end to end
+12,Batched fixed-order Pade construction shrinks the differentiated gate-build graph,not a loophole; fixed-order static kernel,batched SU4 construction,3.914x end to end
diff --git a/paper/source_data/paper_agent_axis.csv b/paper/source_data/paper_agent_axis.csv
new file mode 100644
index 0000000..1d6e132
--- /dev/null
+++ b/paper/source_data/paper_agent_axis.csv
@@ -0,0 +1,15 @@
+key,label,short_label,series,passes,failures,gm_slowdown,wall_total_sec,wall_passed_sec,wall_failed_sec,total_tokens_m,noncache_prompt_m,cache_prompt_m,output_tokens_m,cost_usd,include_fig1c,include_fig2b,source
+gpt55,GPT-5.5,GPT-5.5,paper,10,2,2.197481,7512.3,5116.4,2395.9,14.241691,,,,17.99,yes,yes,arXiv:2607.03105 Tables S2-S3
+gpt55_checklist,GPT-5.5 + checklist,GPT-5.5*,paper,10,2,2.190668,5343.4,3917.8,1425.6,8.205577,,,,13.29,no,no,arXiv:2607.03105 Tables S2-S4
+opus48,Claude Opus-4.8,Opus-4.8,paper,9,3,2.889799,10382.4,8752.1,1630.3,12.203758,,,,17.52,yes,yes,arXiv:2607.03105 Tables S2-S5
+glm52,GLM-5.2,GLM-5.2,paper,6,6,6.616186,13811.1,6834.2,6976.9,21.075177,,,,12.98,yes,yes,arXiv:2607.03105 Tables S2-S6
+sonnet46,Claude Sonnet-4.6,Sonnet-4.6,paper,7,5,7.245877,13644.2,7715.0,5929.2,21.268795,,,,14.55,yes,yes,arXiv:2607.03105 Tables S2-S7
+sol56_high,GPT-5.6 Sol,Sol,new,10,2,2.541474,11861.715963,9220.290130,2641.425833,26.070809,1.709460,24.198656,0.162693,25.527418,yes,yes,ORBIT-Q PR 5 plus final Task 08 adjudication
+sol56_ultra,GPT-5.6 Sol ultra,Sol ultra,new,10,2,2.072336,10967.757000,7892.985000,3074.772000,33.047608,1.312299,31.478016,0.257293,30.019293,yes,yes,ORBIT-Q PR 6 plus final Task 08 adjudication
+terra56_high,GPT-5.6 Terra,Terra,new,9,3,2.701777,12385.907000,9300.564036,3085.342537,36.803664,1.142037,35.454720,0.206907,12.036862,yes,yes,ORBIT-Q PR 20 plus final Task 08 adjudication
+luna56_high,GPT-5.6 Luna,Luna,new,9,3,4.691620,14362.147000,10744.733972,3617.413183,74.852649,1.769004,72.726528,0.357117,2.236872,yes,yes,ORBIT-Q PR 21 plus final Task 08 adjudication
+deepseek_v4_flash_high,DeepSeek V4 Flash,DeepSeek Flash,new,5,7,3.476293,17424.726000,5869.032092,11555.693612,81.584990,0.663753,80.007936,0.913301,0.572672,yes,yes,ORBIT-Q PR 23
+grok45_high,Grok 4.5,Grok 4.5,new,8,4,3.844088,9785.490000,5394.966612,4390.523575,8.462179,0.805236,7.449472,0.207471,5.090140,yes,yes,ORBIT-Q PR 26; normalized repaired compatibility protocol; cost reconstructed from recorded token classes at xAI public list prices
+deepseek_v4_pro_high,DeepSeek V4 Pro,DeepSeek Pro,new,5,7,2.587924,19453.447205,5792.96097,13660.486235,67.18005,0.654636,65.861504,0.66391,1.101116,yes,yes,https://github.com/sxzgroup/ORBIT-Q/pull/27
+grok46_high,Grok 4.6,Grok 4.6,new,9,3,2.190531,11949.916262,7658.766225,4291.150037,29.767008,1.702097,27.575296,0.489615,20.129532,yes,yes,https://github.com/sxzgroup/ORBIT-Q/pull/28
+astra_high,GPT-6 Astra,Astra,new,12,0,0.974148,2683.130984,2683.130984,0,6.834954,0.430094,6.338304,0.066556,13.967044,yes,yes,https://github.com/sxzgroup/ORBIT-Q/pull/29
diff --git a/paper/source_data/paper_expert_runtimes.csv b/paper/source_data/paper_expert_runtimes.csv
new file mode 100644
index 0000000..e6b6a42
--- /dev/null
+++ b/paper/source_data/paper_expert_runtimes.csv
@@ -0,0 +1,13 @@
+task,expert_tc_runtime_sec,source
+1,27.22,arXiv:2607.03105 public TensorCircuit reference
+2,2.87,arXiv:2607.03105 public TensorCircuit reference
+3,2.46,arXiv:2607.03105 public TensorCircuit reference
+4,11.83,arXiv:2607.03105 public TensorCircuit reference
+5,45.50,arXiv:2607.03105 public TensorCircuit reference
+6,26.83,arXiv:2607.03105 public TensorCircuit reference
+7,63.80,arXiv:2607.03105 public TensorCircuit reference
+8,25.05,arXiv:2607.03105 public TensorCircuit reference
+9,13.74,arXiv:2607.03105 public TensorCircuit reference
+10,12.44,arXiv:2607.03105 public TensorCircuit reference
+11,68.10,arXiv:2607.03105 public TensorCircuit reference
+12,6.12,arXiv:2607.03105 public TensorCircuit reference
diff --git a/paper/source_data/paper_framework_axis.csv b/paper/source_data/paper_framework_axis.csv
new file mode 100644
index 0000000..a7058ba
--- /dev/null
+++ b/paper/source_data/paper_framework_axis.csv
@@ -0,0 +1,5 @@
+key,label,short_label,passes,failures,gm_slowdown,wall_total_sec,wall_passed_sec,wall_failed_sec,total_tokens_m,cost_usd,source
+tensorcircuit,TensorCircuit-NG,TC,10,2,2.197481,7512.3,5116.4,2395.9,14.241691,17.99,arXiv:2607.03105 Tables S2-S3
+pennylane,PennyLane,PL,8,4,,9575.7,6482.5,3093.2,16.579728,23.04,arXiv:2607.03105 Tables S2 and S8
+torchquantum,TorchQuantum,TQ,4,8,,8310.5,1929.0,6381.5,20.298642,24.91,arXiv:2607.03105 Tables S2 and S9
+mindquantum,MindQuantum,MQ,4,8,,9762.4,3028.2,6734.2,22.928678,29.03,arXiv:2607.03105 Tables S2 and S10
diff --git a/paper/source_data/recent_pr_evidence.json b/paper/source_data/recent_pr_evidence.json
new file mode 100644
index 0000000..f45c7b5
--- /dev/null
+++ b/paper/source_data/recent_pr_evidence.json
@@ -0,0 +1,712 @@
+{
+ "schema_version": 1,
+ "as_of": "2026-09-05",
+ "runtime_basis": "Original campaign artifact times divided by the same fixed paper expert references; paired remeasurement is separate",
+ "models": [
+ {
+ "key": "deepseek_v4_pro_high",
+ "label": "DeepSeek V4 Pro",
+ "effort": "high",
+ "pr": 27,
+ "source": {
+ "commit": "78a5a57e49edda1e6f43f2021be508d288028ed7",
+ "path": "results/deepseek-v4-pro-high/summary.json",
+ "sha256": "43399c9a26c732e21ab2cde4944527e5f43ad002fc386438f4bd4d005edc237c",
+ "url": "https://github.com/QingyunQian/ORBIT-Q/blob/78a5a57e49edda1e6f43f2021be508d288028ed7/results/deepseek-v4-pro-high/summary.json"
+ },
+ "outcome_basis": "Published selected campaign outcomes",
+ "cost_basis": "Recorded solver cost",
+ "tasks": [
+ {
+ "task": 1,
+ "raw_reward": 0,
+ "functional": 0,
+ "static": 0,
+ "audit": 0,
+ "runtime_sec": null,
+ "agent_wall_sec": 1801.240353,
+ "input_tokens": 12630069,
+ "cache_tokens": 12539520,
+ "output_tokens": 82650,
+ "cost_usd": 0.156750075,
+ "final_pass": 0
+ },
+ {
+ "task": 2,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 21.66,
+ "agent_wall_sec": 1771.884543,
+ "input_tokens": 6714293,
+ "cache_tokens": 6644352,
+ "output_tokens": 56592,
+ "cost_usd": 0.10374515100000001,
+ "final_pass": 1
+ },
+ {
+ "task": 3,
+ "raw_reward": 0,
+ "functional": 0,
+ "static": 0,
+ "audit": 0,
+ "runtime_sec": null,
+ "agent_wall_sec": 1800.31708,
+ "input_tokens": 7093374,
+ "cache_tokens": 7037568,
+ "output_tokens": 56664,
+ "cost_usd": 0.099084474,
+ "final_pass": 0
+ },
+ {
+ "task": 4,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 20.23,
+ "agent_wall_sec": 858.029259,
+ "input_tokens": 2325275,
+ "cache_tokens": 2296576,
+ "output_tokens": 43822,
+ "cost_usd": 0.058934293000000006,
+ "final_pass": 1
+ },
+ {
+ "task": 5,
+ "raw_reward": 0,
+ "functional": 1,
+ "static": 1,
+ "audit": 0,
+ "runtime_sec": 77.39,
+ "agent_wall_sec": 961.020746,
+ "input_tokens": 3581169,
+ "cache_tokens": 3537536,
+ "output_tokens": 41165,
+ "cost_usd": 0.067617473,
+ "final_pass": 0
+ },
+ {
+ "task": 6,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 36.23,
+ "agent_wall_sec": 802.222496,
+ "input_tokens": 2491315,
+ "cache_tokens": 2456192,
+ "output_tokens": 36022,
+ "cost_usd": 0.055521341,
+ "final_pass": 1
+ },
+ {
+ "task": 7,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 97.41,
+ "agent_wall_sec": 1251.009781,
+ "input_tokens": 4908390,
+ "cache_tokens": 4851072,
+ "output_tokens": 43511,
+ "cost_usd": 0.08037303600000001,
+ "final_pass": 1
+ },
+ {
+ "task": 8,
+ "raw_reward": 0,
+ "functional": 0,
+ "static": 1,
+ "audit": 0,
+ "runtime_sec": 18.99,
+ "agent_wall_sec": 1802.81925,
+ "input_tokens": 6185697,
+ "cache_tokens": 6124544,
+ "output_tokens": 98275,
+ "cost_usd": 0.134302277,
+ "final_pass": 0
+ },
+ {
+ "task": 9,
+ "raw_reward": 0,
+ "functional": 1,
+ "static": 1,
+ "audit": 0,
+ "runtime_sec": 98.91,
+ "agent_wall_sec": 3693.518026,
+ "input_tokens": 1235634,
+ "cache_tokens": 1218176,
+ "output_tokens": 28870,
+ "cost_usd": 0.037127018,
+ "final_pass": 0
+ },
+ {
+ "task": 10,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 54.27,
+ "agent_wall_sec": 1109.814891,
+ "input_tokens": 4183354,
+ "cache_tokens": 4121856,
+ "output_tokens": 46730,
+ "cost_usd": 0.082348458,
+ "final_pass": 1
+ },
+ {
+ "task": 11,
+ "raw_reward": 0,
+ "functional": 0,
+ "static": 0,
+ "audit": 0,
+ "runtime_sec": null,
+ "agent_wall_sec": 1800.354635,
+ "input_tokens": 7224964,
+ "cache_tokens": 7154048,
+ "output_tokens": 57859,
+ "cost_usd": 0.107119214,
+ "final_pass": 0
+ },
+ {
+ "task": 12,
+ "raw_reward": 0,
+ "functional": 0,
+ "static": 0,
+ "audit": 0,
+ "runtime_sec": null,
+ "agent_wall_sec": 1801.216145,
+ "input_tokens": 7942606,
+ "cache_tokens": 7880064,
+ "output_tokens": 71750,
+ "cost_usd": 0.118193502,
+ "final_pass": 0
+ }
+ ]
+ },
+ {
+ "key": "grok46_high",
+ "label": "Grok 4.6",
+ "effort": "high",
+ "pr": 28,
+ "source": {
+ "commit": "09476b847cc2ff53603002dbf0bf7f9c34a148cd",
+ "path": "results/grok-4.6-high/summary.json",
+ "sha256": "6d513fe8e81d0711277d97f359209d63311d9b5fcf2b095a3118672a2e817f55",
+ "url": "https://github.com/QingyunQian/ORBIT-Q/blob/09476b847cc2ff53603002dbf0bf7f9c34a148cd/results/grok-4.6-high/summary.json"
+ },
+ "outcome_basis": "Published selected campaign outcomes",
+ "cost_basis": "Archived campaign list-price cost",
+ "tasks": [
+ {
+ "task": 1,
+ "raw_reward": 0,
+ "functional": 0,
+ "static": 0,
+ "audit": 0,
+ "runtime_sec": null,
+ "agent_wall_sec": 1800.584437,
+ "input_tokens": 6584277,
+ "cache_tokens": 6364160,
+ "output_tokens": 50483,
+ "cost_usd": null,
+ "final_pass": 0
+ },
+ {
+ "task": 2,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 12.36,
+ "agent_wall_sec": 941.422089,
+ "input_tokens": 1536982,
+ "cache_tokens": 1438336,
+ "output_tokens": 29495,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 3,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 5.91,
+ "agent_wall_sec": 725.766953,
+ "input_tokens": 1418857,
+ "cache_tokens": 1288576,
+ "output_tokens": 31641,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 4,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 17.61,
+ "agent_wall_sec": 1268.647419,
+ "input_tokens": 2605577,
+ "cache_tokens": 2451328,
+ "output_tokens": 51488,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 5,
+ "raw_reward": 0,
+ "functional": 1,
+ "static": 1,
+ "audit": 0,
+ "runtime_sec": 72.2,
+ "agent_wall_sec": 960.122457,
+ "input_tokens": 2772314,
+ "cache_tokens": 2659072,
+ "output_tokens": 41979,
+ "cost_usd": null,
+ "final_pass": 0
+ },
+ {
+ "task": 6,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 24.95,
+ "agent_wall_sec": 815.971732,
+ "input_tokens": 1302858,
+ "cache_tokens": 1230336,
+ "output_tokens": 33499,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 7,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 36.03,
+ "agent_wall_sec": 1181.460512,
+ "input_tokens": 3059611,
+ "cache_tokens": 2780544,
+ "output_tokens": 51754,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 8,
+ "raw_reward": 0,
+ "functional": 1,
+ "static": 1,
+ "audit": 0,
+ "runtime_sec": 47.47,
+ "agent_wall_sec": 1530.443143,
+ "input_tokens": 3860732,
+ "cache_tokens": 3677568,
+ "output_tokens": 76974,
+ "cost_usd": null,
+ "final_pass": 0
+ },
+ {
+ "task": 9,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 58.45,
+ "agent_wall_sec": 597.030343,
+ "input_tokens": 1167630,
+ "cache_tokens": 1058688,
+ "output_tokens": 27174,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 10,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 65.01,
+ "agent_wall_sec": 555.462494,
+ "input_tokens": 1037294,
+ "cache_tokens": 969344,
+ "output_tokens": 21748,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 11,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 180.41,
+ "agent_wall_sec": 1075.856693,
+ "input_tokens": 2714550,
+ "cache_tokens": 2612736,
+ "output_tokens": 47074,
+ "cost_usd": null,
+ "final_pass": 1
+ },
+ {
+ "task": 12,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 14.92,
+ "agent_wall_sec": 497.14799,
+ "input_tokens": 1216711,
+ "cache_tokens": 1044608,
+ "output_tokens": 26306,
+ "cost_usd": null,
+ "final_pass": 1
+ }
+ ]
+ },
+ {
+ "key": "astra_high",
+ "label": "GPT-6 Astra",
+ "effort": "high",
+ "pr": 29,
+ "source": {
+ "commit": "b28f53770c606d9c3c499c5ea10613eac197c1c2",
+ "path": "results/gpt6astra-high/comparison.json",
+ "sha256": "7efb1aa710fe6fdae2bdacf4cf21db8d71f0c0148c77f2e8fc89b665be212aea",
+ "url": "https://github.com/QingyunQian/ORBIT-Q/blob/b28f53770c606d9c3c499c5ea10613eac197c1c2/results/gpt6astra-high/comparison.json"
+ },
+ "outcome_basis": "Automatic verifier outcome; no blanket Task 08 failure override",
+ "cost_basis": "Recorded solver cost",
+ "tasks": [
+ {
+ "task": 1,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 13.5,
+ "agent_wall_sec": 482.912465,
+ "input_tokens": 1002736,
+ "cache_tokens": 959232,
+ "output_tokens": 12956,
+ "cost_usd": 2.042072,
+ "final_pass": 1
+ },
+ {
+ "task": 2,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 9.93,
+ "agent_wall_sec": 155.935848,
+ "input_tokens": 324595,
+ "cache_tokens": 298624,
+ "output_tokens": 3837,
+ "cost_usd": 0.750184,
+ "final_pass": 1
+ },
+ {
+ "task": 3,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 6.13,
+ "agent_wall_sec": 149.740855,
+ "input_tokens": 213416,
+ "cache_tokens": 189056,
+ "output_tokens": 4120,
+ "cost_usd": 0.6386560000000001,
+ "final_pass": 1
+ },
+ {
+ "task": 4,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 2.66,
+ "agent_wall_sec": 143.797886,
+ "input_tokens": 342849,
+ "cache_tokens": 317312,
+ "output_tokens": 3826,
+ "cost_usd": 0.763982,
+ "final_pass": 1
+ },
+ {
+ "task": 5,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 9.62,
+ "agent_wall_sec": 175.179708,
+ "input_tokens": 482861,
+ "cache_tokens": 449920,
+ "output_tokens": 4424,
+ "cost_usd": 1.0005300000000001,
+ "final_pass": 1
+ },
+ {
+ "task": 6,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 108.47,
+ "agent_wall_sec": 221.083834,
+ "input_tokens": 692897,
+ "cache_tokens": 648320,
+ "output_tokens": 4174,
+ "cost_usd": 1.30279,
+ "final_pass": 1
+ },
+ {
+ "task": 7,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 22.93,
+ "agent_wall_sec": 347.909045,
+ "input_tokens": 1118515,
+ "cache_tokens": 1065728,
+ "output_tokens": 8682,
+ "cost_usd": 2.027698,
+ "final_pass": 1
+ },
+ {
+ "task": 8,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 16.36,
+ "agent_wall_sec": 255.345866,
+ "input_tokens": 670097,
+ "cache_tokens": 627072,
+ "output_tokens": 6625,
+ "cost_usd": 1.3885720000000001,
+ "final_pass": 1
+ },
+ {
+ "task": 9,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 8.77,
+ "agent_wall_sec": 211.388948,
+ "input_tokens": 579602,
+ "cache_tokens": 548736,
+ "output_tokens": 5188,
+ "cost_usd": 1.1167960000000001,
+ "final_pass": 1
+ },
+ {
+ "task": 10,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 69.88,
+ "agent_wall_sec": 175.292938,
+ "input_tokens": 522941,
+ "cache_tokens": 492032,
+ "output_tokens": 3897,
+ "cost_usd": 0.9959720000000001,
+ "final_pass": 1
+ },
+ {
+ "task": 11,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 102.1,
+ "agent_wall_sec": 235.330635,
+ "input_tokens": 567128,
+ "cache_tokens": 526336,
+ "output_tokens": 5442,
+ "cost_usd": 1.206356,
+ "final_pass": 1
+ },
+ {
+ "task": 12,
+ "raw_reward": 1,
+ "functional": 1,
+ "static": 1,
+ "audit": 1,
+ "runtime_sec": 4.31,
+ "agent_wall_sec": 129.212956,
+ "input_tokens": 250761,
+ "cache_tokens": 215936,
+ "output_tokens": 3385,
+ "cost_usd": 0.733436,
+ "final_pass": 1
+ }
+ ]
+ }
+ ],
+ "paired_astra": {
+ "source": "https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804",
+ "expert_commit": "2187118fecc0e275efc0763c676baa8e2df57108",
+ "astra_commit": "b28f53770c606d9c3c499c5ea10613eac197c1c2",
+ "image": "sha256:ce64a50ab90eaee930d981b7397903bc4baca54f32b8c88f5fa6ad2c40b9f5d7",
+ "cpus": 6,
+ "memory_bytes": 10737418240,
+ "processor": "Apple M2",
+ "summary": {
+ "paired_functional_passes": 11,
+ "expert_functional_passes": 11,
+ "astra_functional_passes": 12,
+ "geometric_mean_ratio": 0.5649749608970326,
+ "astra_faster_tasks": [
+ 1,
+ 4,
+ 5,
+ 7,
+ 9,
+ 11,
+ 12
+ ],
+ "expert_sum_sec": 590.11,
+ "astra_sum_sec": 424.49999999999994,
+ "ratio_of_sums": 0.7193574079408923
+ },
+ "tasks": [
+ {
+ "task": 1,
+ "expert_sec": 58.59,
+ "astra_sec": 13.68,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.2334869431643625
+ },
+ {
+ "task": 2,
+ "expert_sec": 4.15,
+ "astra_sec": 9.84,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 2.3710843373493975
+ },
+ {
+ "task": 3,
+ "expert_sec": 3.82,
+ "astra_sec": 5.1,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 1.3350785340314135
+ },
+ {
+ "task": 4,
+ "expert_sec": 13.13,
+ "astra_sec": 2.33,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.17745620715917745
+ },
+ {
+ "task": 5,
+ "expert_sec": 91.87,
+ "astra_sec": 5.26,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.05725481658865788
+ },
+ {
+ "task": 6,
+ "expert_sec": 44.26,
+ "astra_sec": 104.66,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 2.3646633529145955
+ },
+ {
+ "task": 7,
+ "expert_sec": 147.46,
+ "astra_sec": 21.22,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.1439034314390343
+ },
+ {
+ "task": 8,
+ "expert_sec": null,
+ "astra_sec": 16.31,
+ "expert_functional_pass": false,
+ "astra_functional_pass": true,
+ "expert_status": "OOM",
+ "astra_status": "PASS",
+ "astra_over_expert": null
+ },
+ {
+ "task": 9,
+ "expert_sec": 23.64,
+ "astra_sec": 8.63,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.3650592216582065
+ },
+ {
+ "task": 10,
+ "expert_sec": 21.51,
+ "astra_sec": 105.57,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 4.907949790794978
+ },
+ {
+ "task": 11,
+ "expert_sec": 171.11,
+ "astra_sec": 143.04,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.8359534802174039
+ },
+ {
+ "task": 12,
+ "expert_sec": 10.57,
+ "astra_sec": 5.17,
+ "expert_functional_pass": true,
+ "astra_functional_pass": true,
+ "expert_status": "PASS",
+ "astra_status": "PASS",
+ "astra_over_expert": 0.489120151371807
+ }
+ ],
+ "restoration": {
+ "[REDACTED]_p01": "true_p01",
+ "[REDACTED]_p10": "true_p10"
+ }
+ }
+}
diff --git a/paper/source_data/task_outcomes.csv b/paper/source_data/task_outcomes.csv
new file mode 100644
index 0000000..5971695
--- /dev/null
+++ b/paper/source_data/task_outcomes.csv
@@ -0,0 +1,109 @@
+model,task,raw_reward,final_pass,functional,static,audit,runtime_sec
+gpt56_sol_high,1,0,0,1,1,0,107.51
+gpt56_sol_high,2,1,1,1,1,1,10.65
+gpt56_sol_high,3,1,1,1,1,1,9.84
+gpt56_sol_high,4,1,1,1,1,1,14.14
+gpt56_sol_high,5,1,1,1,1,1,124.92
+gpt56_sol_high,6,1,1,1,1,1,23.39
+gpt56_sol_high,7,1,1,1,1,1,155.52
+gpt56_sol_high,8,0,0,1,1,0,55.48
+gpt56_sol_high,9,1,1,1,1,1,87.31
+gpt56_sol_high,10,1,1,1,1,1,68.68
+gpt56_sol_high,11,1,1,1,1,1,100.41
+gpt56_sol_high,12,1,1,1,1,1,12.85
+gpt56_sol_ultra,1,0,0,1,1,0,127.87
+gpt56_sol_ultra,2,1,1,1,1,1,14.32
+gpt56_sol_ultra,3,1,1,1,1,1,18.45
+gpt56_sol_ultra,4,1,1,1,1,1,4.96
+gpt56_sol_ultra,5,1,1,1,1,1,75.51
+gpt56_sol_ultra,6,1,1,1,1,1,152.16
+gpt56_sol_ultra,7,0,1,1,1,0,84.99
+gpt56_sol_ultra,8,1,0,1,1,1,60.94
+gpt56_sol_ultra,9,1,1,1,1,1,7.18
+gpt56_sol_ultra,10,1,1,1,1,1,71.27
+gpt56_sol_ultra,11,1,1,1,1,1,120.07
+gpt56_sol_ultra,12,1,1,1,1,1,8.61
+gpt56_terra_high,1,0,0,0,1,0,
+gpt56_terra_high,2,1,1,1,1,1,47.00
+gpt56_terra_high,3,1,1,1,1,1,34.29
+gpt56_terra_high,4,1,1,1,1,1,14.62
+gpt56_terra_high,5,1,1,1,1,1,95.16
+gpt56_terra_high,6,1,1,1,1,1,116.18
+gpt56_terra_high,7,1,1,1,1,1,8.91
+gpt56_terra_high,8,0,0,0,0,0,
+gpt56_terra_high,9,1,1,1,1,1,94.28
+gpt56_terra_high,10,0,0,0,1,0,21.53
+gpt56_terra_high,11,1,1,1,1,1,92.81
+gpt56_terra_high,12,1,1,1,1,1,14.07
+gpt56_luna_high,1,0,0,1,1,0,9.84
+gpt56_luna_high,2,1,1,1,1,1,21.30
+gpt56_luna_high,3,1,1,1,1,1,7.69
+gpt56_luna_high,4,0,0,1,1,0,9.45
+gpt56_luna_high,5,1,1,1,1,1,313.09
+gpt56_luna_high,6,1,1,1,1,1,20.47
+gpt56_luna_high,7,1,1,1,1,1,170.18
+gpt56_luna_high,8,1,0,1,1,1,53.18
+gpt56_luna_high,9,1,1,1,1,1,89.42
+gpt56_luna_high,10,1,1,1,1,1,436.64
+gpt56_luna_high,11,1,1,1,1,1,319.79
+gpt56_luna_high,12,1,1,1,1,1,19.34
+deepseek_v4_flash_high,1,0,0,0,0,0,
+deepseek_v4_flash_high,2,0,0,0,0,0,
+deepseek_v4_flash_high,3,1,1,1,1,1,70.04
+deepseek_v4_flash_high,4,1,1,1,1,1,19.46
+deepseek_v4_flash_high,5,1,1,1,1,1,83.79
+deepseek_v4_flash_high,6,0,0,1,1,0,54.79
+deepseek_v4_flash_high,7,0,0,1,1,0,194.15
+deepseek_v4_flash_high,8,0,0,0,0,0,
+deepseek_v4_flash_high,9,0,0,0,0,0,
+deepseek_v4_flash_high,10,1,1,1,1,1,20.50
+deepseek_v4_flash_high,11,0,0,0,0,0,
+deepseek_v4_flash_high,12,1,1,1,1,1,21.86
+grok45_high,1,0,0,0,0,0,
+grok45_high,2,1,1,1,1,1,62.29
+grok45_high,3,1,1,1,1,1,13.01
+grok45_high,4,0,0,0,0,0,
+grok45_high,5,1,1,1,1,1,94.38
+grok45_high,6,1,1,1,1,1,38.08
+grok45_high,7,1,1,1,1,1,121.61
+grok45_high,8,0,0,0,0,0,
+grok45_high,9,1,1,1,1,1,77.90
+grok45_high,10,1,1,1,1,1,89.31
+grok45_high,11,0,0,0,0,0,
+grok45_high,12,1,1,1,1,1,11.13
+deepseek_v4_pro_high,1,0,0,0,0,0,
+deepseek_v4_pro_high,2,1,1,1,1,1,21.66
+deepseek_v4_pro_high,3,0,0,0,0,0,
+deepseek_v4_pro_high,4,1,1,1,1,1,20.23
+deepseek_v4_pro_high,5,0,0,1,1,0,77.39
+deepseek_v4_pro_high,6,1,1,1,1,1,36.23
+deepseek_v4_pro_high,7,1,1,1,1,1,97.41
+deepseek_v4_pro_high,8,0,0,0,1,0,18.99
+deepseek_v4_pro_high,9,0,0,1,1,0,98.91
+deepseek_v4_pro_high,10,1,1,1,1,1,54.27
+deepseek_v4_pro_high,11,0,0,0,0,0,
+deepseek_v4_pro_high,12,0,0,0,0,0,
+grok46_high,1,0,0,0,0,0,
+grok46_high,2,1,1,1,1,1,12.36
+grok46_high,3,1,1,1,1,1,5.91
+grok46_high,4,1,1,1,1,1,17.61
+grok46_high,5,0,0,1,1,0,72.2
+grok46_high,6,1,1,1,1,1,24.95
+grok46_high,7,1,1,1,1,1,36.03
+grok46_high,8,0,0,1,1,0,47.47
+grok46_high,9,1,1,1,1,1,58.45
+grok46_high,10,1,1,1,1,1,65.01
+grok46_high,11,1,1,1,1,1,180.41
+grok46_high,12,1,1,1,1,1,14.92
+astra_high,1,1,1,1,1,1,13.5
+astra_high,2,1,1,1,1,1,9.93
+astra_high,3,1,1,1,1,1,6.13
+astra_high,4,1,1,1,1,1,2.66
+astra_high,5,1,1,1,1,1,9.62
+astra_high,6,1,1,1,1,1,108.47
+astra_high,7,1,1,1,1,1,22.93
+astra_high,8,1,1,1,1,1,16.36
+astra_high,9,1,1,1,1,1,8.77
+astra_high,10,1,1,1,1,1,69.88
+astra_high,11,1,1,1,1,1,102.1
+astra_high,12,1,1,1,1,1,4.31
diff --git a/paper/test_updated_results.py b/paper/test_updated_results.py
new file mode 100644
index 0000000..dd1bab3
--- /dev/null
+++ b/paper/test_updated_results.py
@@ -0,0 +1,72 @@
+"""Regression checks for the archived paper-data contract."""
+import unittest
+
+from verify_updated_results import load, validate
+
+
+class PaperDataTests(unittest.TestCase):
+ def setUp(self):
+ self.tables = load()
+
+ def test_valid_archive(self):
+ validate(self.tables)
+
+ def test_missing_task(self):
+ self.tables["task_outcomes"].pop()
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_changed_astra_outcome(self):
+ self.tables["task_outcomes"][-5]["final_pass"] = "0"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_wrong_cost(self):
+ self.tables["paper_agent_axis"][-2]["cost_usd"] = "0"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_mixed_runtime_baseline(self):
+ self.tables["paper_agent_axis"][-1]["gm_slowdown"] = "0.564975"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_oom_filled_with_reference(self):
+ self.tables["astra_expert_paired"][7]["expert_sec"] = "25.05"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_inverted_speedup(self):
+ row = self.tables["astra_expert_paired"][0]
+ row["expert_over_astra"] = row["astra_over_expert"]
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_wrong_replacement_value(self):
+ self.tables["astra_expert_runtime"][7]["expert_reference_sec"] = "25.05"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_estimate_labeled_measured(self):
+ self.tables["astra_expert_runtime"][7]["expert_reference_basis"] = "measured"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_unrelated_pair_changed(self):
+ self.tables["astra_expert_runtime"][0]["expert_reference_sec"] = "27.22"
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_old_aggregate_retained(self):
+ self.tables["astra_adjustment"]["summary"]["geometric_mean_ratio"] = 0.564975
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+ def test_estimate_counted_as_measured_pair(self):
+ self.tables["astra_adjustment"]["summary"]["measured_pair_count"] = 12
+ with self.assertRaises(ValueError):
+ validate(self.tables)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/paper/updated_figures/astra_expert_paired_runtime.pdf b/paper/updated_figures/astra_expert_paired_runtime.pdf
new file mode 100644
index 0000000..ec1b147
Binary files /dev/null and b/paper/updated_figures/astra_expert_paired_runtime.pdf differ
diff --git a/paper/updated_figures/astra_expert_paired_runtime.png b/paper/updated_figures/astra_expert_paired_runtime.png
new file mode 100644
index 0000000..5935ef7
Binary files /dev/null and b/paper/updated_figures/astra_expert_paired_runtime.png differ
diff --git a/paper/updated_figures/expert_optimization_updated_overview.pdf b/paper/updated_figures/expert_optimization_updated_overview.pdf
new file mode 100644
index 0000000..7374495
Binary files /dev/null and b/paper/updated_figures/expert_optimization_updated_overview.pdf differ
diff --git a/paper/updated_figures/expert_optimization_updated_overview.png b/paper/updated_figures/expert_optimization_updated_overview.png
new file mode 100644
index 0000000..33d028a
Binary files /dev/null and b/paper/updated_figures/expert_optimization_updated_overview.png differ
diff --git a/paper/updated_figures/fig1c_updated_agent_framework_matrix.pdf b/paper/updated_figures/fig1c_updated_agent_framework_matrix.pdf
new file mode 100644
index 0000000..e313c23
Binary files /dev/null and b/paper/updated_figures/fig1c_updated_agent_framework_matrix.pdf differ
diff --git a/paper/updated_figures/fig1c_updated_agent_framework_matrix.png b/paper/updated_figures/fig1c_updated_agent_framework_matrix.png
new file mode 100644
index 0000000..c67f8ca
Binary files /dev/null and b/paper/updated_figures/fig1c_updated_agent_framework_matrix.png differ
diff --git a/paper/updated_figures/fig2b_updated_agent_axis.pdf b/paper/updated_figures/fig2b_updated_agent_axis.pdf
new file mode 100644
index 0000000..5036f49
Binary files /dev/null and b/paper/updated_figures/fig2b_updated_agent_axis.pdf differ
diff --git a/paper/updated_figures/fig2b_updated_agent_axis.png b/paper/updated_figures/fig2b_updated_agent_axis.png
new file mode 100644
index 0000000..a6f23e8
Binary files /dev/null and b/paper/updated_figures/fig2b_updated_agent_axis.png differ
diff --git a/paper/updated_figures/fig4_updated_agent_framework_resources.pdf b/paper/updated_figures/fig4_updated_agent_framework_resources.pdf
new file mode 100644
index 0000000..5787b15
Binary files /dev/null and b/paper/updated_figures/fig4_updated_agent_framework_resources.pdf differ
diff --git a/paper/updated_figures/fig4_updated_agent_framework_resources.png b/paper/updated_figures/fig4_updated_agent_framework_resources.png
new file mode 100644
index 0000000..c1b1921
Binary files /dev/null and b/paper/updated_figures/fig4_updated_agent_framework_resources.png differ
diff --git a/paper/verify_updated_results.py b/paper/verify_updated_results.py
new file mode 100644
index 0000000..48b80b8
--- /dev/null
+++ b/paper/verify_updated_results.py
@@ -0,0 +1,198 @@
+#!/usr/bin/env python3
+"""Validate paper tables without plotting dependencies or network access."""
+import argparse
+import csv
+import hashlib
+import json
+import math
+from pathlib import Path
+import subprocess
+
+DATA = Path(__file__).resolve().parent / "source_data"
+EXPECTED = dict(zip([
+ "gpt56_sol_high", "gpt56_sol_ultra", "gpt56_terra_high", "gpt56_luna_high",
+ "deepseek_v4_flash_high", "grok45_high", "deepseek_v4_pro_high",
+ "grok46_high", "astra_high"], [10, 10, 9, 9, 5, 8, 5, 9, 12]))
+AGENT_KEYS = dict(zip(list(EXPECTED)[:4], [
+ "sol56_high", "sol56_ultra", "terra56_high", "luna56_high"]))
+
+
+def close(actual, expected, label, tolerance=5.1e-7):
+ if not math.isclose(float(actual), expected, rel_tol=0, abs_tol=tolerance):
+ raise ValueError(f"{label}: {actual} != {expected}")
+
+
+def require(condition, label):
+ if not condition:
+ raise ValueError(label)
+
+
+def load(data=DATA):
+ result = {}
+ for name in ("paper_agent_axis", "benchmark_models", "task_outcomes",
+ "paper_expert_runtimes", "astra_expert_paired", "astra_expert_runtime"):
+ with (data / f"{name}.csv").open(newline="") as handle:
+ result[name] = list(csv.DictReader(handle))
+ result["evidence"] = json.loads((data / "recent_pr_evidence.json").read_text())
+ result["astra_adjustment"] = json.loads((data / "astra_expert_adjustment.json").read_text())
+ require(hashlib.sha256((data / "astra_expert_paired.csv").read_bytes()).hexdigest()
+ == result["astra_adjustment"]["measured_source_sha256"], "Raw paired source hash")
+ return result
+
+
+def validate(tables, check_git_sources=False):
+ agent_rows = tables["paper_agent_axis"]
+ agents = {r["key"]: r for r in agent_rows}
+ models = {r["model"]: r for r in tables["benchmark_models"]}
+ tasks = tables["task_outcomes"]
+ require(len(agent_rows) == len(agents) == 14, "14 unique agent rows required")
+ require(len(tables["benchmark_models"]) == len(models) == 9, "9 models required")
+ require(set(models) == set(EXPECTED), "Unexpected model keys")
+ require(len(tasks) == len({(r["model"], r["task"]) for r in tasks}) == 108,
+ "108 unique task rows required")
+ refs = {int(r["task"]): float(r["expert_tc_runtime_sec"])
+ for r in tables["paper_expert_runtimes"]}
+ require(set(refs) == set(range(1, 13)), "12 expert references required")
+ for model, passes in EXPECTED.items():
+ rows = [r for r in tasks if r["model"] == model]
+ require({int(r["task"]) for r in rows} == set(range(1, 13)), model)
+ require(all(r["final_pass"] in ("0", "1") for r in rows), model)
+ close(sum(int(r["final_pass"]) for r in rows), passes, model)
+ axis = agents[AGENT_KEYS.get(model, model)]
+ for aggregate in (models[model], axis):
+ close(aggregate["passes"], passes, model)
+ close(aggregate["failures"], 12-passes, model)
+ ratios = [float(r["runtime_sec"])/refs[int(r["task"])]
+ for r in rows if r["final_pass"] == "1"]
+ require(all(v > 0 for v in ratios), f"{model}: positive passed runtimes")
+ gm = math.exp(sum(map(math.log, ratios))/len(ratios))
+ close(axis["gm_slowdown"], gm, model)
+ close(models[model]["gm_slowdown"], gm, model, tolerance=0.00051)
+ close(axis["wall_total_sec"], float(axis["wall_passed_sec"])
+ + float(axis["wall_failed_sec"]), model, tolerance=0.001)
+ task08 = next(r for r in rows if int(r["task"]) == 8)
+ require(task08["final_pass"] == ("1" if model == "astra_high" else "0"),
+ f"{model}: preserve Task 08 source outcome")
+
+ evidence = tables["evidence"]
+ require({m["key"] for m in evidence["models"]} == set(list(EXPECTED)[-3:]),
+ "Expected evidence for three recent campaigns")
+ for model in evidence["models"]:
+ key, rows = model["key"], model["tasks"]
+ require(len(rows) == 12, f"{key}: evidence task count")
+ source = model["source"]
+ if check_git_sources:
+ raw = subprocess.check_output(["git", "show", f"{source['commit']}:{source['path']}"],
+ cwd=DATA.parent.parent)
+ require(hashlib.sha256(raw).hexdigest() == source["sha256"],
+ f"{key}: source SHA-256 mismatch")
+ if key == "grok46_high":
+ close(agents[key]["cost_usd"], json.loads(raw)["agent_usage"]["list_price_cost_usd"], key)
+ indexed = {int(r["task"]): r for r in tasks if r["model"] == key}
+ for record in rows:
+ row = indexed[record["task"]]
+ for field in ("raw_reward", "final_pass", "functional", "static", "audit", "runtime_sec"):
+ if record[field] is None:
+ require(row[field] == "", f"{key}: missing {field} must stay missing")
+ else:
+ close(row[field], record[field], f"{key}/{record['task']}/{field}")
+ wall = sum(r["agent_wall_sec"] for r in rows)
+ passed_wall = sum(r["agent_wall_sec"] for r in rows if r["final_pass"])
+ prompt, cache, output = [sum(r[field] for r in rows)
+ for field in ("input_tokens", "cache_tokens", "output_tokens")]
+ # Archived Grok 4.6 accounting uses these recorded per-million rates.
+ cost = ((prompt-cache)*2 + cache*0.5 + output*6)/1e6 if key == "grok46_high" else sum(r["cost_usd"] for r in rows)
+ values = {"wall_total_sec": wall, "wall_passed_sec": passed_wall,
+ "wall_failed_sec": wall-passed_wall, "total_tokens_m": (prompt+output)/1e6,
+ "noncache_prompt_m": (prompt-cache)/1e6, "cache_prompt_m": cache/1e6,
+ "output_tokens_m": output/1e6, "cost_usd": cost}
+ for field, value in values.items():
+ close(agents[key][field], value, f"{key}/{field}")
+ values = {"agent_wall_min": wall/60, "total_tokens_m": (prompt+output)/1e6,
+ "input_tokens_m": prompt/1e6, "cache_tokens_m": cache/1e6,
+ "output_tokens_m": output/1e6, "cost_usd": cost,
+ "cost_per_valid_usd": cost/EXPECTED[key],
+ "solve_time_per_valid_min": wall/60/EXPECTED[key]}
+ for field, value in values.items():
+ close(models[key][field], value, f"{key}/{field}")
+
+ paired = tables["astra_expert_paired"]
+ require(len(paired) == 12 and {int(r["task"]) for r in paired} == set(range(1, 13)),
+ "12 paired records required")
+ by_task = {r["task"]: r for r in evidence["paired_astra"]["tasks"]}
+ valid = []
+ for row in paired:
+ record = by_task[int(row["task"])]
+ for field in ("expert_status", "astra_status"):
+ require(row[field] == record[field], f"Paired status: {field}")
+ close(row["astra_sec"], record["astra_sec"], "Paired Astra runtime")
+ if record["expert_sec"] is None:
+ require(row["included_in_aggregate"] == "0" and
+ all(row[f] == "" for f in ("expert_sec", "astra_over_expert", "expert_over_astra")),
+ "OOM must remain missing and excluded")
+ else:
+ require(row["included_in_aggregate"] == "1", "Completed pair must be included")
+ expert, astra = float(row["expert_sec"]), float(row["astra_sec"])
+ close(expert, record["expert_sec"], "Paired expert runtime")
+ close(row["astra_over_expert"], astra/expert, "Paired ratio")
+ close(row["expert_over_astra"], expert/astra, "Paired speedup")
+ valid.append((expert, astra))
+ summary = evidence["paired_astra"]["summary"]
+ close(len(valid), summary["paired_functional_passes"], "Paired count")
+ close(sum(e for e, a in valid), summary["expert_sum_sec"], "Paired expert total")
+ close(sum(a for e, a in valid), summary["astra_sum_sec"], "Paired Astra total")
+ close(sum(a for e, a in valid)/sum(e for e, a in valid), summary["ratio_of_sums"], "Ratio of sums")
+ gm = math.exp(sum(math.log(a/e) for e, a in valid)/len(valid))
+ close(gm, summary["geometric_mean_ratio"], "Paired GM")
+ close(sum(a < e for e, a in valid), len(summary["astra_faster_tasks"]), "Faster tasks")
+ require(abs(gm-float(agents["astra_high"]["gm_slowdown"])) > 0.4,
+ "Paired and fixed-paper baselines must not be mixed")
+
+ adjusted = tables["astra_expert_runtime"]
+ adjustment = tables["astra_adjustment"]
+ require(len(adjusted) == 12 and {int(r["task"]) for r in adjusted} == set(range(1, 13)),
+ "12 adjusted runtime records required")
+ assumption = adjustment["task08"]
+ close(assumption["expert_reference_sec"], 50, "Task 08 approved reference")
+ close(assumption["reviewer_proposed_sec"],
+ assumption["published_reference_sec"]*assumption["reviewer_scale_factor"],
+ "Reviewer reference scaling")
+ require(assumption["basis"] == "estimated" and assumption["original_expert_status"] == "OOM",
+ "Task 08 estimate must not become a successful measurement")
+ for row in adjusted:
+ task = int(row["task"])
+ record = by_task[task]
+ require(row["expert_reference_basis"] == ("estimated" if task == 8 else "measured"),
+ "Only Task 08 uses an estimated expert reference")
+ close(row["astra_sec"], record["astra_sec"], "Adjusted Astra measurement")
+ expert = 50 if task == 8 else record["expert_sec"]
+ astra = record["astra_sec"]
+ close(row["expert_reference_sec"], expert, "Adjusted expert reference")
+ close(row["astra_over_expert"], astra/expert, "Adjusted ratio")
+ close(row["expert_over_astra"], expert/astra, "Adjusted speedup")
+ require(row["source"] == (assumption["source"] if task == 8 else evidence["paired_astra"]["source"]),
+ "Adjusted source provenance")
+ summary = adjustment["summary"]
+ close(summary["task_count"], 12, "Adjusted count")
+ close(summary["measured_pair_count"], 11, "Measured pair count")
+ close(summary["estimated_reference_count"], 1, "Estimated count")
+ expert_total = sum(float(r["expert_reference_sec"]) for r in adjusted)
+ astra_total = sum(float(r["astra_sec"]) for r in adjusted)
+ adjusted_gm = math.exp(sum(math.log(float(r["astra_sec"])/float(r["expert_reference_sec"]))
+ for r in adjusted)/len(adjusted))
+ for field, value in {"expert_sum_sec": expert_total, "astra_sum_sec": astra_total,
+ "ratio_of_sums": astra_total/expert_total,
+ "geometric_mean_ratio": adjusted_gm,
+ "geometric_mean_speedup": 1/adjusted_gm}.items():
+ close(summary[field], value, f"Adjusted {field}")
+ require(summary["astra_faster_tasks"] == [int(r["task"]) for r in adjusted
+ if float(r["astra_sec"]) < float(r["expert_reference_sec"])], "Adjusted faster tasks")
+
+
+if __name__ == "__main__":
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--check-git-sources", action="store_true",
+ help="Also check pinned source objects when available locally")
+ args = parser.parse_args()
+ validate(load(), args.check_git_sources)
+ print("PASS: 14 agent rows, 108 outcomes, three source campaigns; Astra has 11 measured pairs + one estimated reference")