diff --git a/optimized_solutions/challenge-01/factor-ablation.svg b/optimized_solutions/challenge-01/factor-ablation.svg new file mode 100644 index 0000000..7d12a1c --- /dev/null +++ b/optimized_solutions/challenge-01/factor-ablation.svg @@ -0,0 +1,691 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.5 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.5 + + + + + + + + + + + + + 2.0 + + + + + + + + + + + + + 2.5 + + + + + + + + + + + + + 3.0 + + + + + + + + + + + + + 3.5 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + 5 matched pairs + + + a + + + 1.000× + + + 3.575× + + + Batched gate construction → scalar construction + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + + + + + + + + + + + + + + 5 matched pairs + + + b + + + 1.000× + + + 1.562× + + + Layer-local fusion → unfused gates + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + + + + + + + + + + + + + + 5 matched pairs + + + c + + + 1.000× + + + 1.512× + + + Direct TFIM MPO → converted MPO + + + + + + + Task 01 factor ablations — batched gate construction is the largest isolated gain + + + Direct matched-parent ratios; final end-to-end speedup 9.636×. Ratios are not multiplicative. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-02/factor-ablation.svg b/optimized_solutions/challenge-02/factor-ablation.svg new file mode 100644 index 0000000..5434b25 --- /dev/null +++ b/optimized_solutions/challenge-02/factor-ablation.svg @@ -0,0 +1,571 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + 6 matched pairs + + + a + + + 1.000× + + + 1.074× + + + Batched exact purity → generic entropy + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + + + + + 6 matched pairs + + + b + + + 1.000× + + + 1.036× + + + Training scan → Python dispatch + + + + + + + + + + + + + + + + + + Recommended + + + + + + + + + + Rejected variant + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.5 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.5 + + + + + + + + + + + + + 2.0 + + + + + + + + + + + + + 2.5 + + + + + + + + + + + + + + + + + 6 matched pairs + + + c + + + 1.000× + + + 2.336× + + + PyTree parameters → packed leaf + + + + + + + Task 02 factor ablations — two modest gains; packed parameters strongly regress + + + Direct matched-parent ratios; final end-to-end speedup 1.116×. The third panel shows the rejected packed-parameter variant. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-03/factor-ablation.svg b/optimized_solutions/challenge-03/factor-ablation.svg new file mode 100644 index 0000000..dbc801b --- /dev/null +++ b/optimized_solutions/challenge-03/factor-ablation.svg @@ -0,0 +1,601 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.5 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.5 + + + + + + + + + + + + + 2.0 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + 6 matched pairs + + + a + + + 1.000× + + + 2.063× + + + Vectorized local maps → scalar maps + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.00 + + + + + + + + + + + + + 0.25 + + + + + + + + + + + + + 0.50 + + + + + + + + + + + + + 0.75 + + + + + + + + + + + + + 1.00 + + + + + + + + + + + + + 1.25 + + + + + + + + + + + + + 1.50 + + + + + + + + + + + + + 1.75 + + + + + + + + + + + + + + + + + 6 matched pairs + + + b + + + 1.000× + + + 1.660× + + + Product-state reduction → generic network + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + 6 matched pairs + + + c + + + 1.000× + + + 1.149× + + + Training scan → Python dispatch + + + + + + + Task 03 factor ablations — exact product structure enables the largest gains + + + Direct matched-parent ratios; final end-to-end speedup 4.435×. The smaller observable batching factor (1.093×) remains in the table. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-04/factor-ablation.svg b/optimized_solutions/challenge-04/factor-ablation.svg new file mode 100644 index 0000000..5886950 --- /dev/null +++ b/optimized_solutions/challenge-04/factor-ablation.svg @@ -0,0 +1,571 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.5 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.5 + + + + + + + + + + + + + 2.0 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + 6 matched pairs + + + a + + + 1.000× + + + 2.165× + + + Probe K.vmap → four rebuilt networks + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + 6 matched pairs + + + b + + + 1.000× + + + 1.104× + + + Fused RXX–Kraus → separate nodes + + + + + + + + + + + + + + + + + + With factor + + + + + + + + + + Without factor + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + + + + + 6 matched pairs + + + c + + + 1.000× + + + 1.053× + + + Paired Kraus → separate nodes + + + + + + + Task 04 factor ablations — shared vectorized probe networks dominate + + + Direct matched-parent ratios; final end-to-end speedup 2.602×. The two node-fusion factors are secondary. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-05/factor-ablation.svg b/optimized_solutions/challenge-05/factor-ablation.svg new file mode 100644 index 0000000..27b438b --- /dev/null +++ b/optimized_solutions/challenge-05/factor-ablation.svg @@ -0,0 +1,591 @@ + + + + + + + + 2026-07-30T16:30:53.042730 + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Final TC-native + + + + + + + + + + Public expert + + + + + + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + + + + + + + + + + 1.8 + + + + Paired runtime ratio + + + + + + + + + + + + + + + + 3.14× + + + a + + + 1.00× + + + 1.668× + + + 5/5 final wins; 95% t-CI [1.105, 2.772] · confirmed + + + End-to-end replacement + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + OMECo 4×4 + + + + + + + + + + OMECo 1×1 + + + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + + + + + + + + + + 1.8 + + + + Paired runtime ratio + + + + + + + + + + + + + + + + b + + + 1.00× + + + 1.485× + + + 5/5 4×4 wins; 95% t-CI [1.181, 1.659] · confirmed + + + Remove path-search budget + + + + + + + + + + + + + + + + + + + + + + + + + + + Fused graph + + + + + + + + + + Unfused graph + + + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + + + + + + + + + + 1.8 + + + + Paired runtime ratio + + + + + + + + + + + + + + + + c + + + 1.00× + + + 1.175× + + + 4/5 fused wins; 95% t-CI [0.916, 1.480] · unresolved + + + Remove pair fusion + + + + + + + + + + + + + Task 05 — TensorCircuit-native contraction-path ablation + + + Adequate OMECo search removes bad paths; gate fusion alone is not established + + + Bars: median paired ratio; dots: five rotated-order cold-process pairs. All cells passed. Ratios are not multiplied. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-06/factor-ablation.svg b/optimized_solutions/challenge-06/factor-ablation.svg new file mode 100644 index 0000000..5738920 --- /dev/null +++ b/optimized_solutions/challenge-06/factor-ablation.svg @@ -0,0 +1,663 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + With jaxode + 27.75 s + + + + + + + + + + Without + 42.41 s + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + single canonical screen + + + a + + + 1.000× + + + 1.528× + + + Native jaxode → Diffrax + + + + + + + + + + + + + + + + + + Fused + + + + + + + + + + Unfused + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + cold kernel; steady gain 1.004× + + + b + + + 1.000× + + + 1.152× + + + Fused Euler gate → three gates + + + + + + + + + + + + + + + + + + Recommended + + + + + + + + + + Rejected BCOO + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.5 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.5 + + + + + + + + + + + + + 2.0 + + + + + + + + + + + + + 2.5 + + + + + + + + + + + + + 3.0 + + + + + + + + + + + + + 3.5 + + + + + + + + + + + + + + + + + isolated actions: 3.4–3.5× slower + + + c + + + 1.000× + + + 3.4–3.5× + + + Termwise MVP → sparse BCOO + + + + + + + Task 06 factor ablations — TensorCircuit native jaxode supplies the end-to-end gain + + + Panels use the tracked canonical screen or isolated component profile named above; they are explanatory, not multiplicative. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-07/factor-ablation.svg b/optimized_solutions/challenge-07/factor-ablation.svg new file mode 100644 index 0000000..7670022 --- /dev/null +++ b/optimized_solutions/challenge-07/factor-ablation.svg @@ -0,0 +1,558 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + Reduced + 3.07 s + + + + + + + + + + Public expert + 140.08 s + + + + + + + + + + + + + + + + + + 1 + + + + + + + + + + + + + 3 + + + + + + + + + + + + + 10 + + + + + + + + + + + + + 30 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + 6 matched pairs; paired mean + + + a + + + 1.000× + + + 45.758× + + + Exact controller reduction → full 16q × 64 + + + + + + + + + + + + + + + + + + Recommended + + + + + + + + + + Rejected scan + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + + + + + post-reduction screen + + + b + + + 1.000× + + + 1.070× + + + Python loop → training scan + + + + + + + + + + + + + + + + + + Recommended + + + + + + + + + + Rejected fusion + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + post-reduction screen + + + c + + + 1.000× + + + 1.178× + + + Native gates → dense local fusion + + + + + + + Task 07 factor ablations — exact classical reduction removes nearly all work + + + Panel a is the exact challenge-design reduction. Panels b–c test secondary changes after the graph is already reduced. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-08/factor-ablation.svg b/optimized_solutions/challenge-08/factor-ablation.svg new file mode 100644 index 0000000..00dcbfc --- /dev/null +++ b/optimized_solutions/challenge-08/factor-ablation.svg @@ -0,0 +1,607 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + 256-shot + 123.19 s + + + + + + + + + + Public expert + 126.68 s + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + paired mean 1.045×; CI [0.818, 1.273] + + + a + + + 1.000× + + + 1.028× + + + Complete 8192 shots + + + + + + + + + + + + + + + + + + 256 shots + 44.03 s + + + + + + + + + + 512 shots + 50.84 s + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + single full-workload screen + + + b + + + 1.000× + + + 1.155× + + + 256-shot chunks → 512-shot chunks + + + + + + + + + + + + + + + + + + 256 shots + 44.03 s + + + + + + + + + + 128 shots + 55.18 s + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + single full-workload screen + + + c + + + 1.000× + + + 1.253× + + + 256-shot chunks → 128-shot chunks + + + + + + + Task 08 factor ablations — 256-shot chunks minimize the screen; full-run speedup is unconfirmed + + + Panel a: five complete 8192-shot pairs; paired 95% CI includes 1. Panels b–c: one full-workload screen per chunk size. + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-09/factor-ablation.svg b/optimized_solutions/challenge-09/factor-ablation.svg new file mode 100644 index 0000000..677aa87 --- /dev/null +++ b/optimized_solutions/challenge-09/factor-ablation.svg @@ -0,0 +1,639 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + Compact + 8.77 s + + + + + + + + + + Public expert + 33.50 s + + + + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.5 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.5 + + + + + + + + + + + + + 2.0 + + + + + + + + + + + + + 2.5 + + + + + + + + + + + + + 3.0 + + + + + + + + + + + + + 3.5 + + + + + + + + + + + + + 4.0 + + + + Runtime normalized to recommended + + + + + + + + + + + + + + + + 6 matched pairs; paired mean + + + a + + + 1.000× + + + 3.822× + + + Compact causal cones → full graph + + + + + + + + + + + + + + + + + + Enabled + 8.77 s + + + + + + + + + + Disabled + >300 s + + + + + + + + + + + + + + + 1 + + + + + + + + + + + + + 3 + + + + + + + + + + + + + 10 + + + + + + + + + + + + + 30 + + + + + + + + + + + + + + + + + ablation timed out + + + b + + + 1.000× + + + >34.2× + + + Inner light-cone on → disabled + + + + + + + + + + + + + + + + + + Separated + 7.14 s + + + + + + + + + + Combined + 7.72 s + + + + + + + + + + + + + + + 0.0 + + + + + + + + + + + + + 0.2 + + + + + + + + + + + + + 0.4 + + + + + + + + + + + + + 0.6 + + + + + + + + + + + + + 0.8 + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + + + + + 3 matched screening pairs + + + c + + + 1.000× + + + 1.082× + + + Separated cones → combined loss + + + + + + + Task 09 factor ablations — compact causal cones and inner light-cone cancellation are both essential + + + Panel a is the bundled final comparison. Panel b is a lower bound from the 300 s timeout; panel c is a direct 3-pair removal screen. + + + + + + + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-10/factor-ablation.svg b/optimized_solutions/challenge-10/factor-ablation.svg new file mode 100644 index 0000000..520cfd7 --- /dev/null +++ b/optimized_solutions/challenge-10/factor-ablation.svg @@ -0,0 +1,300 @@ + + + + + + + + 2026-07-31T10:48:26.036746 + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 1.0 + + + + + + + + + + + + + 1.2 + + + + + + + + + + + + + 1.4 + + + + + + + + + + + + + 1.6 + + + + + + + + + + + + + 1.8 + + + + Paired speedup (before / after); error bars are 95% t-CIs + + + + + + + Cache path + (native hyperedge) + + + + + + MPO gate + (fixed path) + + + + + + Fuse rotations + (fixed path) + + + + + + Local MPS program + (fused, fixed-path control) + + + + + + + + + + + + + + + + + + + + + + + + 1.626× + + + + + + + + + + + + + + + + + + + 1.011× + + + + + + + + + + + + + + + + 1.210× + + + + + + + + + + + + + + + + 1.158× + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-11/factor-ablation.svg b/optimized_solutions/challenge-11/factor-ablation.svg new file mode 100644 index 0000000..ac7b22a --- /dev/null +++ b/optimized_solutions/challenge-11/factor-ablation.svg @@ -0,0 +1,2527 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/optimized_solutions/challenge-12/factor-ablation.svg b/optimized_solutions/challenge-12/factor-ablation.svg new file mode 100644 index 0000000..8fad390 --- /dev/null +++ b/optimized_solutions/challenge-12/factor-ablation.svg @@ -0,0 +1,2499 @@ + + + + + + + + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/paper/FIGURE_CONTRACT.md b/paper/FIGURE_CONTRACT.md new file mode 100644 index 0000000..e0a9ad4 --- /dev/null +++ b/paper/FIGURE_CONTRACT.md @@ -0,0 +1,39 @@ +# Figure contract + +The paper update follows the visual and metric contracts of +[arXiv:2607.03105](https://arxiv.org/abs/2607.03105). + +1. Updated Fig. 1c preserves the agent-by-framework matrix, places all + configurations in one continuous layout without an old/new separator, and + includes nine added TensorCircuit-NG campaigns through PR #29. +2. Updated Fig. 2b retains all thirteen model points and the previous non-Astra + coordinates. Astra uses eleven paired measurements plus the approved + 50 s estimated Task 08 expert reference. Both axes remain linear + with extra padding left of zero failure and below unit runtime. No footer, + reference-reuse marker, or numeric runtime annotation is added to the figure. +3. Updated Fig. 4 preserves the paper's 2 × 3 wall-time, token, and efficiency + layout. All token bars use the same total-token encoding because the source + tables do not publish components for every configuration. +4. The expert-optimization overview reports paired end-to-end measurements; + exact task reductions are distinguished visually; Task 08 remains a valid + optimized implementation with a small measured point estimate. +5. Fable 5 and DeepSeek max remain excluded. Published outcomes are preserved + per campaign: Task 08 is failed for the eight pre-Astra configurations and + passed for Astra. Astra retains its automatic result; no blanket failure + override or retroactive adjudication is applied. +6. Figure labels omit the repeated `high` qualifier. The accompanying text + states that Sol, Terra, Luna, DeepSeek V4 Flash/Pro, Grok 4.5/4.6 and Astra use high + thinking effort; the separately named Sol ultra configuration uses ultra. +7. Astra uses purple in model-specific encodings. Passed/failed and total-token + bars retain their common colors; the matrix retains its count scale. +8. The Astra paired runtime figure supplies the Astra ratio in Fig. 2b: it uses + same-host remeasurements for eleven tasks and the estimated Task 08 reference. + The raw Task 08 OOM record is preserved separately; the replacement is + not reported as a completed measurement. Its source is given in the data + and accompanying text, not as an added figure footer. + Its right panel reports expert/Astra speedup, not Astra/expert runtime. + +Vector PDF is the publication deliverable; PNG copies are non-publication +previews for inline GitHub/PR display only. +Agent wall time, token use, artifact-runtime ratio, and expert-implementation +speedup remain separate quantities. diff --git a/paper/FIGURE_QA.md b/paper/FIGURE_QA.md new file mode 100644 index 0000000..df0857b --- /dev/null +++ b/paper/FIGURE_QA.md @@ -0,0 +1,57 @@ +# Figure QA record + +Date: 2026-09-05 + +- Regenerated Fig. 1c, Fig. 2b and Fig. 4; added a separate Astra/expert + paired-runtime figure. Each has a vector PDF master and matched 300-dpi PNG. +- Rendered all four changed PDFs with Poppler and visually inspected them. + Adjusted scatter labels to remove overlap; retained direct labels without + colored leader lines. Astra uses purple; the count matrix, passed/failed + bars, and total-token bars retain their shared encodings. +- All five PDFs, including the unchanged expert-optimization overview, have + one page, embedded TrueType fonts, and no raster image objects. SVG/TIFF + are intentionally omitted; PDF is the publication master. +- Validated 14 agent rows and 108 unique task outcomes. Added-campaign pass + counts are **10, 10, 9, 9, 5, 8, 5, 9, 12**. Task 08 source outcomes are + preserved: failed for the first eight campaigns, passed for Astra. +- Cross-checked the 36 new task records and resource totals against pinned + PR #27–29 evidence; verified the three original source-file SHA-256 hashes + against local Git objects. +- Fig. 2b retains all thirteen model points and all previous non-Astra + coordinates. Astra alone uses the updated artifact comparison. Both axes are linear, + with limits (-10, 72) and (-0.6, 9.2) for visual padding, not negative data. +- Inspected the revised PDF after Poppler rendering: Astra is below the + reference anchor; no footer, dagger, numeric Astra runtime annotation, + colored leader lines, or label overlap is present. +- Independently checked the updated ratios and reciprocal speedups: **12** + comparisons, consisting of **11 measured pairs and one estimated reference**. + Task 08 uses **50.00 s / 16.31 s**, giving **3.065604×** expert/Astra speedup. + Astra is faster on **8/12** tasks; geometric mean Astra/expert is + **0.539698×**, with totals **440.81 / 640.11 s** (ratio **0.688647×**). + The raw Task 08 OOM record and original paired-source hash are unchanged. +- Re-rendered both runtime figures after the replacement and inspected the + actual PDF pages. The per-task plot now contains all 24 runtime bars. + Programmatically checked all thirteen scatter coordinates, including + Astra at (0, 0.53969753); every non-Astra coordinate remains unchanged. + The matrix and solver-resource PNG hashes are also unchanged. +- Twelve regression tests pass, including rejection of missing tasks, changed + outcomes, zero Grok cost, mixed runtime baselines, filled OOM times, and + inverted speedups, an incorrect replacement, an estimate labeled as measured, + changes to unrelated pairs, stale aggregates, and inflated measured-pair + counts. `git diff --check` passes. +- The earlier 72 task records and all original numeric agent rows are retained. + The original framework and expert-reference tables, expert optimizations, + insights, and expert-overview PDF/PNG are unchanged from the PR head. +- Fable 5 and DeepSeek max remain excluded. No benchmark was rerun for this + paper-summary refresh, and unrelated worktree changes were not included. + +Reproduce from the repository root: + +```bash +python3 paper/verify_updated_results.py +python3 -m unittest discover -s paper -p test_updated_results.py +python3 paper/make_updated_paper_figures.py +``` + +With the pinned campaign commits available locally, additionally run +`python3 paper/verify_updated_results.py --check-git-sources`. diff --git a/paper/README.md b/paper/README.md new file mode 100644 index 0000000..ad0fb08 --- /dev/null +++ b/paper/README.md @@ -0,0 +1,129 @@ +# ORBIT-Q paper-update figures + +This directory provides paper-ready updates to Fig. 1c, Fig. 2b, and Fig. 4 +of [arXiv:2607.03105](https://arxiv.org/abs/2607.03105), plus one new summary +figure for AI-assisted optimization of the 12 human-expert implementations +and a separate paired Astra/expert runtime comparison. Updated 2026-09-05 +with the complete campaigns in PRs #27, #28 and #29. +The manuscript masters are vector PDF; matched PNG files are included only for +inline GitHub/PR previews. +Fable 5 and DeepSeek max are intentionally excluded. Sol, Terra, Luna, +DeepSeek V4 Flash/Pro, Grok 4.5/4.6, and Astra use high thinking effort; +the explicitly named Sol ultra configuration uses ultra. + +## Campaign results + +| Model | Passed | Solve time (min) | Tokens (M) | Cost (USD) | Runtime / expert (Fig. 2b) | Source | +|---|---:|---:|---:|---:|---:|---| +| GPT-5.6 Sol | 10/12 | 197.70 | 26.071 | 25.527 | 2.541× | [#5](https://github.com/sxzgroup/ORBIT-Q/pull/5) | +| GPT-5.6 Sol ultra | 10/12 | 182.80 | 33.048 | 30.019 | 2.072× | [#6](https://github.com/sxzgroup/ORBIT-Q/pull/6) | +| GPT-5.6 Terra | 9/12 | 206.43 | 36.804 | 12.037 | 2.702× | [#20](https://github.com/sxzgroup/ORBIT-Q/pull/20) | +| GPT-5.6 Luna | 9/12 | 239.37 | 74.853 | 2.237 | 4.692× | [#21](https://github.com/sxzgroup/ORBIT-Q/pull/21) | +| DeepSeek V4 Flash | 5/12 | 290.41 | 81.585 | 0.573 | 3.476× | [#23](https://github.com/sxzgroup/ORBIT-Q/pull/23) | +| Grok 4.5 | 8/12 | 163.09 | 8.462 | 5.090 | 3.844× | [#26](https://github.com/sxzgroup/ORBIT-Q/pull/26) | +| DeepSeek V4 Pro | 5/12 | 324.22 | 67.180 | 1.101 | 2.588× | [#27](https://github.com/sxzgroup/ORBIT-Q/pull/27) | +| Grok 4.6 | 9/12 | 199.17 | 29.767 | 20.130 | 2.191× | [#28](https://github.com/sxzgroup/ORBIT-Q/pull/28) | +| GPT-6 Astra | 12/12 | 44.72 | 6.835 | 13.967 | 0.540× | [#29](https://github.com/sxzgroup/ORBIT-Q/pull/29) | + +Reported outcomes retain each campaign's published audit/adjudication basis. +Astra's 12/12 is the original automatic verifier result, not a newly applied +blanket human adjudication. The maintainer's subsequent source review accepts +its Task 08 reduction ([review](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5548928438)). +Earlier campaigns' Task 08 failures are not retroactively changed. + +## Updated Fig. 1c — agent–framework benchmark matrix + +[PDF](updated_figures/fig1c_updated_agent_framework_matrix.pdf) + +The matrix presents all reported configurations in one continuous layout, +without visually separating earlier and later evaluations. The added +TensorCircuit-NG results are Sol **10/12**, Sol ultra **10/12**, Terra **9/12**, +Luna **9/12**, DeepSeek V4 Flash **5/12**, Grok 4.5 **8/12**, DeepSeek V4 Pro +**5/12**, Grok 4.6 **9/12**, and Astra **12/12**. The count-based matrix palette +is unchanged; model-specific scatter markers and the Astra label use purple. + +## Updated Fig. 2b — failure rate versus artifact runtime + +[PDF](updated_figures/fig2b_updated_agent_axis.pdf) + +All thirteen model points are retained. Astra uses the runtime comparison +below; other points retain their previous fixed-paper-reference values. +The linear axes add space left of zero failure and below unit runtime. + +## Updated Fig. 4 — benchmark resource use + +[PDF](updated_figures/fig4_updated_agent_framework_resources.pdf) + +The original 2 × 3 layout is preserved. Panels (a–c) combine all fourteen agent +configurations; panels (d–f) retain the paper's original framework comparison. +Panels (b) and (e) use one consistent total-token encoding for every +configuration because the paper tables do not publish token components for +every legacy configuration. Bubble area represents total solver cost. All +three added models appear in wall-time, token, and cost panels. Their costs +per valid solution are **$0.220223** (DeepSeek Pro), **$2.236615** (Grok 4.6), +and **$1.163920** (Astra). Cost values retain the source campaigns' accounting. + +## Astra / expert artifact runtime + +[PDF](updated_figures/astra_expert_paired_runtime.pdf) + +The [PR #29 runtime reply](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804) +reports fresh runs on the same Apple M2 host, pinned TensorCircuit image and +6 CPU / 10 GiB limits. Following the [reviewer's suggestion](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776), +Task 08 uses an **estimated 50 s** expert reference (25.05 s × 2, rounded), +compared with Astra's measured 16.31 s. The other eleven pairs are unchanged. +With this replacement, Astra is faster on **8/12** tasks. The geometric mean +Astra/expert ratio is **0.540×** (equivalently **1.85×** speedup), and totals +are **440.81 s / 640.11 s = 0.689×**. This includes 11 measured pairs and +one estimated expert reference, not 12 completed paired measurements. +The plot's right panel uses the reciprocal expert/Astra speedup, so >1 means +Astra is faster. Only the two exported Task 04 field names were restored in +its execution copy; the algorithms and original archive were unchanged. +The [updated table](astra_runtime_update.md) and +[plot data](source_data/astra_expert_runtime.csv) record the replacement. +The original [paired measurements](source_data/astra_expert_paired.csv), +including the missing expert Task 08 runtime, remain unchanged. + +## New figure — human expert + AI co-optimization + +[PDF](updated_figures/expert_optimization_updated_overview.pdf) + +Panel (a) compares end-to-end runtime before and after optimization on a log +scale. Panel (b) reports the paired end-to-end speedup and the dominant insight +retained after ablation. The largest result is the exact Task 07 reduction +(**45.76×**); Task 01 reaches **9.64×**, while Tasks 03, 09, 10, and 12 reach +**3.82–4.90×**. Task 08 is included as a valid optimized implementation with a +small **1.04×** paired point estimate. +This concerns the optimized human-expert implementation and is separate from +the Task 08 benchmark outcomes of the solver campaigns above. +The legal TensorCircuit-native Task 05 result is used (**1.939× mean paired**). +This is the arithmetic mean of five paired public-expert/final-candidate runtime +ratios, not the ratio of the two reported mean runtimes. The candidate won all +five pairs; the median is **1.668×**. One `180.22 s` expert-reference long tail +raises the mean, while the four-pair sensitivity mean is **1.639×**. The direct +ablation attributes **1.420×** to the retained OMECo `4×4` path-search budget; +gate fusion alone remains unresolved. + +Individual factor-removal plots remain available as supplementary evidence: + +| Task | End-to-end result | Upstream PR | +|---:|---:|---:| +| 01 | **9.636×** | [#8](https://github.com/sxzgroup/ORBIT-Q/pull/8) | +| 02 | **1.116×** | [#9](https://github.com/sxzgroup/ORBIT-Q/pull/9) | +| 03 | **4.894×** | [#10](https://github.com/sxzgroup/ORBIT-Q/pull/10) | +| 04 | **2.602×** | [#11](https://github.com/sxzgroup/ORBIT-Q/pull/11) | +| 05 | **1.939× mean paired** | [#19](https://github.com/sxzgroup/ORBIT-Q/pull/19) | +| 06 | **1.504×** | [#13](https://github.com/sxzgroup/ORBIT-Q/pull/13) | +| 07 | **45.758×** | [#7](https://github.com/sxzgroup/ORBIT-Q/pull/7) | +| 08 | **1.045×** | [#18](https://github.com/sxzgroup/ORBIT-Q/pull/18) | +| 09 | **3.822×** | [#14](https://github.com/sxzgroup/ORBIT-Q/pull/14) | +| 10 | **4.898×** | [#15](https://github.com/sxzgroup/ORBIT-Q/pull/15) | +| 11 | **1.464×** | [#16](https://github.com/sxzgroup/ORBIT-Q/pull/16) | +| 12 | **3.914×** | [#17](https://github.com/sxzgroup/ORBIT-Q/pull/17) | + +Source tables, adjudication notes, and the reproducible plotting script are in +[`source_data/`](source_data/) and +[`make_updated_paper_figures.py`](make_updated_paper_figures.py). +Run `python3 paper/verify_updated_results.py` from the repository root to check +the tables without plotting dependencies; see [FIGURE_QA.md](FIGURE_QA.md) +for regression tests and rendering checks. diff --git a/paper/astra_runtime_update.md b/paper/astra_runtime_update.md new file mode 100644 index 0000000..0e764a4 --- /dev/null +++ b/paper/astra_runtime_update.md @@ -0,0 +1,28 @@ +# Astra artifact runtime update + +Following the [reviewer's suggestion](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776), Task 08 uses an estimated expert reference of **50 s** (25.05 s × 2, rounded). Its Astra runtime remains the measured **16.31 s**. The other eleven pairs are unchanged same-host measurements. Original measurement records are preserved separately. + +Across all 12 tasks, the geometric mean Astra/expert runtime ratio is **0.540×**, equivalent to **1.85×** geometric-mean speedup. Astra is faster on **8/12** tasks. Summed times are **440.81 s / 640.11 s = 0.689×**. This comparison contains 11 measured pairs and one estimated expert reference. + +| Task | Expert reference (s) | Astra (s) | Astra / expert | +|---|---:|---:|---:| +| 01 | 58.59 | 13.68 | 0.233× | +| 02 | 4.15 | 9.84 | 2.371× | +| 03 | 3.82 | 5.10 | 1.335× | +| 04 | 13.13 | 2.33 | 0.177× | +| 05 | 91.87 | 5.26 | 0.057× | +| 06 | 44.26 | 104.66 | 2.365× | +| 07 | 147.46 | 21.22 | 0.144× | +| 08 | 50.00 | 16.31 | 0.326× | +| 09 | 23.64 | 8.63 | 0.365× | +| 10 | 21.51 | 105.57 | 4.908× | +| 11 | 171.11 | 143.04 | 0.836× | +| 12 | 10.57 | 5.17 | 0.489× | + +![Astra artifact runtime](updated_figures/astra_expert_paired_runtime.png) + +Panel (b) shows expert/Astra speedup, the reciprocal of the table's runtime ratio. + +![Artifact runtime efficiency](updated_figures/fig2b_updated_agent_axis.png) + +Only Astra's artifact-runtime coordinate changes; all other model points and solver-cost figures are unchanged. diff --git a/paper/make_updated_paper_figures.py b/paper/make_updated_paper_figures.py new file mode 100644 index 0000000..c594450 --- /dev/null +++ b/paper/make_updated_paper_figures.py @@ -0,0 +1,510 @@ +#!/usr/bin/env python3 +"""Generate vector-ready ORBIT-Q paper-update figures. + +The layouts deliberately follow arXiv:2607.03105: Fig. 1c is an +agent-by-framework matrix, Fig. 2b is the TC agent-axis failure/runtime +scatter, and Fig. 4 keeps the original 2 x 3 resource-use structure. +""" + +from __future__ import annotations + +import csv +import math +from pathlib import Path + +import matplotlib as mpl +mpl.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +from matplotlib.colors import LinearSegmentedColormap, Normalize +from matplotlib.patches import Patch, Rectangle + + +ROOT = Path(__file__).resolve().parent +DATA = ROOT / "source_data" +OUT = ROOT / "updated_figures" + +mpl.rcParams.update( + { + "font.family": "sans-serif", + "font.sans-serif": ["Arial", "Helvetica", "DejaVu Sans"], + "font.size": 8.5, + "axes.labelsize": 9, + "axes.titlesize": 10, + "xtick.labelsize": 8, + "ytick.labelsize": 8, + "legend.fontsize": 7.5, + "axes.linewidth": 0.8, + "grid.color": "#D6D6D6", + "grid.linewidth": 0.6, + "grid.alpha": 0.70, + "svg.fonttype": "none", + "pdf.fonttype": 42, + "ps.fonttype": 42, + "savefig.facecolor": "white", + } +) + +COLORS = { + "gpt55": "#0072B2", + "gpt55_checklist": "#56B4E9", + "opus48": "#D55E00", + "glm52": "#A05195", + "sonnet46": "#E69F00", + "sol56_high": "#2563EB", + "sol56_ultra": "#F97316", + "terra56_high": "#009E73", + "luna56_high": "#CC79A7", + "deepseek_v4_flash_high": "#4D4D4D", + "grok45_high": "#111111", + "deepseek_v4_pro_high": "#007F87", + "grok46_high": "#566573", + "astra_high": "#7B2CBF", + "tensorcircuit": "#0072B2", + "pennylane": "#CC79A7", + "torchquantum": "#E69F00", + "mindquantum": "#009E73", +} + +PASS = "#009E73" +FAIL = "#FFFFFF" +CACHE = "#56B4E9" + + +def read_csv(name: str) -> list[dict[str, str]]: + with (DATA / name).open(newline="", encoding="utf-8") as f: + return list(csv.DictReader(f)) + + +def num(row: dict[str, str], key: str) -> float: + value = row.get(key, "") + return float(value) if value not in (None, "") else np.nan + + +def clean_axes(ax: mpl.axes.Axes, *, grid_axis: str = "both") -> None: + ax.grid(True, axis=grid_axis, zorder=0) + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.tick_params(direction="out", length=3, width=0.8) + + +def panel_label(ax: mpl.axes.Axes, label: str) -> None: + ax.text( + -0.11, + 1.04, + label, + transform=ax.transAxes, + fontsize=12, + fontweight="bold", + va="bottom", + ha="left", + ) + + +def save_all(fig: mpl.figure.Figure, stem: str) -> None: + OUT.mkdir(parents=True, exist_ok=True) + fig.savefig(OUT / f"{stem}.pdf", bbox_inches="tight") + fig.savefig(OUT / f"{stem}.png", dpi=300, bbox_inches="tight") + plt.close(fig) + + +def validate_new_slowdowns(agent_rows: list[dict[str, str]]) -> None: + """Check archived fixed-paper ratios; these no longer drive Fig. 2b.""" + refs = {int(r["task"]): num(r, "expert_tc_runtime_sec") for r in read_csv("paper_expert_runtimes.csv")} + outcomes = read_csv("task_outcomes.csv") + model_to_agent = { + "gpt56_sol_high": "sol56_high", + "gpt56_sol_ultra": "sol56_ultra", + "gpt56_terra_high": "terra56_high", + "gpt56_luna_high": "luna56_high", + "deepseek_v4_flash_high": "deepseek_v4_flash_high", + "grok45_high": "grok45_high", + "deepseek_v4_pro_high": "deepseek_v4_pro_high", + "grok46_high": "grok46_high", + "astra_high": "astra_high", + } + expected = {r["key"]: num(r, "gm_slowdown") for r in agent_rows if r["series"] == "new"} + for model, agent_key in model_to_agent.items(): + ratios = [ + num(r, "runtime_sec") / refs[int(r["task"])] + for r in outcomes + if r["model"] == model and r["final_pass"] == "1" + ] + recomputed = math.exp(sum(math.log(v) for v in ratios) / len(ratios)) + if not math.isclose(recomputed, expected[agent_key], rel_tol=0.0, abs_tol=5e-7): + raise ValueError(f"{model}: stored GM {expected[agent_key]} != recomputed {recomputed}") + + +def make_fig1c(agent_rows: list[dict[str, str]], framework_rows: list[dict[str, str]]) -> None: + agents = [r for r in agent_rows if r["include_fig1c"] == "yes"] + order = ["gpt55", "opus48", "glm52", "sonnet46", "sol56_high", "sol56_ultra", "terra56_high", "luna56_high", "deepseek_v4_flash_high", "grok45_high", "deepseek_v4_pro_high", "grok46_high", "astra_high"] + agents = sorted(agents, key=lambda r: order.index(r["key"])) + frameworks = {r["key"]: r for r in framework_rows} + + values = np.full((4, len(agents)), np.nan) + values[0, :] = [int(r["passes"]) for r in agents] + values[1, 0] = int(frameworks["pennylane"]["passes"]) + values[2, 0] = int(frameworks["torchquantum"]["passes"]) + values[3, 0] = int(frameworks["mindquantum"]["passes"]) + + cmap = LinearSegmentedColormap.from_list( + "orbit_passes", ["#E9B5AB", "#F3D7A2", "#CFE1C9", "#B8D8D8"] + ) + norm = Normalize(3, 12) + fig, ax = plt.subplots(figsize=(10.2, 2.75)) + ax.set_xlim(-0.85, len(agents)) + ax.set_ylim(4.25, -1.1) + ax.axis("off") + + for i in range(4): + for j in range(len(agents)): + value = values[i, j] + color = "#F7F7F7" if np.isnan(value) else cmap(norm(value)) + edge = "#D8D8D8" if np.isnan(value) else mpl.colors.to_hex(np.array(mpl.colors.to_rgb(color)) * 0.82) + ax.add_patch(Rectangle((j + 0.06, i + 0.06), 0.88, 0.88, facecolor=color, edgecolor=edge, linewidth=1.0)) + text = "–" if np.isnan(value) else f"{int(value)}/12" + ax.text(j + 0.50, i + 0.50, text, ha="center", va="center", fontweight="bold" if not np.isnan(value) else "normal", color="#333333") + + row_labels = ["TC", "PL", "TQ", "MQ"] + for i, label in enumerate(row_labels): + ax.text(-0.08, i + 0.50, label, ha="right", va="center", fontsize=9.5) + + labels = [ + "GPT-5.5", + "Opus-4.8", + "GLM-5.2", + "Sonnet-4.6", + "5.6 Sol", + "5.6 Sol\nultra", + "5.6 Terra", + "5.6 Luna", + "DeepSeek\nV4 Flash", + "Grok 4.5", + "DeepSeek\nV4 Pro", + "Grok 4.6", + "GPT-6\nAstra", + ] + for j, label in enumerate(labels): + ax.text(j + 0.50, -0.02, label, ha="center", va="bottom", fontsize=7.7, + color=COLORS["astra_high"] if agents[j]["key"] == "astra_high" else "#222222") + + ax.text(-0.68, 2.0, "Framework", rotation=90, ha="center", va="center", fontsize=9.5, fontweight="bold", color="#666666") + ax.set_title("Updated agent–framework benchmarking matrix", loc="left", fontweight="bold", pad=18) + save_all(fig, "fig1c_updated_agent_framework_matrix") + + +def make_fig2b(agent_rows: list[dict[str, str]]) -> None: + rows = [r for r in agent_rows if r["include_fig2b"] == "yes"] + paired = read_csv("astra_expert_runtime.csv") + paired_ratios = [num(r, "astra_sec") / num(r, "expert_reference_sec") + for r in paired] + astra_ratio = math.exp(sum(math.log(v) for v in paired_ratios) / len(paired_ratios)) + fig, ax = plt.subplots(figsize=(5.3, 4.4)) + clean_axes(ax) + ax.axhline(1.0, color="#888888", ls=(0, (3, 3)), lw=0.9, zorder=1) + ax.axvline(0.0, color="#888888", ls=(0, (3, 3)), lw=0.9, zorder=1) + # The diamond is the normalization anchor, not a measured failure rate + # for the expert (Astra's Task 08 expert run did not complete). + ax.scatter([0], [1], marker="D", s=82, facecolor="none", edgecolor="#888888", zorder=2) + ax.text(1.3, 1.40, "Expert TC\nreference", fontsize=7.5, color="#777777", ha="left", va="center") + + # Preserve the original paper's four agent colors and label positions. + # Additional configurations use distinct colors in the same axis. + fig2_colors = { + "gpt55": "#0072B2", + "opus48": "#E69F00", + "sonnet46": "#CC79A7", + "glm52": "#D55E00", + "sol56_high": "#A65628", + "sol56_ultra": "#009E73", + "terra56_high": "#56B4E9", + "luna56_high": "#8C6D31", + "deepseek_v4_flash_high": "#4D4D4D", + "grok45_high": "#111111", + "deepseek_v4_pro_high": COLORS["deepseek_v4_pro_high"], + "grok46_high": COLORS["grok46_high"], + "astra_high": COLORS["astra_high"], + } + label_positions = { + "gpt55": (18.2, 1.30, "left"), + "opus48": (24.0, 3.70, "right"), + "sonnet46": (47.2, 8.15, "left"), + "glm52": (57.0, 7.10, "left"), + "sol56_high": (14.6, 3.00, "right"), + "sol56_ultra": (14.2, 2.15, "right"), + "terra56_high": (29.5, 2.78, "left"), + "luna56_high": (28.2, 5.00, "left"), + "deepseek_v4_flash_high": (58.3, 4.08, "center"), + "grok45_high": (36.0, 4.12, "left"), + "deepseek_v4_pro_high": (58.3, 2.02, "center"), + "grok46_high": (30.0, 2.00, "left"), + "astra_high": (3.5, 0.08, "left"), + } + for row in rows: + x = 100.0 * int(row["failures"]) / 12.0 + y = astra_ratio if row["key"] == "astra_high" else num(row, "gm_slowdown") + color = fig2_colors[row["key"]] + ax.scatter(x, y, marker="o", s=34, facecolor=color, edgecolor="#222222", linewidth=0.8, alpha=0.9, zorder=3) + label_x, label_y, align = label_positions[row["key"]] + label = f"{row['short_label']}\n({row['passes']}/12)" + ax.text( + label_x, + label_y, + label, + fontsize=7.2, + fontweight="bold", + color=color, + ha=align, + va="center", + ) + + # Negative limits provide visual padding only; data coordinates are unchanged. + ax.set_xlim(-10, 72) + ax.set_ylim(-0.6, 9.2) + ax.set_xticks(np.arange(0, 71, 10)) + ax.set_yticks([0, 1, 3, 5, 7, 9]) + ax.set_xlabel("Failure rate (%)") + ax.set_ylabel("Runtime / expert TC reference") + panel_label(ax, "(b)") + save_all(fig, "fig2b_updated_agent_axis") + + +def make_fig4(agent_rows: list[dict[str, str]], framework_rows: list[dict[str, str]]) -> None: + fig = plt.figure(figsize=(16.0, 9.4)) + gs = fig.add_gridspec(2, 3, width_ratios=[1.15, 1.15, 1.15], + height_ratios=[1.25, 1], wspace=0.50, hspace=0.42) + axes = [fig.add_subplot(gs[i, j]) for i in range(2) for j in range(3)] + axa, axb, axc, axd, axe, axf = axes + + # Agent axis: resource totals for every reported configuration. + y = np.arange(len(agent_rows)) + passed_h = np.array([num(r, "wall_passed_sec") / 3600 for r in agent_rows]) + failed_h = np.array([num(r, "wall_failed_sec") / 3600 for r in agent_rows]) + axa.barh(y, passed_h, color=PASS, edgecolor="#222222", linewidth=0.7, label="Passed tasks", zorder=2) + axa.barh(y, failed_h, left=passed_h, color=FAIL, edgecolor="#222222", linewidth=0.7, label="Failed tasks", zorder=2) + axa.set_yticks(y, [r["short_label"] for r in agent_rows]) + axa.invert_yaxis() + axa.set_xlabel("Agent solve wall time (h)") + axa.set_title("Agent axis", fontweight="bold") + axa.legend(loc="lower right", frameon=False) + clean_axes(axa, grid_axis="x") + panel_label(axa, "(a)") + + axb.barh( + y, + [num(row, "total_tokens_m") for row in agent_rows], + color=CACHE, + edgecolor="#222222", + linewidth=0.7, + zorder=2, + ) + axb.set_yticks(y, [r["short_label"] for r in agent_rows]) + axb.invert_yaxis() + axb.set_xlabel("Total solving-side tokens (million)") + clean_axes(axb, grid_axis="x") + panel_label(axb, "(b)") + + agent_label_positions = { + "gpt55": (6.2, 1.98), + "gpt55_checklist": (9.0, 1.10), + "opus48": (22.7, 1.82), + "glm52": (34.0, 2.50), + "sonnet46": (32.0, 1.72), + "sol56_high": (11.5, 2.74), + "sol56_ultra": (6.0, 3.10), + "terra56_high": (26.5, 1.47), + "luna56_high": (25.5, 0.43), + "deepseek_v4_flash_high": (45.0, 0.35), + "grok45_high": (17.0, 0.79), + "deepseek_v4_pro_high": (60.5, 0.60), + "grok46_high": (25.0, 2.30), + "astra_high": (0.8, 0.80), + } + for row in agent_rows: + x = num(row, "wall_total_sec") / 60 / int(row["passes"]) + yy = num(row, "cost_usd") / int(row["passes"]) + if np.isnan(yy): + continue + size = 30 + 11 * num(row, "cost_usd") + color = COLORS[row["key"]] + axc.scatter(x, yy, s=size, marker="o", facecolor=color, edgecolor="#222222", linewidth=0.8, alpha=0.95, zorder=3) + label_x, label_y = agent_label_positions[row["key"]] + axc.annotate( + f"{row['short_label']}\n({row['passes']}/12)", + (x, yy), + xytext=(label_x, label_y), + textcoords="data", + fontsize=6.4, + fontweight="bold", + color=color, + va="center", + ) + axc.set_xlim(0, 76) + axc.set_ylim(0, 3.35) + axc.set_xlabel("Solve time per valid solution (min)") + axc.set_ylabel("Solver cost per valid solution (USD)") + axc.text(0.98, 0.97, "Marker area scales with total solver cost", transform=axc.transAxes, ha="right", va="top", fontsize=7.2, color="#444444") + clean_axes(axc) + panel_label(axc, "(c)") + + # Framework axis: unchanged original-paper comparison. + fy = np.arange(len(framework_rows)) + fpass = np.array([num(r, "wall_passed_sec") / 3600 for r in framework_rows]) + ffail = np.array([num(r, "wall_failed_sec") / 3600 for r in framework_rows]) + axd.barh(fy, fpass, color=PASS, edgecolor="#222222", linewidth=0.7, zorder=2) + axd.barh(fy, ffail, left=fpass, color=FAIL, edgecolor="#222222", linewidth=0.7, zorder=2) + axd.set_yticks(fy, [r["short_label"] for r in framework_rows]) + axd.invert_yaxis() + axd.set_xlabel("Agent solve wall time (h)") + axd.set_title("Framework axis", fontweight="bold") + clean_axes(axd, grid_axis="x") + panel_label(axd, "(d)") + + axe.barh(fy, [num(r, "total_tokens_m") for r in framework_rows], color=CACHE, edgecolor="#222222", linewidth=0.7, zorder=2) + axe.set_yticks(fy, [r["short_label"] for r in framework_rows]) + axe.invert_yaxis() + axe.set_xlabel("Total solving-side tokens (million)") + clean_axes(axe, grid_axis="x") + panel_label(axe, "(e)") + + foffsets = {"tensorcircuit": (8, 6), "pennylane": (8, 5), "torchquantum": (-48, -13), "mindquantum": (-48, 8)} + for row in framework_rows: + x = num(row, "wall_total_sec") / 60 / int(row["passes"]) + yy = num(row, "cost_usd") / int(row["passes"]) + size = 30 + 11 * num(row, "cost_usd") + color = COLORS[row["key"]] + axf.scatter(x, yy, s=size, facecolor=color, edgecolor="#222222", linewidth=0.8, zorder=3) + dx, dy = foffsets[row["key"]] + axf.annotate(f"{row['short_label']}\n({row['passes']}/12)", (x, yy), xytext=(dx, dy), textcoords="offset points", fontsize=7.2, fontweight="bold", color=color, va="center") + axf.set_xlim(7, 45) + axf.set_ylim(0.8, 8.1) + axf.set_xlabel("Solve time per valid solution (min)") + axf.set_ylabel("Recorded cost per valid solution (USD)") + clean_axes(axf) + panel_label(axf, "(f)") + + fig.suptitle("Updated ORBIT-Q benchmark resource use", fontsize=13, fontweight="bold", y=0.995) + save_all(fig, "fig4_updated_agent_framework_resources") + + +def make_expert_optimization() -> None: + rows = read_csv("expert_optimization.csv") + tasks = [r["task"] for r in rows] + baseline = np.array([num(r, "baseline_sec") for r in rows]) + optimized = np.array([num(r, "optimized_sec") for r in rows]) + speedup = np.array([num(r, "speedup") for r in rows]) + short = { + "01": "batched gate construction", + "02": "batched exact purity", + "03": "exact product-state reduction", + "04": "batched probe networks", + "05": "tuned OMECo path search", + "06": "TC-native jaxode", + "07": "exact ancilla/branch reduction", + "08": "bounded mapped sampling", + "09": "causal-cone pruning", + "10": "fixed contraction program", + "11": "layer and onsite fusion", + "12": "batched Padé SU4", + } + + fig, (axa, axb) = plt.subplots(1, 2, figsize=(12.2, 5.25), gridspec_kw={"width_ratios": [1.08, 1.35], "wspace": 0.35}) + x = np.arange(len(tasks)) + width = 0.38 + axa.bar(x - width / 2, baseline, width, color="#7A7A7A", edgecolor="#222222", linewidth=0.6, label="Original expert", zorder=2) + axa.bar(x + width / 2, optimized, width, color=PASS, edgecolor="#222222", linewidth=0.6, label="Human + AI", zorder=2) + axa.set_yscale("log") + axa.set_xticks(x, tasks) + axa.set_xlabel("Challenge") + axa.set_ylabel("End-to-end runtime (s, log scale)") + axa.legend(frameon=False, loc="upper right") + clean_axes(axa, grid_axis="y") + panel_label(axa, "(a)") + + y = np.arange(len(tasks)) + bar_colors = ["#D55E00" if t in {"03", "07"} else PASS for t in tasks] + bars = axb.barh(y, speedup, color=bar_colors, edgecolor="#222222", linewidth=0.75, zorder=2) + axb.axvline(1.0, color="#777777", ls=(0, (3, 3)), lw=0.9) + axb.set_xscale("log") + axb.set_xlim(0.9, 78) + axb.set_yticks(y, [f"Task {t}" for t in tasks]) + axb.invert_yaxis() + axb.set_xlabel("End-to-end speedup (×, log scale)") + clean_axes(axb, grid_axis="x") + panel_label(axb, "(b)") + for i, (t, value) in enumerate(zip(tasks, speedup)): + axb.text(value * 1.06, i, f"{value:.2f}× {short[t]}", va="center", ha="left", fontsize=7.2, color="#333333") + + axb.legend( + handles=[ + Patch(facecolor=PASS, edgecolor="#222222", label="Framework-native optimization"), + Patch(facecolor="#D55E00", edgecolor="#222222", label="Exact task reduction"), + ], + loc="lower right", + frameon=False, + ) + fig.suptitle("Human-expert implementations after AI-assisted optimization", fontsize=12.5, fontweight="bold", y=0.995) + save_all(fig, "expert_optimization_updated_overview") + + +def make_astra_paired() -> None: + """Use eleven measured pairs plus the reviewer-approved Task 08 estimate.""" + rows = read_csv("astra_expert_runtime.csv") + tasks = [r["task"] for r in rows] + expert = np.array([num(r, "expert_reference_sec") for r in rows]) + astra = np.array([num(r, "astra_sec") for r in rows]) + speedups = np.array([num(r, "expert_over_astra") for r in rows]) + fig, (ax, bx) = plt.subplots(1, 2, figsize=(12.2, 5.25), + gridspec_kw={"width_ratios": [1.08, 1.35], "wspace": 0.35}) + x = np.arange(len(rows)) + width = 0.38 + purple = COLORS["astra_high"] + ax.bar(x-width/2, expert, width, color="#7A7A7A", edgecolor="#222222", + linewidth=0.6, label="Expert reference", zorder=2) + ax.bar(x+width/2, astra, width, color=purple, edgecolor="#222222", + linewidth=0.6, label="Astra high", zorder=2) + ax.set_yscale("log") + ax.set_ylim(1, 400) + ax.set_xticks(x, tasks) + ax.set_xlabel("Challenge") + ax.set_ylabel("End-to-end runtime (s, log scale)") + ax.legend(frameon=False, loc="upper right") + for i, r in enumerate(rows): + if np.isnan(expert[i]): + ax.text(i-width/2, 1.16, r["expert_status"], color="#7A7A7A", + fontsize=6.8, ha="center", va="bottom", rotation=90) + bx.barh(x, speedups, color=purple, edgecolor="#222222", linewidth=0.75, zorder=2) + bx.axvline(1, color="#777777", ls=(0,(3,3)), lw=0.9) + bx.set_xscale("log") + bx.set_xlim(0.1, 35) + bx.set_yticks(x, [f"Task {t}" for t in tasks]) + bx.invert_yaxis() + bx.set_xlabel("End-to-end speedup (expert / Astra, ×, log scale)") + for i, value in enumerate(speedups): + if np.isnan(value): + bx.text(1.08, i, f"Expert {rows[i]['expert_status']}", color="#555555", va="center", fontsize=7.2) + else: + bx.text(value*1.08, i, f"{value:.2f}×", va="center", ha="left", fontsize=7.2, color="#333333") + for panel, grid_axis, label in [(ax,"y","(a)"),(bx,"x","(b)")]: + clean_axes(panel, grid_axis=grid_axis) + panel.set_axisbelow(True) + panel_label(panel, label) + fig.suptitle("Astra artifact runtime relative to the expert reference", + fontsize=12.5, fontweight="bold", y=0.995) + save_all(fig, "astra_expert_paired_runtime") + + +def main() -> None: + agents = read_csv("paper_agent_axis.csv") + frameworks = read_csv("paper_framework_axis.csv") + validate_new_slowdowns(agents) + make_fig1c(agents, frameworks) + make_fig2b(agents) + make_fig4(agents, frameworks) + make_astra_paired() + # The earlier expert-optimization measurements and their figure are unchanged. + # Call make_expert_optimization() explicitly to reproduce that archived figure. + + +if __name__ == "__main__": + main() diff --git a/paper/source_data/README.md b/paper/source_data/README.md new file mode 100644 index 0000000..fcf39d0 --- /dev/null +++ b/paper/source_data/README.md @@ -0,0 +1,59 @@ +# Source-data notes + +- `paper_agent_axis.csv`: original paper agent-axis totals plus Sol, Sol ultra, + Terra, Luna, DeepSeek V4 Flash/Pro, Grok 4.5/4.6, and Astra (14 rows). It drives updated Fig. 1c, + and the top row of Fig. 4. Its `gm_slowdown` and `include_fig2b` fields + drive Fig. 2b except that Astra uses `astra_expert_runtime.csv`. + Unqualified added configurations use + high thinking effort; Sol ultra uses ultra. +- `paper_framework_axis.csv`: original paper framework-axis totals for the + unchanged bottom row of Fig. 4. +- `paper_expert_runtimes.csv`: original public per-task TensorCircuit expert + references retained to validate the archived fixed-denominator calculations. +- `benchmark_models.csv`: compact new-campaign aggregate table. +- `task_outcomes.csv`: 9 configurations × 12 tasks. `raw_reward` preserves the + verifier result; `final_pass` retains each source campaign's reported outcome, + including its published adjudication where applicable. +- `recent_pr_evidence.json`: pinned commits, source-file SHA-256 hashes, and + normalized per-task evidence for PRs #27–29, plus paired-run provenance. +- `astra_expert_paired.csv`: 12 fresh paired-runtime records from the + [PR #29 reply](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804). + Task 08 has a missing expert time (OOM) and is excluded from aggregates. +- `astra_expert_runtime.csv`: derived plot data for all 12 Astra tasks. + Eleven pairs retain the fresh measurements; Task 08 uses the estimated + **50 s** expert reference requested in the [PR #29 discussion](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776) + (25.05 s × 2, rounded). Its measured Astra time is 16.31 s. + This table supplies the Astra Fig. 2b point (**0.539698×**) and runtime bars. + `expert_reference_basis` distinguishes the estimate from measurements. +- `astra_expert_adjustment.json`: replacement provenance, original-source + hash, and recomputed aggregate values. Neither the original paired table + nor the historical `paper_agent_axis.csv` measurements are overwritten. +- `expert_optimization.csv`: original and optimized expert runtimes plus the + dominant factor retained after ablation. +- `insights.csv`: short take-home messages for the expert optimizations. + +The original values come from the main and supplementary tables of +[arXiv:2607.03105](https://arxiv.org/abs/2607.03105). The new campaign totals +come from ORBIT-Q PRs #5, #6, #20, #21, #23, #26, #27, #28, and #29. +Fig. 2b retains the fixed-paper-reference coordinates for all non-Astra models; +Astra uses its paired remeasurement with the Task 08 replacement described above. +The original CSV measurements are unchanged. + +The original paper tables expose total token use for the legacy configurations +but not the full cache/non-cache/output decomposition. Updated Fig. 4b/e +therefore compare total tokens with one consistent color for every bar. + +Task 08 is final `F` for the first eight added solver configurations. Luna and Sol ultra +retain raw reward `1` in `task_outcomes.csv`; `final_pass=0` records the +paper-facing decision. Astra retains its original 12/12 automatic result; +the maintainer subsequently accepted its Task 08 implementation in +[review](https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5548928438). +The newly added campaigns are not described as uniformly human-adjudicated. +This solver-campaign adjudication is separate from the +valid Task 08 human-expert optimization recorded in `expert_optimization.csv`. +Task 05 uses the legal TensorCircuit-native result from +[PR #19](https://github.com/sxzgroup/ORBIT-Q/pull/19): **1.939×** mean paired +end-to-end speedup (median **1.668×**). This is the mean of the five paired +expert/candidate ratios; one `180.22 s` expert-reference long tail raises it, +and the four-pair sensitivity mean is **1.639×**. The candidate won 5/5 pairs. +The earlier custom no-QR MPS result is excluded. diff --git a/paper/source_data/astra_expert_adjustment.json b/paper/source_data/astra_expert_adjustment.json new file mode 100644 index 0000000..af75a5e --- /dev/null +++ b/paper/source_data/astra_expert_adjustment.json @@ -0,0 +1,38 @@ +{ + "schema_version": 1, + "date": "2026-09-05", + "measured_source": "astra_expert_paired.csv", + "measured_source_sha256": "9d151a187c4b73e08c95ced4cfcbaae29afcd34bffdddecb376537710cd34e48", + "runtime_table": "astra_expert_runtime.csv", + "task08": { + "expert_reference_sec": 50, + "basis": "estimated", + "source": "https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776", + "published_reference_sec": 25.05, + "reviewer_scale_factor": 2, + "reviewer_proposed_sec": 50.1, + "rounding": "Nearest whole second (50 s), as approved for this figure update.", + "original_expert_status": "OOM", + "astra_sec": 16.31 + }, + "summary": { + "task_count": 12, + "measured_pair_count": 11, + "estimated_reference_count": 1, + "geometric_mean_ratio": 0.539697529957958, + "geometric_mean_speedup": 1.8528897104233535, + "expert_sum_sec": 640.11, + "astra_sum_sec": 440.81, + "ratio_of_sums": 0.6886472637515427, + "astra_faster_tasks": [ + 1, + 4, + 5, + 7, + 8, + 9, + 11, + 12 + ] + } +} diff --git a/paper/source_data/astra_expert_paired.csv b/paper/source_data/astra_expert_paired.csv new file mode 100644 index 0000000..9ab4c98 --- /dev/null +++ b/paper/source_data/astra_expert_paired.csv @@ -0,0 +1,13 @@ +task,expert_sec,astra_sec,astra_over_expert,expert_over_astra,expert_status,astra_status,included_in_aggregate,source +01,58.59,13.68,0.233487,4.282895,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +02,4.15,9.84,2.371084,0.421748,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +03,3.82,5.1,1.335079,0.74902,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +04,13.13,2.33,0.177456,5.635193,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +05,91.87,5.26,0.057255,17.465779,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +06,44.26,104.66,2.364663,0.422893,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +07,147.46,21.22,0.143903,6.949105,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +08,,16.31,,,OOM,PASS,0,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +09,23.64,8.63,0.365059,2.739282,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +10,21.51,105.57,4.90795,0.203751,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +11,171.11,143.04,0.835953,1.196239,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +12,10.57,5.17,0.48912,2.044487,PASS,PASS,1,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 diff --git a/paper/source_data/astra_expert_runtime.csv b/paper/source_data/astra_expert_runtime.csv new file mode 100644 index 0000000..54654c6 --- /dev/null +++ b/paper/source_data/astra_expert_runtime.csv @@ -0,0 +1,13 @@ +task,expert_reference_sec,astra_sec,astra_over_expert,expert_over_astra,expert_reference_basis,source +01,58.59,13.68,0.2334869431643625,4.282894736842105,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +02,4.15,9.84,2.3710843373493975,0.42174796747967486,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +03,3.82,5.1,1.3350785340314135,0.7490196078431373,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +04,13.13,2.33,0.17745620715917745,5.635193133047211,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +05,91.87,5.26,0.05725481658865788,17.46577946768061,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +06,44.26,104.66,2.3646633529145955,0.422893177909421,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +07,147.46,21.22,0.1439034314390343,6.949104618284638,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +08,50,16.31,0.3262,3.065603923973023,estimated,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549266776 +09,23.64,8.63,0.3650592216582065,2.73928157589803,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +10,21.51,105.57,4.907949790794978,0.20375106564364878,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +11,171.11,143.04,0.8359534802174039,1.1962388143176736,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 +12,10.57,5.17,0.489120151371807,2.044487427466151,measured,https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804 diff --git a/paper/source_data/benchmark_models.csv b/paper/source_data/benchmark_models.csv new file mode 100644 index 0000000..5715d60 --- /dev/null +++ b/paper/source_data/benchmark_models.csv @@ -0,0 +1,10 @@ +model,label,effort,passes,failures,agent_wall_min,total_tokens_m,input_tokens_m,cache_tokens_m,output_tokens_m,cost_usd,cost_per_valid_usd,solve_time_per_valid_min,gm_slowdown,resource_comparable,notes +gpt56_sol_high,GPT-5.6 Sol,high,10,2,197.70,26.071,25.908,24.199,0.163,25.527,2.553,19.770,2.541,yes,final Task 05 API adjudication; Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references +gpt56_sol_ultra,GPT-5.6 Sol,ultra,10,2,182.80,33.048,32.790,31.478,0.257,30.019,3.002,18.280,2.072,yes,Task 07 source adjudication pass; Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references +gpt56_terra_high,GPT-5.6 Terra,high,9,3,206.43,36.804,36.597,35.455,0.207,12.037,1.337,22.937,2.702,yes,Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references +gpt56_luna_high,GPT-5.6 Luna,high,9,3,239.37,74.853,74.496,72.727,0.357,2.237,0.249,26.597,4.692,yes,Task 08 final expert failure; slowdown recomputed against the original paper expert-TC references +deepseek_v4_flash_high,DeepSeek V4 Flash,high,5,7,290.41,81.585,80.672,80.008,0.913,0.573,0.115,58.082,3.476,yes,official DeepSeek integration; five final passes; slowdown recomputed against the original paper expert-TC references +grok45_high,Grok 4.5,high,8,4,163.09,8.462,8.255,7.449,0.207,5.090,0.636,20.386,3.844,yes,normalized repaired compatibility protocol; cost reconstructed from recorded token classes at xAI public list prices; slowdown recomputed against the original paper expert-TC references +deepseek_v4_pro_high,DeepSeek V4 Pro,high,5,7,324.22412,67.18005,66.51614,65.861504,0.66391,1.101116,0.220223,64.844824,2.587924,yes,https://github.com/sxzgroup/ORBIT-Q/pull/27; published campaign outcome; runtime uses original paper expert references +grok46_high,Grok 4.6,high,9,3,199.165271,29.767008,29.277393,27.575296,0.489615,20.129532,2.236615,22.129475,2.190531,yes,https://github.com/sxzgroup/ORBIT-Q/pull/28; published campaign outcome; runtime uses original paper expert references +astra_high,GPT-6 Astra,high,12,0,44.71885,6.834954,6.768398,6.338304,0.066556,13.967044,1.16392,3.726571,0.974148,yes,https://github.com/sxzgroup/ORBIT-Q/pull/29; automatic verifier outcome; runtime uses original paper expert references diff --git a/paper/source_data/expert_optimization.csv b/paper/source_data/expert_optimization.csv new file mode 100644 index 0000000..3531e2c --- /dev/null +++ b/paper/source_data/expert_optimization.csv @@ -0,0 +1,13 @@ +task,baseline_sec,optimized_sec,speedup,dominant_factor,dominant_factor_speedup,dominant_factor_context,source_branch,measurement_note +01,60.651144,6.359807,9.636410,batched closed-form gate construction,3.575,end-to-end factor screen,codex/task-01-final-expert-optimization,six matched local pairs +02,4.495463,4.031039,1.115649,batched exact purity,1.074,end-to-end factor screen,codex/task-02-final-expert-optimization,six matched local pairs +03,4.259896,0.870991,4.893978,TensorCircuit K.vmap local conditional maps,2.063,incremental factor screen,codex/task-03-final-expert-optimization,original workload; exact reduction +04,14.742286,5.672258,2.602133,probe-network K.vmap,2.165,incremental factor screen,codex/task-04-final-expert-optimization,six matched local pairs +05,115.708000,60.114000,1.939000,OMECo 4x4 path-search budget,1.420,direct fused-circuit ablation,codex/task-05-tc-native-fused,five cold fresh-process triples; mean paired speedup; median 1.668x +06,41.425900,27.536600,1.504460,TensorCircuit native jaxode,1.529,single-screen dominant factor,codex/task-06-final-expert-optimization,five matched local pairs +07,140.076441,3.070839,45.757921,classical-ancilla and duplicate-trajectory reduction,45.758,end-to-end factor screen,codex/task-07-classical-ancilla-design-report,six matched local pairs; exact design loophole +08,126.675000,123.188000,1.045000,bounded mapped sampling batch,1.045,small paired end-to-end gain,codex/task-08-final-expert-optimization,five matched pairs; valid optimized implementation +09,33.503727,8.766516,3.821725,preconstructed causal cones plus compiled trajectory,3.8217,bundled end-to-end factor,codex/task-09-final-expert-optimization,six matched local pairs +10,18.931000,3.869000,4.898000,native hyperedge with fixed contraction path,1.626,largest isolated factor,codex/task-10-final-expert-optimization,five paired cold-specialization runs +11,168.361539,114.968325,1.464430,layer fusion plus diagonal onsite reduction,81.215,isolated onsite-term factor; not end-to-end,codex/task-11-final-expert-optimization,six matched local pairs +12,9.082742,2.320613,3.914003,batched fixed-order Pade SU4 construction,3.109,isolated gate-build/gradient kernel,codex/task-12-final-expert-optimization,six matched local pairs diff --git a/paper/source_data/insights.csv b/paper/source_data/insights.csv new file mode 100644 index 0000000..596aa02 --- /dev/null +++ b/paper/source_data/insights.csv @@ -0,0 +1,7 @@ +task,insight,loophole_or_design_finding,primary_factor,measured_effect +03,Post-selection after every brickwork layer leaves six independent one-qubit survivors,exact product-state reduction in original task,product reduction,4.894x end to end +07,Measured ancillas form an analytically sampled classical controller with only two unique branches,exact classical-ancilla/trajectory reduction in published task,branch and trajectory merging,45.758x end to end +10,The advantage is cold specialization and fixed contraction planning rather than MPO gate arithmetic,hyperedge path search and graph preprocessing dominate; fixed-path hyperedge versus MPO is 1.011x,native hyperedge with fixed path,4.898x end to end +05,Keep the quantum process TensorCircuit-native and give OMECo enough search budget to avoid bad contraction paths,gate fusion alone is unresolved; OMECo 4x4 is the dominant measured factor,OMECo path-search budget,1.939x mean paired end to end +09,Explicit causal-cone pruning plus compact TensorCircuit graphs removes irrelevant global work,not a loophole; framework-native local contraction,causal cones and compiled trajectory,3.822x end to end +12,Batched fixed-order Pade construction shrinks the differentiated gate-build graph,not a loophole; fixed-order static kernel,batched SU4 construction,3.914x end to end diff --git a/paper/source_data/paper_agent_axis.csv b/paper/source_data/paper_agent_axis.csv new file mode 100644 index 0000000..1d6e132 --- /dev/null +++ b/paper/source_data/paper_agent_axis.csv @@ -0,0 +1,15 @@ +key,label,short_label,series,passes,failures,gm_slowdown,wall_total_sec,wall_passed_sec,wall_failed_sec,total_tokens_m,noncache_prompt_m,cache_prompt_m,output_tokens_m,cost_usd,include_fig1c,include_fig2b,source +gpt55,GPT-5.5,GPT-5.5,paper,10,2,2.197481,7512.3,5116.4,2395.9,14.241691,,,,17.99,yes,yes,arXiv:2607.03105 Tables S2-S3 +gpt55_checklist,GPT-5.5 + checklist,GPT-5.5*,paper,10,2,2.190668,5343.4,3917.8,1425.6,8.205577,,,,13.29,no,no,arXiv:2607.03105 Tables S2-S4 +opus48,Claude Opus-4.8,Opus-4.8,paper,9,3,2.889799,10382.4,8752.1,1630.3,12.203758,,,,17.52,yes,yes,arXiv:2607.03105 Tables S2-S5 +glm52,GLM-5.2,GLM-5.2,paper,6,6,6.616186,13811.1,6834.2,6976.9,21.075177,,,,12.98,yes,yes,arXiv:2607.03105 Tables S2-S6 +sonnet46,Claude Sonnet-4.6,Sonnet-4.6,paper,7,5,7.245877,13644.2,7715.0,5929.2,21.268795,,,,14.55,yes,yes,arXiv:2607.03105 Tables S2-S7 +sol56_high,GPT-5.6 Sol,Sol,new,10,2,2.541474,11861.715963,9220.290130,2641.425833,26.070809,1.709460,24.198656,0.162693,25.527418,yes,yes,ORBIT-Q PR 5 plus final Task 08 adjudication +sol56_ultra,GPT-5.6 Sol ultra,Sol ultra,new,10,2,2.072336,10967.757000,7892.985000,3074.772000,33.047608,1.312299,31.478016,0.257293,30.019293,yes,yes,ORBIT-Q PR 6 plus final Task 08 adjudication +terra56_high,GPT-5.6 Terra,Terra,new,9,3,2.701777,12385.907000,9300.564036,3085.342537,36.803664,1.142037,35.454720,0.206907,12.036862,yes,yes,ORBIT-Q PR 20 plus final Task 08 adjudication +luna56_high,GPT-5.6 Luna,Luna,new,9,3,4.691620,14362.147000,10744.733972,3617.413183,74.852649,1.769004,72.726528,0.357117,2.236872,yes,yes,ORBIT-Q PR 21 plus final Task 08 adjudication +deepseek_v4_flash_high,DeepSeek V4 Flash,DeepSeek Flash,new,5,7,3.476293,17424.726000,5869.032092,11555.693612,81.584990,0.663753,80.007936,0.913301,0.572672,yes,yes,ORBIT-Q PR 23 +grok45_high,Grok 4.5,Grok 4.5,new,8,4,3.844088,9785.490000,5394.966612,4390.523575,8.462179,0.805236,7.449472,0.207471,5.090140,yes,yes,ORBIT-Q PR 26; normalized repaired compatibility protocol; cost reconstructed from recorded token classes at xAI public list prices +deepseek_v4_pro_high,DeepSeek V4 Pro,DeepSeek Pro,new,5,7,2.587924,19453.447205,5792.96097,13660.486235,67.18005,0.654636,65.861504,0.66391,1.101116,yes,yes,https://github.com/sxzgroup/ORBIT-Q/pull/27 +grok46_high,Grok 4.6,Grok 4.6,new,9,3,2.190531,11949.916262,7658.766225,4291.150037,29.767008,1.702097,27.575296,0.489615,20.129532,yes,yes,https://github.com/sxzgroup/ORBIT-Q/pull/28 +astra_high,GPT-6 Astra,Astra,new,12,0,0.974148,2683.130984,2683.130984,0,6.834954,0.430094,6.338304,0.066556,13.967044,yes,yes,https://github.com/sxzgroup/ORBIT-Q/pull/29 diff --git a/paper/source_data/paper_expert_runtimes.csv b/paper/source_data/paper_expert_runtimes.csv new file mode 100644 index 0000000..e6b6a42 --- /dev/null +++ b/paper/source_data/paper_expert_runtimes.csv @@ -0,0 +1,13 @@ +task,expert_tc_runtime_sec,source +1,27.22,arXiv:2607.03105 public TensorCircuit reference +2,2.87,arXiv:2607.03105 public TensorCircuit reference +3,2.46,arXiv:2607.03105 public TensorCircuit reference +4,11.83,arXiv:2607.03105 public TensorCircuit reference +5,45.50,arXiv:2607.03105 public TensorCircuit reference +6,26.83,arXiv:2607.03105 public TensorCircuit reference +7,63.80,arXiv:2607.03105 public TensorCircuit reference +8,25.05,arXiv:2607.03105 public TensorCircuit reference +9,13.74,arXiv:2607.03105 public TensorCircuit reference +10,12.44,arXiv:2607.03105 public TensorCircuit reference +11,68.10,arXiv:2607.03105 public TensorCircuit reference +12,6.12,arXiv:2607.03105 public TensorCircuit reference diff --git a/paper/source_data/paper_framework_axis.csv b/paper/source_data/paper_framework_axis.csv new file mode 100644 index 0000000..a7058ba --- /dev/null +++ b/paper/source_data/paper_framework_axis.csv @@ -0,0 +1,5 @@ +key,label,short_label,passes,failures,gm_slowdown,wall_total_sec,wall_passed_sec,wall_failed_sec,total_tokens_m,cost_usd,source +tensorcircuit,TensorCircuit-NG,TC,10,2,2.197481,7512.3,5116.4,2395.9,14.241691,17.99,arXiv:2607.03105 Tables S2-S3 +pennylane,PennyLane,PL,8,4,,9575.7,6482.5,3093.2,16.579728,23.04,arXiv:2607.03105 Tables S2 and S8 +torchquantum,TorchQuantum,TQ,4,8,,8310.5,1929.0,6381.5,20.298642,24.91,arXiv:2607.03105 Tables S2 and S9 +mindquantum,MindQuantum,MQ,4,8,,9762.4,3028.2,6734.2,22.928678,29.03,arXiv:2607.03105 Tables S2 and S10 diff --git a/paper/source_data/recent_pr_evidence.json b/paper/source_data/recent_pr_evidence.json new file mode 100644 index 0000000..f45c7b5 --- /dev/null +++ b/paper/source_data/recent_pr_evidence.json @@ -0,0 +1,712 @@ +{ + "schema_version": 1, + "as_of": "2026-09-05", + "runtime_basis": "Original campaign artifact times divided by the same fixed paper expert references; paired remeasurement is separate", + "models": [ + { + "key": "deepseek_v4_pro_high", + "label": "DeepSeek V4 Pro", + "effort": "high", + "pr": 27, + "source": { + "commit": "78a5a57e49edda1e6f43f2021be508d288028ed7", + "path": "results/deepseek-v4-pro-high/summary.json", + "sha256": "43399c9a26c732e21ab2cde4944527e5f43ad002fc386438f4bd4d005edc237c", + "url": "https://github.com/QingyunQian/ORBIT-Q/blob/78a5a57e49edda1e6f43f2021be508d288028ed7/results/deepseek-v4-pro-high/summary.json" + }, + "outcome_basis": "Published selected campaign outcomes", + "cost_basis": "Recorded solver cost", + "tasks": [ + { + "task": 1, + "raw_reward": 0, + "functional": 0, + "static": 0, + "audit": 0, + "runtime_sec": null, + "agent_wall_sec": 1801.240353, + "input_tokens": 12630069, + "cache_tokens": 12539520, + "output_tokens": 82650, + "cost_usd": 0.156750075, + "final_pass": 0 + }, + { + "task": 2, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 21.66, + "agent_wall_sec": 1771.884543, + "input_tokens": 6714293, + "cache_tokens": 6644352, + "output_tokens": 56592, + "cost_usd": 0.10374515100000001, + "final_pass": 1 + }, + { + "task": 3, + "raw_reward": 0, + "functional": 0, + "static": 0, + "audit": 0, + "runtime_sec": null, + "agent_wall_sec": 1800.31708, + "input_tokens": 7093374, + "cache_tokens": 7037568, + "output_tokens": 56664, + "cost_usd": 0.099084474, + "final_pass": 0 + }, + { + "task": 4, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 20.23, + "agent_wall_sec": 858.029259, + "input_tokens": 2325275, + "cache_tokens": 2296576, + "output_tokens": 43822, + "cost_usd": 0.058934293000000006, + "final_pass": 1 + }, + { + "task": 5, + "raw_reward": 0, + "functional": 1, + "static": 1, + "audit": 0, + "runtime_sec": 77.39, + "agent_wall_sec": 961.020746, + "input_tokens": 3581169, + "cache_tokens": 3537536, + "output_tokens": 41165, + "cost_usd": 0.067617473, + "final_pass": 0 + }, + { + "task": 6, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 36.23, + "agent_wall_sec": 802.222496, + "input_tokens": 2491315, + "cache_tokens": 2456192, + "output_tokens": 36022, + "cost_usd": 0.055521341, + "final_pass": 1 + }, + { + "task": 7, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 97.41, + "agent_wall_sec": 1251.009781, + "input_tokens": 4908390, + "cache_tokens": 4851072, + "output_tokens": 43511, + "cost_usd": 0.08037303600000001, + "final_pass": 1 + }, + { + "task": 8, + "raw_reward": 0, + "functional": 0, + "static": 1, + "audit": 0, + "runtime_sec": 18.99, + "agent_wall_sec": 1802.81925, + "input_tokens": 6185697, + "cache_tokens": 6124544, + "output_tokens": 98275, + "cost_usd": 0.134302277, + "final_pass": 0 + }, + { + "task": 9, + "raw_reward": 0, + "functional": 1, + "static": 1, + "audit": 0, + "runtime_sec": 98.91, + "agent_wall_sec": 3693.518026, + "input_tokens": 1235634, + "cache_tokens": 1218176, + "output_tokens": 28870, + "cost_usd": 0.037127018, + "final_pass": 0 + }, + { + "task": 10, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 54.27, + "agent_wall_sec": 1109.814891, + "input_tokens": 4183354, + "cache_tokens": 4121856, + "output_tokens": 46730, + "cost_usd": 0.082348458, + "final_pass": 1 + }, + { + "task": 11, + "raw_reward": 0, + "functional": 0, + "static": 0, + "audit": 0, + "runtime_sec": null, + "agent_wall_sec": 1800.354635, + "input_tokens": 7224964, + "cache_tokens": 7154048, + "output_tokens": 57859, + "cost_usd": 0.107119214, + "final_pass": 0 + }, + { + "task": 12, + "raw_reward": 0, + "functional": 0, + "static": 0, + "audit": 0, + "runtime_sec": null, + "agent_wall_sec": 1801.216145, + "input_tokens": 7942606, + "cache_tokens": 7880064, + "output_tokens": 71750, + "cost_usd": 0.118193502, + "final_pass": 0 + } + ] + }, + { + "key": "grok46_high", + "label": "Grok 4.6", + "effort": "high", + "pr": 28, + "source": { + "commit": "09476b847cc2ff53603002dbf0bf7f9c34a148cd", + "path": "results/grok-4.6-high/summary.json", + "sha256": "6d513fe8e81d0711277d97f359209d63311d9b5fcf2b095a3118672a2e817f55", + "url": "https://github.com/QingyunQian/ORBIT-Q/blob/09476b847cc2ff53603002dbf0bf7f9c34a148cd/results/grok-4.6-high/summary.json" + }, + "outcome_basis": "Published selected campaign outcomes", + "cost_basis": "Archived campaign list-price cost", + "tasks": [ + { + "task": 1, + "raw_reward": 0, + "functional": 0, + "static": 0, + "audit": 0, + "runtime_sec": null, + "agent_wall_sec": 1800.584437, + "input_tokens": 6584277, + "cache_tokens": 6364160, + "output_tokens": 50483, + "cost_usd": null, + "final_pass": 0 + }, + { + "task": 2, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 12.36, + "agent_wall_sec": 941.422089, + "input_tokens": 1536982, + "cache_tokens": 1438336, + "output_tokens": 29495, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 3, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 5.91, + "agent_wall_sec": 725.766953, + "input_tokens": 1418857, + "cache_tokens": 1288576, + "output_tokens": 31641, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 4, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 17.61, + "agent_wall_sec": 1268.647419, + "input_tokens": 2605577, + "cache_tokens": 2451328, + "output_tokens": 51488, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 5, + "raw_reward": 0, + "functional": 1, + "static": 1, + "audit": 0, + "runtime_sec": 72.2, + "agent_wall_sec": 960.122457, + "input_tokens": 2772314, + "cache_tokens": 2659072, + "output_tokens": 41979, + "cost_usd": null, + "final_pass": 0 + }, + { + "task": 6, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 24.95, + "agent_wall_sec": 815.971732, + "input_tokens": 1302858, + "cache_tokens": 1230336, + "output_tokens": 33499, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 7, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 36.03, + "agent_wall_sec": 1181.460512, + "input_tokens": 3059611, + "cache_tokens": 2780544, + "output_tokens": 51754, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 8, + "raw_reward": 0, + "functional": 1, + "static": 1, + "audit": 0, + "runtime_sec": 47.47, + "agent_wall_sec": 1530.443143, + "input_tokens": 3860732, + "cache_tokens": 3677568, + "output_tokens": 76974, + "cost_usd": null, + "final_pass": 0 + }, + { + "task": 9, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 58.45, + "agent_wall_sec": 597.030343, + "input_tokens": 1167630, + "cache_tokens": 1058688, + "output_tokens": 27174, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 10, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 65.01, + "agent_wall_sec": 555.462494, + "input_tokens": 1037294, + "cache_tokens": 969344, + "output_tokens": 21748, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 11, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 180.41, + "agent_wall_sec": 1075.856693, + "input_tokens": 2714550, + "cache_tokens": 2612736, + "output_tokens": 47074, + "cost_usd": null, + "final_pass": 1 + }, + { + "task": 12, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 14.92, + "agent_wall_sec": 497.14799, + "input_tokens": 1216711, + "cache_tokens": 1044608, + "output_tokens": 26306, + "cost_usd": null, + "final_pass": 1 + } + ] + }, + { + "key": "astra_high", + "label": "GPT-6 Astra", + "effort": "high", + "pr": 29, + "source": { + "commit": "b28f53770c606d9c3c499c5ea10613eac197c1c2", + "path": "results/gpt6astra-high/comparison.json", + "sha256": "7efb1aa710fe6fdae2bdacf4cf21db8d71f0c0148c77f2e8fc89b665be212aea", + "url": "https://github.com/QingyunQian/ORBIT-Q/blob/b28f53770c606d9c3c499c5ea10613eac197c1c2/results/gpt6astra-high/comparison.json" + }, + "outcome_basis": "Automatic verifier outcome; no blanket Task 08 failure override", + "cost_basis": "Recorded solver cost", + "tasks": [ + { + "task": 1, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 13.5, + "agent_wall_sec": 482.912465, + "input_tokens": 1002736, + "cache_tokens": 959232, + "output_tokens": 12956, + "cost_usd": 2.042072, + "final_pass": 1 + }, + { + "task": 2, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 9.93, + "agent_wall_sec": 155.935848, + "input_tokens": 324595, + "cache_tokens": 298624, + "output_tokens": 3837, + "cost_usd": 0.750184, + "final_pass": 1 + }, + { + "task": 3, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 6.13, + "agent_wall_sec": 149.740855, + "input_tokens": 213416, + "cache_tokens": 189056, + "output_tokens": 4120, + "cost_usd": 0.6386560000000001, + "final_pass": 1 + }, + { + "task": 4, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 2.66, + "agent_wall_sec": 143.797886, + "input_tokens": 342849, + "cache_tokens": 317312, + "output_tokens": 3826, + "cost_usd": 0.763982, + "final_pass": 1 + }, + { + "task": 5, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 9.62, + "agent_wall_sec": 175.179708, + "input_tokens": 482861, + "cache_tokens": 449920, + "output_tokens": 4424, + "cost_usd": 1.0005300000000001, + "final_pass": 1 + }, + { + "task": 6, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 108.47, + "agent_wall_sec": 221.083834, + "input_tokens": 692897, + "cache_tokens": 648320, + "output_tokens": 4174, + "cost_usd": 1.30279, + "final_pass": 1 + }, + { + "task": 7, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 22.93, + "agent_wall_sec": 347.909045, + "input_tokens": 1118515, + "cache_tokens": 1065728, + "output_tokens": 8682, + "cost_usd": 2.027698, + "final_pass": 1 + }, + { + "task": 8, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 16.36, + "agent_wall_sec": 255.345866, + "input_tokens": 670097, + "cache_tokens": 627072, + "output_tokens": 6625, + "cost_usd": 1.3885720000000001, + "final_pass": 1 + }, + { + "task": 9, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 8.77, + "agent_wall_sec": 211.388948, + "input_tokens": 579602, + "cache_tokens": 548736, + "output_tokens": 5188, + "cost_usd": 1.1167960000000001, + "final_pass": 1 + }, + { + "task": 10, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 69.88, + "agent_wall_sec": 175.292938, + "input_tokens": 522941, + "cache_tokens": 492032, + "output_tokens": 3897, + "cost_usd": 0.9959720000000001, + "final_pass": 1 + }, + { + "task": 11, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 102.1, + "agent_wall_sec": 235.330635, + "input_tokens": 567128, + "cache_tokens": 526336, + "output_tokens": 5442, + "cost_usd": 1.206356, + "final_pass": 1 + }, + { + "task": 12, + "raw_reward": 1, + "functional": 1, + "static": 1, + "audit": 1, + "runtime_sec": 4.31, + "agent_wall_sec": 129.212956, + "input_tokens": 250761, + "cache_tokens": 215936, + "output_tokens": 3385, + "cost_usd": 0.733436, + "final_pass": 1 + } + ] + } + ], + "paired_astra": { + "source": "https://github.com/sxzgroup/ORBIT-Q/pull/29#issuecomment-5549123804", + "expert_commit": "2187118fecc0e275efc0763c676baa8e2df57108", + "astra_commit": "b28f53770c606d9c3c499c5ea10613eac197c1c2", + "image": "sha256:ce64a50ab90eaee930d981b7397903bc4baca54f32b8c88f5fa6ad2c40b9f5d7", + "cpus": 6, + "memory_bytes": 10737418240, + "processor": "Apple M2", + "summary": { + "paired_functional_passes": 11, + "expert_functional_passes": 11, + "astra_functional_passes": 12, + "geometric_mean_ratio": 0.5649749608970326, + "astra_faster_tasks": [ + 1, + 4, + 5, + 7, + 9, + 11, + 12 + ], + "expert_sum_sec": 590.11, + "astra_sum_sec": 424.49999999999994, + "ratio_of_sums": 0.7193574079408923 + }, + "tasks": [ + { + "task": 1, + "expert_sec": 58.59, + "astra_sec": 13.68, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.2334869431643625 + }, + { + "task": 2, + "expert_sec": 4.15, + "astra_sec": 9.84, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 2.3710843373493975 + }, + { + "task": 3, + "expert_sec": 3.82, + "astra_sec": 5.1, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 1.3350785340314135 + }, + { + "task": 4, + "expert_sec": 13.13, + "astra_sec": 2.33, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.17745620715917745 + }, + { + "task": 5, + "expert_sec": 91.87, + "astra_sec": 5.26, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.05725481658865788 + }, + { + "task": 6, + "expert_sec": 44.26, + "astra_sec": 104.66, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 2.3646633529145955 + }, + { + "task": 7, + "expert_sec": 147.46, + "astra_sec": 21.22, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.1439034314390343 + }, + { + "task": 8, + "expert_sec": null, + "astra_sec": 16.31, + "expert_functional_pass": false, + "astra_functional_pass": true, + "expert_status": "OOM", + "astra_status": "PASS", + "astra_over_expert": null + }, + { + "task": 9, + "expert_sec": 23.64, + "astra_sec": 8.63, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.3650592216582065 + }, + { + "task": 10, + "expert_sec": 21.51, + "astra_sec": 105.57, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 4.907949790794978 + }, + { + "task": 11, + "expert_sec": 171.11, + "astra_sec": 143.04, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.8359534802174039 + }, + { + "task": 12, + "expert_sec": 10.57, + "astra_sec": 5.17, + "expert_functional_pass": true, + "astra_functional_pass": true, + "expert_status": "PASS", + "astra_status": "PASS", + "astra_over_expert": 0.489120151371807 + } + ], + "restoration": { + "[REDACTED]_p01": "true_p01", + "[REDACTED]_p10": "true_p10" + } + } +} diff --git a/paper/source_data/task_outcomes.csv b/paper/source_data/task_outcomes.csv new file mode 100644 index 0000000..5971695 --- /dev/null +++ b/paper/source_data/task_outcomes.csv @@ -0,0 +1,109 @@ +model,task,raw_reward,final_pass,functional,static,audit,runtime_sec +gpt56_sol_high,1,0,0,1,1,0,107.51 +gpt56_sol_high,2,1,1,1,1,1,10.65 +gpt56_sol_high,3,1,1,1,1,1,9.84 +gpt56_sol_high,4,1,1,1,1,1,14.14 +gpt56_sol_high,5,1,1,1,1,1,124.92 +gpt56_sol_high,6,1,1,1,1,1,23.39 +gpt56_sol_high,7,1,1,1,1,1,155.52 +gpt56_sol_high,8,0,0,1,1,0,55.48 +gpt56_sol_high,9,1,1,1,1,1,87.31 +gpt56_sol_high,10,1,1,1,1,1,68.68 +gpt56_sol_high,11,1,1,1,1,1,100.41 +gpt56_sol_high,12,1,1,1,1,1,12.85 +gpt56_sol_ultra,1,0,0,1,1,0,127.87 +gpt56_sol_ultra,2,1,1,1,1,1,14.32 +gpt56_sol_ultra,3,1,1,1,1,1,18.45 +gpt56_sol_ultra,4,1,1,1,1,1,4.96 +gpt56_sol_ultra,5,1,1,1,1,1,75.51 +gpt56_sol_ultra,6,1,1,1,1,1,152.16 +gpt56_sol_ultra,7,0,1,1,1,0,84.99 +gpt56_sol_ultra,8,1,0,1,1,1,60.94 +gpt56_sol_ultra,9,1,1,1,1,1,7.18 +gpt56_sol_ultra,10,1,1,1,1,1,71.27 +gpt56_sol_ultra,11,1,1,1,1,1,120.07 +gpt56_sol_ultra,12,1,1,1,1,1,8.61 +gpt56_terra_high,1,0,0,0,1,0, +gpt56_terra_high,2,1,1,1,1,1,47.00 +gpt56_terra_high,3,1,1,1,1,1,34.29 +gpt56_terra_high,4,1,1,1,1,1,14.62 +gpt56_terra_high,5,1,1,1,1,1,95.16 +gpt56_terra_high,6,1,1,1,1,1,116.18 +gpt56_terra_high,7,1,1,1,1,1,8.91 +gpt56_terra_high,8,0,0,0,0,0, +gpt56_terra_high,9,1,1,1,1,1,94.28 +gpt56_terra_high,10,0,0,0,1,0,21.53 +gpt56_terra_high,11,1,1,1,1,1,92.81 +gpt56_terra_high,12,1,1,1,1,1,14.07 +gpt56_luna_high,1,0,0,1,1,0,9.84 +gpt56_luna_high,2,1,1,1,1,1,21.30 +gpt56_luna_high,3,1,1,1,1,1,7.69 +gpt56_luna_high,4,0,0,1,1,0,9.45 +gpt56_luna_high,5,1,1,1,1,1,313.09 +gpt56_luna_high,6,1,1,1,1,1,20.47 +gpt56_luna_high,7,1,1,1,1,1,170.18 +gpt56_luna_high,8,1,0,1,1,1,53.18 +gpt56_luna_high,9,1,1,1,1,1,89.42 +gpt56_luna_high,10,1,1,1,1,1,436.64 +gpt56_luna_high,11,1,1,1,1,1,319.79 +gpt56_luna_high,12,1,1,1,1,1,19.34 +deepseek_v4_flash_high,1,0,0,0,0,0, +deepseek_v4_flash_high,2,0,0,0,0,0, +deepseek_v4_flash_high,3,1,1,1,1,1,70.04 +deepseek_v4_flash_high,4,1,1,1,1,1,19.46 +deepseek_v4_flash_high,5,1,1,1,1,1,83.79 +deepseek_v4_flash_high,6,0,0,1,1,0,54.79 +deepseek_v4_flash_high,7,0,0,1,1,0,194.15 +deepseek_v4_flash_high,8,0,0,0,0,0, +deepseek_v4_flash_high,9,0,0,0,0,0, +deepseek_v4_flash_high,10,1,1,1,1,1,20.50 +deepseek_v4_flash_high,11,0,0,0,0,0, +deepseek_v4_flash_high,12,1,1,1,1,1,21.86 +grok45_high,1,0,0,0,0,0, +grok45_high,2,1,1,1,1,1,62.29 +grok45_high,3,1,1,1,1,1,13.01 +grok45_high,4,0,0,0,0,0, +grok45_high,5,1,1,1,1,1,94.38 +grok45_high,6,1,1,1,1,1,38.08 +grok45_high,7,1,1,1,1,1,121.61 +grok45_high,8,0,0,0,0,0, +grok45_high,9,1,1,1,1,1,77.90 +grok45_high,10,1,1,1,1,1,89.31 +grok45_high,11,0,0,0,0,0, +grok45_high,12,1,1,1,1,1,11.13 +deepseek_v4_pro_high,1,0,0,0,0,0, +deepseek_v4_pro_high,2,1,1,1,1,1,21.66 +deepseek_v4_pro_high,3,0,0,0,0,0, +deepseek_v4_pro_high,4,1,1,1,1,1,20.23 +deepseek_v4_pro_high,5,0,0,1,1,0,77.39 +deepseek_v4_pro_high,6,1,1,1,1,1,36.23 +deepseek_v4_pro_high,7,1,1,1,1,1,97.41 +deepseek_v4_pro_high,8,0,0,0,1,0,18.99 +deepseek_v4_pro_high,9,0,0,1,1,0,98.91 +deepseek_v4_pro_high,10,1,1,1,1,1,54.27 +deepseek_v4_pro_high,11,0,0,0,0,0, +deepseek_v4_pro_high,12,0,0,0,0,0, +grok46_high,1,0,0,0,0,0, +grok46_high,2,1,1,1,1,1,12.36 +grok46_high,3,1,1,1,1,1,5.91 +grok46_high,4,1,1,1,1,1,17.61 +grok46_high,5,0,0,1,1,0,72.2 +grok46_high,6,1,1,1,1,1,24.95 +grok46_high,7,1,1,1,1,1,36.03 +grok46_high,8,0,0,1,1,0,47.47 +grok46_high,9,1,1,1,1,1,58.45 +grok46_high,10,1,1,1,1,1,65.01 +grok46_high,11,1,1,1,1,1,180.41 +grok46_high,12,1,1,1,1,1,14.92 +astra_high,1,1,1,1,1,1,13.5 +astra_high,2,1,1,1,1,1,9.93 +astra_high,3,1,1,1,1,1,6.13 +astra_high,4,1,1,1,1,1,2.66 +astra_high,5,1,1,1,1,1,9.62 +astra_high,6,1,1,1,1,1,108.47 +astra_high,7,1,1,1,1,1,22.93 +astra_high,8,1,1,1,1,1,16.36 +astra_high,9,1,1,1,1,1,8.77 +astra_high,10,1,1,1,1,1,69.88 +astra_high,11,1,1,1,1,1,102.1 +astra_high,12,1,1,1,1,1,4.31 diff --git a/paper/test_updated_results.py b/paper/test_updated_results.py new file mode 100644 index 0000000..dd1bab3 --- /dev/null +++ b/paper/test_updated_results.py @@ -0,0 +1,72 @@ +"""Regression checks for the archived paper-data contract.""" +import unittest + +from verify_updated_results import load, validate + + +class PaperDataTests(unittest.TestCase): + def setUp(self): + self.tables = load() + + def test_valid_archive(self): + validate(self.tables) + + def test_missing_task(self): + self.tables["task_outcomes"].pop() + with self.assertRaises(ValueError): + validate(self.tables) + + def test_changed_astra_outcome(self): + self.tables["task_outcomes"][-5]["final_pass"] = "0" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_wrong_cost(self): + self.tables["paper_agent_axis"][-2]["cost_usd"] = "0" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_mixed_runtime_baseline(self): + self.tables["paper_agent_axis"][-1]["gm_slowdown"] = "0.564975" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_oom_filled_with_reference(self): + self.tables["astra_expert_paired"][7]["expert_sec"] = "25.05" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_inverted_speedup(self): + row = self.tables["astra_expert_paired"][0] + row["expert_over_astra"] = row["astra_over_expert"] + with self.assertRaises(ValueError): + validate(self.tables) + + def test_wrong_replacement_value(self): + self.tables["astra_expert_runtime"][7]["expert_reference_sec"] = "25.05" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_estimate_labeled_measured(self): + self.tables["astra_expert_runtime"][7]["expert_reference_basis"] = "measured" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_unrelated_pair_changed(self): + self.tables["astra_expert_runtime"][0]["expert_reference_sec"] = "27.22" + with self.assertRaises(ValueError): + validate(self.tables) + + def test_old_aggregate_retained(self): + self.tables["astra_adjustment"]["summary"]["geometric_mean_ratio"] = 0.564975 + with self.assertRaises(ValueError): + validate(self.tables) + + def test_estimate_counted_as_measured_pair(self): + self.tables["astra_adjustment"]["summary"]["measured_pair_count"] = 12 + with self.assertRaises(ValueError): + validate(self.tables) + + +if __name__ == "__main__": + unittest.main() diff --git a/paper/updated_figures/astra_expert_paired_runtime.pdf b/paper/updated_figures/astra_expert_paired_runtime.pdf new file mode 100644 index 0000000..ec1b147 Binary files /dev/null and b/paper/updated_figures/astra_expert_paired_runtime.pdf differ diff --git a/paper/updated_figures/astra_expert_paired_runtime.png b/paper/updated_figures/astra_expert_paired_runtime.png new file mode 100644 index 0000000..5935ef7 Binary files /dev/null and b/paper/updated_figures/astra_expert_paired_runtime.png differ diff --git a/paper/updated_figures/expert_optimization_updated_overview.pdf b/paper/updated_figures/expert_optimization_updated_overview.pdf new file mode 100644 index 0000000..7374495 Binary files /dev/null and b/paper/updated_figures/expert_optimization_updated_overview.pdf differ diff --git a/paper/updated_figures/expert_optimization_updated_overview.png b/paper/updated_figures/expert_optimization_updated_overview.png new file mode 100644 index 0000000..33d028a Binary files /dev/null and b/paper/updated_figures/expert_optimization_updated_overview.png differ diff --git a/paper/updated_figures/fig1c_updated_agent_framework_matrix.pdf b/paper/updated_figures/fig1c_updated_agent_framework_matrix.pdf new file mode 100644 index 0000000..e313c23 Binary files /dev/null and b/paper/updated_figures/fig1c_updated_agent_framework_matrix.pdf differ diff --git a/paper/updated_figures/fig1c_updated_agent_framework_matrix.png b/paper/updated_figures/fig1c_updated_agent_framework_matrix.png new file mode 100644 index 0000000..c67f8ca Binary files /dev/null and b/paper/updated_figures/fig1c_updated_agent_framework_matrix.png differ diff --git a/paper/updated_figures/fig2b_updated_agent_axis.pdf b/paper/updated_figures/fig2b_updated_agent_axis.pdf new file mode 100644 index 0000000..5036f49 Binary files /dev/null and b/paper/updated_figures/fig2b_updated_agent_axis.pdf differ diff --git a/paper/updated_figures/fig2b_updated_agent_axis.png b/paper/updated_figures/fig2b_updated_agent_axis.png new file mode 100644 index 0000000..a6f23e8 Binary files /dev/null and b/paper/updated_figures/fig2b_updated_agent_axis.png differ diff --git a/paper/updated_figures/fig4_updated_agent_framework_resources.pdf b/paper/updated_figures/fig4_updated_agent_framework_resources.pdf new file mode 100644 index 0000000..5787b15 Binary files /dev/null and b/paper/updated_figures/fig4_updated_agent_framework_resources.pdf differ diff --git a/paper/updated_figures/fig4_updated_agent_framework_resources.png b/paper/updated_figures/fig4_updated_agent_framework_resources.png new file mode 100644 index 0000000..c1b1921 Binary files /dev/null and b/paper/updated_figures/fig4_updated_agent_framework_resources.png differ diff --git a/paper/verify_updated_results.py b/paper/verify_updated_results.py new file mode 100644 index 0000000..48b80b8 --- /dev/null +++ b/paper/verify_updated_results.py @@ -0,0 +1,198 @@ +#!/usr/bin/env python3 +"""Validate paper tables without plotting dependencies or network access.""" +import argparse +import csv +import hashlib +import json +import math +from pathlib import Path +import subprocess + +DATA = Path(__file__).resolve().parent / "source_data" +EXPECTED = dict(zip([ + "gpt56_sol_high", "gpt56_sol_ultra", "gpt56_terra_high", "gpt56_luna_high", + "deepseek_v4_flash_high", "grok45_high", "deepseek_v4_pro_high", + "grok46_high", "astra_high"], [10, 10, 9, 9, 5, 8, 5, 9, 12])) +AGENT_KEYS = dict(zip(list(EXPECTED)[:4], [ + "sol56_high", "sol56_ultra", "terra56_high", "luna56_high"])) + + +def close(actual, expected, label, tolerance=5.1e-7): + if not math.isclose(float(actual), expected, rel_tol=0, abs_tol=tolerance): + raise ValueError(f"{label}: {actual} != {expected}") + + +def require(condition, label): + if not condition: + raise ValueError(label) + + +def load(data=DATA): + result = {} + for name in ("paper_agent_axis", "benchmark_models", "task_outcomes", + "paper_expert_runtimes", "astra_expert_paired", "astra_expert_runtime"): + with (data / f"{name}.csv").open(newline="") as handle: + result[name] = list(csv.DictReader(handle)) + result["evidence"] = json.loads((data / "recent_pr_evidence.json").read_text()) + result["astra_adjustment"] = json.loads((data / "astra_expert_adjustment.json").read_text()) + require(hashlib.sha256((data / "astra_expert_paired.csv").read_bytes()).hexdigest() + == result["astra_adjustment"]["measured_source_sha256"], "Raw paired source hash") + return result + + +def validate(tables, check_git_sources=False): + agent_rows = tables["paper_agent_axis"] + agents = {r["key"]: r for r in agent_rows} + models = {r["model"]: r for r in tables["benchmark_models"]} + tasks = tables["task_outcomes"] + require(len(agent_rows) == len(agents) == 14, "14 unique agent rows required") + require(len(tables["benchmark_models"]) == len(models) == 9, "9 models required") + require(set(models) == set(EXPECTED), "Unexpected model keys") + require(len(tasks) == len({(r["model"], r["task"]) for r in tasks}) == 108, + "108 unique task rows required") + refs = {int(r["task"]): float(r["expert_tc_runtime_sec"]) + for r in tables["paper_expert_runtimes"]} + require(set(refs) == set(range(1, 13)), "12 expert references required") + for model, passes in EXPECTED.items(): + rows = [r for r in tasks if r["model"] == model] + require({int(r["task"]) for r in rows} == set(range(1, 13)), model) + require(all(r["final_pass"] in ("0", "1") for r in rows), model) + close(sum(int(r["final_pass"]) for r in rows), passes, model) + axis = agents[AGENT_KEYS.get(model, model)] + for aggregate in (models[model], axis): + close(aggregate["passes"], passes, model) + close(aggregate["failures"], 12-passes, model) + ratios = [float(r["runtime_sec"])/refs[int(r["task"])] + for r in rows if r["final_pass"] == "1"] + require(all(v > 0 for v in ratios), f"{model}: positive passed runtimes") + gm = math.exp(sum(map(math.log, ratios))/len(ratios)) + close(axis["gm_slowdown"], gm, model) + close(models[model]["gm_slowdown"], gm, model, tolerance=0.00051) + close(axis["wall_total_sec"], float(axis["wall_passed_sec"]) + + float(axis["wall_failed_sec"]), model, tolerance=0.001) + task08 = next(r for r in rows if int(r["task"]) == 8) + require(task08["final_pass"] == ("1" if model == "astra_high" else "0"), + f"{model}: preserve Task 08 source outcome") + + evidence = tables["evidence"] + require({m["key"] for m in evidence["models"]} == set(list(EXPECTED)[-3:]), + "Expected evidence for three recent campaigns") + for model in evidence["models"]: + key, rows = model["key"], model["tasks"] + require(len(rows) == 12, f"{key}: evidence task count") + source = model["source"] + if check_git_sources: + raw = subprocess.check_output(["git", "show", f"{source['commit']}:{source['path']}"], + cwd=DATA.parent.parent) + require(hashlib.sha256(raw).hexdigest() == source["sha256"], + f"{key}: source SHA-256 mismatch") + if key == "grok46_high": + close(agents[key]["cost_usd"], json.loads(raw)["agent_usage"]["list_price_cost_usd"], key) + indexed = {int(r["task"]): r for r in tasks if r["model"] == key} + for record in rows: + row = indexed[record["task"]] + for field in ("raw_reward", "final_pass", "functional", "static", "audit", "runtime_sec"): + if record[field] is None: + require(row[field] == "", f"{key}: missing {field} must stay missing") + else: + close(row[field], record[field], f"{key}/{record['task']}/{field}") + wall = sum(r["agent_wall_sec"] for r in rows) + passed_wall = sum(r["agent_wall_sec"] for r in rows if r["final_pass"]) + prompt, cache, output = [sum(r[field] for r in rows) + for field in ("input_tokens", "cache_tokens", "output_tokens")] + # Archived Grok 4.6 accounting uses these recorded per-million rates. + cost = ((prompt-cache)*2 + cache*0.5 + output*6)/1e6 if key == "grok46_high" else sum(r["cost_usd"] for r in rows) + values = {"wall_total_sec": wall, "wall_passed_sec": passed_wall, + "wall_failed_sec": wall-passed_wall, "total_tokens_m": (prompt+output)/1e6, + "noncache_prompt_m": (prompt-cache)/1e6, "cache_prompt_m": cache/1e6, + "output_tokens_m": output/1e6, "cost_usd": cost} + for field, value in values.items(): + close(agents[key][field], value, f"{key}/{field}") + values = {"agent_wall_min": wall/60, "total_tokens_m": (prompt+output)/1e6, + "input_tokens_m": prompt/1e6, "cache_tokens_m": cache/1e6, + "output_tokens_m": output/1e6, "cost_usd": cost, + "cost_per_valid_usd": cost/EXPECTED[key], + "solve_time_per_valid_min": wall/60/EXPECTED[key]} + for field, value in values.items(): + close(models[key][field], value, f"{key}/{field}") + + paired = tables["astra_expert_paired"] + require(len(paired) == 12 and {int(r["task"]) for r in paired} == set(range(1, 13)), + "12 paired records required") + by_task = {r["task"]: r for r in evidence["paired_astra"]["tasks"]} + valid = [] + for row in paired: + record = by_task[int(row["task"])] + for field in ("expert_status", "astra_status"): + require(row[field] == record[field], f"Paired status: {field}") + close(row["astra_sec"], record["astra_sec"], "Paired Astra runtime") + if record["expert_sec"] is None: + require(row["included_in_aggregate"] == "0" and + all(row[f] == "" for f in ("expert_sec", "astra_over_expert", "expert_over_astra")), + "OOM must remain missing and excluded") + else: + require(row["included_in_aggregate"] == "1", "Completed pair must be included") + expert, astra = float(row["expert_sec"]), float(row["astra_sec"]) + close(expert, record["expert_sec"], "Paired expert runtime") + close(row["astra_over_expert"], astra/expert, "Paired ratio") + close(row["expert_over_astra"], expert/astra, "Paired speedup") + valid.append((expert, astra)) + summary = evidence["paired_astra"]["summary"] + close(len(valid), summary["paired_functional_passes"], "Paired count") + close(sum(e for e, a in valid), summary["expert_sum_sec"], "Paired expert total") + close(sum(a for e, a in valid), summary["astra_sum_sec"], "Paired Astra total") + close(sum(a for e, a in valid)/sum(e for e, a in valid), summary["ratio_of_sums"], "Ratio of sums") + gm = math.exp(sum(math.log(a/e) for e, a in valid)/len(valid)) + close(gm, summary["geometric_mean_ratio"], "Paired GM") + close(sum(a < e for e, a in valid), len(summary["astra_faster_tasks"]), "Faster tasks") + require(abs(gm-float(agents["astra_high"]["gm_slowdown"])) > 0.4, + "Paired and fixed-paper baselines must not be mixed") + + adjusted = tables["astra_expert_runtime"] + adjustment = tables["astra_adjustment"] + require(len(adjusted) == 12 and {int(r["task"]) for r in adjusted} == set(range(1, 13)), + "12 adjusted runtime records required") + assumption = adjustment["task08"] + close(assumption["expert_reference_sec"], 50, "Task 08 approved reference") + close(assumption["reviewer_proposed_sec"], + assumption["published_reference_sec"]*assumption["reviewer_scale_factor"], + "Reviewer reference scaling") + require(assumption["basis"] == "estimated" and assumption["original_expert_status"] == "OOM", + "Task 08 estimate must not become a successful measurement") + for row in adjusted: + task = int(row["task"]) + record = by_task[task] + require(row["expert_reference_basis"] == ("estimated" if task == 8 else "measured"), + "Only Task 08 uses an estimated expert reference") + close(row["astra_sec"], record["astra_sec"], "Adjusted Astra measurement") + expert = 50 if task == 8 else record["expert_sec"] + astra = record["astra_sec"] + close(row["expert_reference_sec"], expert, "Adjusted expert reference") + close(row["astra_over_expert"], astra/expert, "Adjusted ratio") + close(row["expert_over_astra"], expert/astra, "Adjusted speedup") + require(row["source"] == (assumption["source"] if task == 8 else evidence["paired_astra"]["source"]), + "Adjusted source provenance") + summary = adjustment["summary"] + close(summary["task_count"], 12, "Adjusted count") + close(summary["measured_pair_count"], 11, "Measured pair count") + close(summary["estimated_reference_count"], 1, "Estimated count") + expert_total = sum(float(r["expert_reference_sec"]) for r in adjusted) + astra_total = sum(float(r["astra_sec"]) for r in adjusted) + adjusted_gm = math.exp(sum(math.log(float(r["astra_sec"])/float(r["expert_reference_sec"])) + for r in adjusted)/len(adjusted)) + for field, value in {"expert_sum_sec": expert_total, "astra_sum_sec": astra_total, + "ratio_of_sums": astra_total/expert_total, + "geometric_mean_ratio": adjusted_gm, + "geometric_mean_speedup": 1/adjusted_gm}.items(): + close(summary[field], value, f"Adjusted {field}") + require(summary["astra_faster_tasks"] == [int(r["task"]) for r in adjusted + if float(r["astra_sec"]) < float(r["expert_reference_sec"])], "Adjusted faster tasks") + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--check-git-sources", action="store_true", + help="Also check pinned source objects when available locally") + args = parser.parse_args() + validate(load(), args.check_git_sources) + print("PASS: 14 agent rows, 108 outcomes, three source campaigns; Astra has 11 measured pairs + one estimated reference")