Skip to content

Commit 8e8bfcb

Browse files
committed
feat(eval): canonicalize repeat policy schema
1 parent 23567a2 commit 8e8bfcb

32 files changed

Lines changed: 2750 additions & 751 deletions

File tree

‎README.md‎

Lines changed: 13 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -19,7 +19,7 @@ Test AI targets on real repo tasks and measure what actually works.
1919
- **Workspace / fixtures / graders** are task-owned context: repos, setup scripts, files, fixtures, isolation, deterministic checks, and LLM grading prompts.
2020
- **Target** is the system under test: an agent, provider, gateway, replay target, CLI wrapper, transcript provider, or future app/service wrapper. Each eval selects one `target`, either by name from `targets.yaml` or with an eval-local target object.
2121
- **Experiment** is the run/result grouping label being measured over that corpus, such as `backend-with-skills` or `backend-without-skills`.
22-
- **Run controls** configure repeats, exits, timeouts, budgets, thresholds, and completion hooks with top-level fields such as `runs`, `early_exit`, `timeout_seconds`, `budget_usd`, `threshold`, and `on_run_complete`.
22+
- **Run controls** configure repeats, timeouts, budgets, thresholds, and completion hooks with fields such as `repeat`, `timeout_seconds`, `budget_usd`, `threshold`, and `on_run_complete`.
2323
- **Run** is one concrete execution of an experiment against a resolved target that writes portable artifacts for readers such as Dashboard, compare, and trend.
2424

2525
```mermaid
@@ -60,8 +60,10 @@ agentv init
6060
description: Code generation quality
6161
experiment: backend-with-skills
6262
target: copilot-sdk
63-
runs: 3
64-
early_exit: false
63+
repeat:
64+
count: 3
65+
strategy: pass_any
66+
early_exit: false
6567
timeout_seconds: 600
6668
threshold: 0.8
6769
budget_usd: 5
@@ -91,7 +93,9 @@ target:
9193
extends: codex-gpt5
9294
model: gpt-5.1
9395
reasoning_effort: high
94-
runs: 2
96+
repeat:
97+
count: 2
98+
strategy: pass_any
9599
timeout_seconds: 900
96100
threshold: 0.85
97101

@@ -186,8 +190,11 @@ export default defineEval({
186190
extends: 'copilot-sdk',
187191
model: 'claude-sonnet-4.6',
188192
},
189-
runs: 3,
190-
earlyExit: false,
193+
repeat: {
194+
count: 3,
195+
strategy: 'pass_any',
196+
earlyExit: false,
197+
},
191198
timeoutSeconds: 600,
192199
threshold: 0.8,
193200
budgetUsd: 5,

‎apps/cli/src/commands/eval/run-eval.ts‎

Lines changed: 2 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -803,17 +803,10 @@ function buildExperimentTrialsConfig(experiment: ExperimentConfig): TrialsConfig
803803
...(experiment.repeat.costLimitUsd !== undefined && {
804804
costLimitUsd: experiment.repeat.costLimitUsd,
805805
}),
806-
...(experiment.earlyExit !== undefined && { earlyExit: experiment.earlyExit }),
806+
...(experiment.repeat.earlyExit !== undefined && { earlyExit: experiment.repeat.earlyExit }),
807807
};
808808
}
809-
if (!experiment.runs || experiment.runs <= 1) {
810-
return undefined;
811-
}
812-
return {
813-
count: experiment.runs,
814-
strategy: 'pass_at_k',
815-
...(experiment.earlyExit !== undefined && { earlyExit: experiment.earlyExit }),
816-
};
809+
return undefined;
817810
}
818811

819812
type EffectiveRunPolicy = {

‎apps/cli/test/commands/eval/artifact-writer.test.ts‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -177,7 +177,7 @@ describe('buildGradingArtifact', () => {
177177
},
178178
],
179179
aggregation: {
180-
strategy: 'pass_at_k',
180+
strategy: 'pass_any',
181181
passedAttempts: 1,
182182
totalAttempts: 2,
183183
},
@@ -202,7 +202,7 @@ describe('buildGradingArtifact', () => {
202202
},
203203
]);
204204
expect(grading.aggregation).toEqual({
205-
strategy: 'pass_at_k',
205+
strategy: 'pass_any',
206206
passed_attempts: 1,
207207
total_attempts: 2,
208208
});

‎apps/cli/test/eval.integration.test.ts‎

Lines changed: 11 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -611,7 +611,10 @@ describe('agentv eval CLI', () => {
611611
'timeout_seconds: 12',
612612
'threshold: 0.8',
613613
'budget_usd: 3',
614-
'runs: 2',
614+
'repeat:',
615+
' count: 2',
616+
' strategy: pass_any',
617+
' early_exit: true',
615618
'tests:',
616619
' - include: sample.test.yaml',
617620
' type: suite',
@@ -623,6 +626,7 @@ describe('agentv eval CLI', () => {
623626
' repeat:',
624627
' count: 3',
625628
' strategy: pass_all',
629+
' early_exit: true',
626630
'',
627631
].join('\n'),
628632
'utf8',
@@ -646,6 +650,7 @@ describe('agentv eval CLI', () => {
646650
trials: {
647651
count: 3,
648652
strategy: 'pass_all',
653+
earlyExit: true,
649654
},
650655
});
651656

@@ -655,7 +660,11 @@ describe('agentv eval CLI', () => {
655660
expect(benchmark.metadata?.experiment).toBe('native-exp');
656661
expect(benchmark.metadata?.experiment_config).toMatchObject({
657662
target: 'codex-target',
658-
runs: 2,
663+
repeat: {
664+
count: 2,
665+
strategy: 'pass_any',
666+
early_exit: true,
667+
},
659668
threshold: 0.8,
660669
budget_usd: 3,
661670
timeout_seconds: 12,

‎apps/web/src/content/docs/docs/evaluation/eval-files.mdx‎

Lines changed: 7 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -5,7 +5,7 @@ sidebar:
55
order: 1
66
---
77

8-
Evaluation files define the test cases, graders, workspace lifecycle, and run controls for an evaluation run. Top-level `experiment` is the run/result grouping label, top-level `target` identifies the system under test, and flat fields such as `runs`, `threshold`, `timeout_seconds`, and `budget_usd` control repeated attempts and gates. Workspace reuse belongs under `workspace.isolation`; Docker/container binding belongs under `workspace.docker`. Install, build, and reset commands belong under `workspace.hooks`; runner-specific setup belongs in the `target` object or `targets.yaml`. AgentV supports two eval data formats: YAML and JSONL.
8+
Evaluation files define the test cases, graders, workspace lifecycle, and run controls for an evaluation run. Top-level `experiment` is the run/result grouping label, top-level `target` identifies the system under test, and fields such as `repeat`, `threshold`, `timeout_seconds`, and `budget_usd` control repeated attempts and gates. Workspace reuse belongs under `workspace.isolation`; Docker/container binding belongs under `workspace.docker`. Install, build, and reset commands belong under `workspace.hooks`; runner-specific setup belongs in the `target` object or `targets.yaml`. AgentV supports two eval data formats: YAML and JSONL.
99

1010
YAML is the canonical portable model. TypeScript helpers, generated fixtures, and Python scripts should lower to the same YAML/JSONL shapes rather than inventing a separate eval contract.
1111
Eval files describe the task, target binding, and run controls. Concurrency is an operator/run setting: pass `--workers` or set `execution.workers` in `agentv.config.*` / `.agentv/config.yaml` instead of authoring `workers` in eval YAML.
@@ -24,7 +24,7 @@ experiment format.
2424
it with `imports.tests`, `tests: ./cases.yaml`, or string shorthand; parent
2525
suite context applies because raw cases do not carry their own suite context.
2626
- A **wrapper eval** is eval YAML that imports one or more suites with
27-
`imports.suites` and binds run controls with top-level `target`, `runs`,
27+
`imports.suites` and binds run controls with top-level `target`, `repeat`,
2828
`threshold`, `timeout_seconds`, and `budget_usd`.
2929
Wrapper evals can live anywhere in the repo. A wrapper that imports suites
3030
with `imports.suites` must not define parent `workspace`; imported suites own
@@ -64,7 +64,9 @@ A wrapper eval stays ordinary eval YAML while choosing a target and run controls
6464
# experiments/refunds-codex.eval.yaml
6565
name: refunds-codex
6666
target: codex-gpt5
67-
runs: 2
67+
repeat:
68+
count: 2
69+
strategy: pass_any
6870

6971
imports:
7072
suites:
@@ -115,13 +117,12 @@ tests:
115117
| `category` | Optional slash-delimited analytics taxonomy path. Overrides the category derived from the eval file path. |
116118
| `target` | Named system under test from `.agentv/targets.yaml` or `--targets` |
117119
| `experiment` | Optional run/result grouping label |
118-
| `runs` | Optional number of attempts per case |
119-
| `early_exit` | Optional early exit for repeated attempts |
120+
| `repeat` | Optional repeat policy with `count`, `strategy`, and `early_exit` |
120121
| `timeout_seconds` | Optional per-case timeout |
121122
| `budget_usd` | Optional suite budget |
122123
| `threshold` | Optional suite quality threshold |
123124
| `workspace` | Suite-level task environment — inline object or string path to an [external workspace file](/docs/guides/workspace-pool/#external-workspace-config). Repo entries declare identity and checkout pins; acquisition is covered in [Workspace Architecture](/docs/guides/workspace-architecture/#repo-provenance-vs-acquisition). |
124-
| `imports` | Optional import groups. `imports.suites` imports full child eval suites with their task context. `imports.tests` imports raw test rows into this file's context. Import entries may use scoped `run:` overrides for `threshold`, `timeout_seconds`, and `budget_usd`. |
125+
| `imports` | Optional import groups. `imports.suites` imports full child eval suites with their task context. `imports.tests` imports raw test rows into this file's context. Import entries may use scoped `run:` overrides for `threshold`, `repeat`, `timeout_seconds`, and `budget_usd`. |
125126
| `tests` | Inline raw tests or a string path to an external raw-case file or directory. Legacy `tests[].include` entries still load with a migration warning; prefer `imports.suites` or `imports.tests`. |
126127
| `assertions` | Suite-level graders appended to each test unless `execution.skip_defaults: true` is set on the test |
127128
| `input` | Suite-level input messages prepended to each test's input unless `execution.skip_defaults: true` is set on the test |

‎apps/web/src/content/docs/docs/evaluation/experiments.mdx‎

Lines changed: 26 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -8,7 +8,7 @@ sidebar:
88
AgentV eval files are the runnable authoring artifact. Use top-level
99
`description` for display metadata, `experiment` as the run/result grouping
1010
label, `target` for the system under test, and flat top-level run controls such
11-
as `runs`, `timeout_seconds`, `budget_usd`, and `threshold`.
11+
as `repeat`, `timeout_seconds`, `budget_usd`, and `threshold`.
1212
Concurrency is outside eval YAML. Use `agentv eval --workers N` or project
1313
config defaults such as `agentv.config.*` / `.agentv/config.yaml`
1414
`execution.workers` for operator-side parallelism.
@@ -21,7 +21,9 @@ target:
2121
extends: codex-gpt5
2222
model: gpt-5.1
2323
reasoning_effort: high
24-
runs: 4
24+
repeat:
25+
count: 4
26+
strategy: pass_any
2527
timeout_seconds: 720
2628
budget_usd: 2.00
2729

@@ -151,7 +153,9 @@ test.run > import run > parent top-level run controls
151153
```yaml
152154
target: agent
153155
threshold: 0.8
154-
runs: 3
156+
repeat:
157+
count: 3
158+
strategy: pass_any
155159
156160
imports:
157161
suites:
@@ -177,10 +181,10 @@ tests:
177181
budget_usd: 0.50
178182
```
179183

180-
Scoped `run:` supports `threshold`, `timeout_seconds`, and `budget_usd` for
181-
public eval authoring. Candidate-changing fields stay parent-level. Workspace
182-
mutation belongs in `workspace.hooks`, and provider-specific setup belongs in
183-
target configuration.
184+
Scoped `run:` supports `threshold`, `repeat`, `timeout_seconds`, and
185+
`budget_usd` for public eval authoring. Candidate-changing fields stay
186+
parent-level. Workspace mutation belongs in `workspace.hooks`, and
187+
provider-specific setup belongs in target configuration.
184188

185189
## Lifecycle Ownership
186190

@@ -194,7 +198,7 @@ target-specific runner state.
194198
| Configure an agent runner or provider variant | `target` object or `targets.yaml` |
195199
| Choose the target | top-level `target` |
196200
| Override the target's default model | `target.model` |
197-
| Configure run count, budget, timeout, threshold | top-level `runs`, `budget_usd`, `timeout_seconds`, `threshold` |
201+
| Configure repeat policy, budget, timeout, threshold | top-level `repeat`, `budget_usd`, `timeout_seconds`, `threshold` |
198202
| Bind an existing local workspace directory | `--workspace-path` or `.agentv/config.local.yaml` |
199203

200204
```yaml
@@ -208,7 +212,9 @@ target:
208212
hooks:
209213
before_each:
210214
command: ["sh", "-c", "cp -R skills \"{{workspace_path}}/.codex/skills\""]
211-
runs: 3
215+
repeat:
216+
count: 3
217+
strategy: pass_any
212218
```
213219

214220
Existing local workspace paths are machine-local bindings: pass
@@ -219,16 +225,21 @@ top-level or case-level `workspace`.
219225

220226
## Repeat Runs
221227

222-
Use top-level `runs` when you want AgentV to try each case more than once:
228+
Use top-level `repeat` when you want AgentV to try each case more than once:
223229

224230
```yaml
225-
runs: 3
231+
repeat:
232+
count: 3
233+
strategy: pass_any
234+
early_exit: false
226235
```
227236

228-
The current repeated-attempt aggregation treats the case as successful when any
229-
attempt succeeds. AgentV does not expose this as `pass@k`; in ML and
230-
code-generation evaluation, pass@k is a statistical metric with different
231-
semantics.
237+
`repeat.strategy` controls verdict aggregation. `pass_any` treats the case as
238+
successful when any completed attempt passes; `pass_all` requires every
239+
completed attempt to pass. `mean` and `confidence_interval` aggregate scores
240+
where supported today. `repeat.early_exit` is only a scheduling and cost
241+
optimization: `pass_any` may stop at the first pass, and `pass_all` may stop at
242+
the first fail. Leave it unset or `false` when you want complete variance data.
232243

233244
## Result Layout
234245

‎examples/features/README.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -81,7 +81,7 @@ Focused examples for specific AgentV capabilities. Find your use case below, the
8181
| Example | Description |
8282
|---------|-------------|
8383
| [benchmark-tooling](benchmark-tooling/) | N-way benchmarking with `agentv compare` over completed runs |
84-
| [trials](trials/) | Configure repeated attempts with `runs` |
84+
| [trials](trials/) | Configure repeated attempts with `repeat` |
8585
| [trial-output-consistency](trial-output-consistency/) | Measure output consistency across trials using pairwise cosine similarity |
8686
| [compare](compare/) | Compare a run against a stored baseline |
8787

‎examples/features/trials/README.md‎

Lines changed: 6 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
# Repeat Runs
22

33
This example keeps the runnable contract in one eval file. Top-level `target`
4-
selects the system under test and top-level `runs` configures repeated attempts.
4+
selects the system under test and top-level `repeat` configures repeated attempts.
55

66
## Files
77

@@ -13,9 +13,12 @@ selects the system under test and top-level `runs` configures repeated attempts.
1313
bun agentv eval examples/features/trials/evals/dataset.eval.yaml
1414
```
1515

16-
Edit `runs` to change how many attempts AgentV makes for each case:
16+
Edit `repeat.count` to change how many attempts AgentV makes for each case:
1717

1818
```yaml
19-
runs: 2
19+
repeat:
20+
count: 2
21+
strategy: pass_any
22+
early_exit: false
2023
budget_usd: 1.00
2124
```
Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,2 @@
1-
{"timestamp":"2026-02-20T21:40:25.928Z","test_id":"capital-knowledge","suite":"dataset","score":1,"target":"default","trials":[{"attempt":0,"score":1,"verdict":"pass"}],"aggregation":{"strategy":"pass_at_k","passed_attempts":1,"total_attempts":1},"assertions":[{"text":"Correctly identifies Canberra as the capital of Australia","passed":true,"evidence":"The candidate answer provides the correct and complete information, fully matching the reference answer."}]}
2-
{"timestamp":"2026-02-20T21:40:26.593Z","test_id":"math-basics","suite":"dataset","score":1,"target":"default","trials":[{"attempt":0,"score":1,"verdict":"pass"}],"aggregation":{"strategy":"pass_at_k","passed_attempts":1,"total_attempts":1},"assertions":[{"text":"Explains step-by-step reasoning","passed":true,"evidence":"The candidate answer breaks down the calculation clearly, explains each step, and arrives at the correct answer, matching the reference reasoning."},{"text":"Splits 15 into 10 and 5 for easier calculation","passed":true},{"text":"Calculates partial products (10\u00d77 and 5\u00d77)","passed":true},{"text":"Arrives at correct final answer (105)","passed":true}]}
1+
{"timestamp":"2026-02-20T21:40:25.928Z","test_id":"capital-knowledge","suite":"dataset","score":1,"target":"default","trials":[{"attempt":0,"score":1,"verdict":"pass"},{"attempt":1,"score":1,"verdict":"pass"}],"aggregation":{"strategy":"pass_any","passed_attempts":2,"total_attempts":2},"assertions":[{"text":"Correctly identifies Canberra as the capital of Australia","passed":true,"evidence":"The candidate answer provides the correct and complete information, fully matching the reference answer."}]}
2+
{"timestamp":"2026-02-20T21:40:26.593Z","test_id":"math-basics","suite":"dataset","score":1,"target":"default","trials":[{"attempt":0,"score":1,"verdict":"pass"},{"attempt":1,"score":1,"verdict":"pass"}],"aggregation":{"strategy":"pass_any","passed_attempts":2,"total_attempts":2},"assertions":[{"text":"Explains step-by-step reasoning","passed":true,"evidence":"The candidate answer breaks down the calculation clearly, explains each step, and arrives at the correct answer, matching the reference reasoning."},{"text":"Splits 15 into 10 and 5 for easier calculation","passed":true},{"text":"Calculates partial products (10\u00d77 and 5\u00d77)","passed":true},{"text":"Arrives at correct final answer (105)","passed":true}]}

‎examples/features/trials/evals/dataset.eval.yaml‎

Lines changed: 5 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,11 +1,14 @@
11
# AgentV Repeat Runs Example
2-
# Demonstrates policy runs for handling LLM non-determinism
2+
# Demonstrates repeat policy for handling LLM non-determinism
33

44
name: trials
55
description: Repeat runs example with 2 attempts configured inline
66

77
target: llm
8-
runs: 2
8+
repeat:
9+
count: 2
10+
strategy: pass_any
11+
early_exit: false
912
budget_usd: 1.00
1013

1114
tests:

0 commit comments

Comments
 (0)