Skip to content

Commit 810864b

Browse files
committed
write file changes diff sidecar
1 parent f6824a0 commit 810864b

12 files changed

Lines changed: 185 additions & 34 deletions

File tree

‎apps/cli/src/commands/eval/artifact-writer.ts‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -105,6 +105,7 @@ export function buildIndexArtifactEntry(
105105
transcriptPath?: string;
106106
transcriptRawPath?: string;
107107
metricsPath?: string;
108+
fileChangesPath?: string;
108109
rawProviderLogPath?: string;
109110
responsePath?: string;
110111
taskBundle?: MaterializedTaskBundlePaths;

‎apps/cli/test/commands/eval/artifact-writer.test.ts‎

Lines changed: 40 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -4,6 +4,7 @@ import { mkdir, readFile, readdir, rm, writeFile } from 'node:fs/promises';
44
import path from 'node:path';
55

66
import {
7+
CANONICAL_FILE_CHANGES_ARTIFACT_PATH,
78
CANONICAL_METRICS_ARTIFACT_PATH,
89
CANONICAL_TRANSCRIPT_ARTIFACT_PATH,
910
type EvalTest,
@@ -290,6 +291,10 @@ describe('buildGradingArtifact', () => {
290291
'@@ -1 +1 @@',
291292
'-old',
292293
'+new',
294+
'--- a/deleted.ts',
295+
'+++ /dev/null',
296+
'@@ -1 +0,0 @@',
297+
'-gone',
293298
].join('\n');
294299

295300
const result = makeResult({ fileChanges: diff });
@@ -298,6 +303,9 @@ describe('buildGradingArtifact', () => {
298303
expect(grading.workspace_changes).toBeDefined();
299304
expect(grading.workspace_changes?.files_created).toBe(1);
300305
expect(grading.workspace_changes?.files_modified).toBe(1);
306+
expect(grading.workspace_changes?.files_deleted).toBe(1);
307+
expect(grading.workspace_changes?.deleted_file_paths).toEqual(['deleted.ts']);
308+
expect(grading.workspace_changes).not.toHaveProperty('diff_summary');
301309
});
302310

303311
it('includes conversation when conversationId present', () => {
@@ -766,6 +774,17 @@ describe('parseJsonlResults', () => {
766774
expect(() => parseJsonlResults(content)).toThrow(/Use "artifact_pointers"/);
767775
});
768776

777+
it('rejects camelCase file changes path rows for the new wire field', () => {
778+
const content = `${JSON.stringify({
779+
test_id: 'file-changes-row',
780+
target: 'codex',
781+
score: 1,
782+
fileChangesPath: 'file-changes-row/run-1/outputs/file_changes.diff',
783+
})}\n`;
784+
785+
expect(() => parseJsonlResults(content)).toThrow(/Use "file_changes_path"/);
786+
});
787+
769788
it('does not treat parsed raw provider log pointers as fresh source artifacts', () => {
770789
const content = `${JSON.stringify({
771790
test_id: 'raw-log-case',
@@ -1398,6 +1417,10 @@ describe('writeArtifactsFromResults', () => {
13981417
'+++ b/src/new.ts',
13991418
'@@ -0,0 +1 @@',
14001419
'+created',
1420+
'--- a/src/gone.ts',
1421+
'+++ /dev/null',
1422+
'@@ -1 +0,0 @@',
1423+
'-deleted',
14011424
].join('\n');
14021425
const results = [
14031426
makeResult({
@@ -1433,6 +1456,21 @@ describe('writeArtifactsFromResults', () => {
14331456
const rowDir = expectRowDir(indexLine, 'summary-case');
14341457

14351458
expect(indexLine?.metrics_path).toBe(`${rowDir}/run-1/metrics.json`);
1459+
expect(indexLine?.file_changes_path).toBe(
1460+
`${rowDir}/run-1/${CANONICAL_FILE_CHANGES_ARTIFACT_PATH}`,
1461+
);
1462+
await expect(
1463+
readFile(
1464+
runArtifactPath(testDir, indexLine, 'run-1', 'outputs', 'file_changes.diff'),
1465+
'utf8',
1466+
),
1467+
).resolves.toBe(fileChanges);
1468+
1469+
const runResult = JSON.parse(
1470+
await readFile(runArtifactPath(testDir, indexLine, 'run-1', 'result.json'), 'utf8'),
1471+
);
1472+
expect(runResult.file_changes_path).toBe('./outputs/file_changes.diff');
1473+
expect(runResult.output_paths.file_changes).toBe('./outputs/file_changes.diff');
14361474

14371475
const summary = MetricsArtifactWireSchema.parse(
14381476
JSON.parse(
@@ -1451,6 +1489,7 @@ describe('writeArtifactsFromResults', () => {
14511489
transcript_path: 'transcript.jsonl',
14521490
grading_path: 'grading.json',
14531491
timing_path: 'timing.json',
1492+
file_changes_path: CANONICAL_FILE_CHANGES_ARTIFACT_PATH,
14541493
});
14551494
expect(summary.source_artifacts).not.toHaveProperty('trace_path');
14561495
await expect(
@@ -1504,6 +1543,7 @@ describe('writeArtifactsFromResults', () => {
15041543
source: 'file_changes',
15051544
});
15061545
expect(summary.metrics.files_created).toEqual(['src/new.ts']);
1546+
expect(summary.metrics.files_deleted).toEqual(['src/gone.ts']);
15071547
expect(summary.metrics.web_fetches).toEqual([
15081548
{
15091549
url: 'https://example.com/spec',

‎apps/cli/test/commands/grade/grade-prepared.test.ts‎

Lines changed: 7 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -197,8 +197,14 @@ describe('agentv grade prepared attempts', () => {
197197
});
198198
expect(typeof row.metadata.prepared_attempt.baseline_commit).toBe('string');
199199

200+
expect(row.file_changes_path).toMatch(/\/run-1\/outputs\/file_changes\.diff$/);
201+
await expect(readFile(path.join(runDir, row.file_changes_path), 'utf8')).resolves.toContain(
202+
'+manual edit',
203+
);
204+
200205
const grading = JSON.parse(await readFile(path.join(runDir, row.grading_path), 'utf8'));
201-
expect(grading.workspace_changes.diff_summary).toContain('+manual edit');
206+
expect(grading.workspace_changes).not.toHaveProperty('diff_summary');
207+
expect(grading.workspace_changes.files_modified).toBeGreaterThanOrEqual(1);
202208
}, 20_000);
203209

204210
it('fails clearly when the prepared manifest is missing', async () => {

‎apps/web/src/content/docs/docs/evaluation/running-evals.mdx‎

Lines changed: 8 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -141,6 +141,7 @@ my-results/
141141
transcript.jsonl
142142
transcript-raw.jsonl
143143
outputs/answer.md
144+
outputs/file_changes.diff # when workspace changes are captured
144145
test/
145146
EVAL.yaml
146147
targets.yaml
@@ -149,13 +150,13 @@ my-results/
149150
```
150151

151152
The `index.jsonl` row links to these generated paths with snake_case fields such
152-
as `result_dir`, `test_dir`, `eval_path`, `targets_path`, `files_path`, and
153-
`graders_path`. Treat those paths as relative to the run directory. When you need
154-
a portable artifact for audit, review, Dashboard inspection, or rerun workflows,
155-
share the generated run directory and its `index.jsonl` manifest. Source-side
156-
case directories are still useful for organizing bulky prompts, fixtures, or
157-
tests while authoring an eval, but they are optional input organization rather
158-
than a separate artifact schema.
153+
as `result_dir`, `test_dir`, `eval_path`, `targets_path`, `files_path`,
154+
`file_changes_path`, and `graders_path`. Treat those paths as relative to the
155+
run directory. When you need a portable artifact for audit, review, Dashboard
156+
inspection, or rerun workflows, share the generated run directory and its
157+
`index.jsonl` manifest. Source-side case directories are still useful for
158+
organizing bulky prompts, fixtures, or tests while authoring an eval, but they
159+
are optional input organization rather than a separate artifact schema.
159160

160161
For the full root layout, per-attempt sidecars, pointer rules, and integration
161162
guidance, use the [Result Artifact Contract](/docs/reference/result-artifacts/).

‎apps/web/src/content/docs/docs/reference/result-artifacts.mdx‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -45,6 +45,7 @@ The default local layout is:
4545
transcript-raw.jsonl
4646
outputs/
4747
answer.md
48+
file_changes.diff
4849
run-2/
4950
result.json
5051
grading.json
@@ -54,6 +55,7 @@ The default local layout is:
5455
transcript-raw.jsonl
5556
outputs/
5657
answer.md
58+
file_changes.diff
5759
```
5860

5961
The `<experiment>` and `<run_id>` directories are storage allocation. They help
@@ -83,6 +85,7 @@ query.
8385
| `result.json` | Compact per-attempt manifest for one attempt directory. | Loading one attempt without scanning the whole run index. |
8486
| `grading.json` | Grader outputs, assertions, rubric evidence, execution-metric grader facts, and scoring provenance. | Explaining why a row passed or failed. |
8587
| `metrics.json` | Derived executor behavior summary, such as tool calls, files touched, shell commands, errors, turns, and output sizes. | Dashboard behavior views, metric-style graders, adapter projections, and lightweight analysis. |
88+
| `outputs/file_changes.diff` | Full unified diff of workspace file changes when file changes are captured. | Human review and external artifact inspection; LLM and code graders still receive the same full diff through `file_changes`. |
8689
| `timing.json` | Duration, token usage, cost usage, and source labels such as `provider_reported`, `token_estimated`, `aggregate`, or `unavailable`. | Cost/latency reporting and provider-accounting audits. |
8790
| `transcript.jsonl` | AgentV-normalized transcript/timeline rows. | Portable human review, replay, transcript-aware graders, and tool-trajectory analysis. |
8891
| `transcript-raw.jsonl` | Native provider or harness evidence when available. | Parser debugging, forensic review, and preserving source bytes without making provider schemas public AgentV fields. |
@@ -132,6 +135,7 @@ Example row:
132135
"transcript_raw_path": "refund-eligibility--4f9a7c2d1b6e/run-1/transcript-raw.jsonl",
133136
"output_path": "refund-eligibility--4f9a7c2d1b6e/run-1/outputs/answer.md",
134137
"answer_path": "refund-eligibility--4f9a7c2d1b6e/run-1/outputs/answer.md",
138+
"file_changes_path": "refund-eligibility--4f9a7c2d1b6e/run-1/outputs/file_changes.diff",
135139
"test_dir": "refund-eligibility--4f9a7c2d1b6e/test"
136140
}
137141
```

‎apps/web/src/content/docs/docs/tools/results.mdx‎

Lines changed: 9 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -130,8 +130,10 @@ token/cost usage.
130130
Every case uses aggregate `summary.json`, then stores attempt details under
131131
`run-N/`. Each `run-N/` contains a compact per-attempt manifest `result.json`,
132132
`grading.json`, `metrics.json`, `timing.json`, `transcript.jsonl`,
133-
`transcript-raw.jsonl`, and `outputs/answer.md`. The `result.json` file carries
134-
`grading_path`, `metrics_path`, transcript, and output paths.
133+
`transcript-raw.jsonl`, `outputs/answer.md`, and `outputs/file_changes.diff`
134+
when workspace changes were captured. The `result.json` file carries
135+
`grading_path`, `metrics_path`, transcript, output, and `file_changes_path`
136+
paths.
135137

136138
`transcript-raw.jsonl` preserves native provider or harness transcript bytes
137139
when they are available, while `transcript.jsonl` is the normalized
@@ -141,10 +143,10 @@ systems can be linked through safe `external_trace` metadata when available.
141143
`summary.json` remains the run-level aggregate summary. `index.jsonl` is the
142144
canonical row index for the run: one row per result, attempt, or case, carrying
143145
lightweight explicit paths such as `transcript_path`, `transcript_raw_path`,
144-
and `metrics_path` plus artifact pointers only when detached payload publishing
145-
needs them. Dashboard search indexes, SQLite indexes, and other read models are
146-
derived projections over these run artifacts, not replacements for
147-
`index.jsonl`.
146+
`file_changes_path`, and `metrics_path` plus artifact pointers only when
147+
detached payload publishing needs them. Dashboard search indexes, SQLite
148+
indexes, and other read models are derived projections over these run artifacts,
149+
not replacements for `index.jsonl`.
148150
Duration, token, and cost usage remains in `timing.json`, including source
149151
labels such as `provider_reported`, `token_estimated`, `aggregate`, or
150152
`unavailable`.
@@ -154,7 +156,7 @@ while adding AgentV/Vercel-style detail:
154156

155157
| Field group | Purpose |
156158
|-------------|---------|
157-
| `tool_calls`, `total_tool_calls`, `total_steps`, `errors_encountered`, `output_chars`, `transcript_chars`, `files_created` | Agent Skills-compatible executor metrics |
159+
| `tool_calls`, `total_tool_calls`, `total_steps`, `errors_encountered`, `output_chars`, `transcript_chars`, `files_created`, `files_deleted` | Agent Skills-compatible executor metrics |
158160
| `tool_call_events`, `tool_call_counts`, `tool_category_counts`, `shell_commands`, `files_read`, `files_modified`, `web_fetches`, `errors`, `reasoning_blocks`, `thinking_blocks`, `total_turns` | AgentV/Vercel-style behavior summary when source data includes it |
159161

160162
Vercel `@vercel/agent-eval` `results.o11y` maps into AgentV like this:

‎docs/adr/0011-result-output-artifact-contract.md‎

Lines changed: 4 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -61,6 +61,8 @@ An AgentV result output is a run-centric bundle with this root contract:
6161
transcript.jsonl
6262
transcript-raw.jsonl
6363
outputs/
64+
answer.md # when target output exists
65+
file_changes.diff # when workspace file changes exist
6466
```
6567

6668
`summary.json` and `index.jsonl` are complementary:
@@ -82,8 +84,8 @@ aggregate summaries.
8284
ordinary per-case sidecars through explicit fields such as `result_dir`,
8385
`summary_path`, `grading_path`, `metrics_path`, `timing_path`,
8486
`transcript_path`, `transcript_raw_path`, `answer_path`, `output_path`,
85-
`test_dir`, `eval_path`, `targets_path`, `files_path`, and `graders_path` when
86-
those artifacts exist.
87+
`file_changes_path`, `test_dir`, `eval_path`, `targets_path`, `files_path`, and
88+
`graders_path` when those artifacts exist.
8789

8890
`artifact_pointers` remain an offload indirection for large detached payload
8991
bytes. They are not the discovery path for ordinary sidecars that live in the

‎packages/core/src/evaluation/metrics.ts‎

Lines changed: 31 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -152,6 +152,7 @@ export const MetricsWireSchema = z
152152
files_read: z.array(FileReferenceWireSchema),
153153
files_modified: z.array(FileReferenceWireSchema),
154154
files_created: z.array(z.string()),
155+
files_deleted: z.array(z.string()).default([]),
155156
web_fetches: z.array(WebFetchWireSchema),
156157
errors: z.array(ExecutionErrorWireSchema),
157158
errors_encountered: z.number().int().nonnegative(),
@@ -184,6 +185,7 @@ export const MetricsArtifactWireSchema = z
184185
transcript_path: z.string().optional(),
185186
grading_path: z.string().optional(),
186187
timing_path: z.string().optional(),
188+
file_changes_path: z.string().optional(),
187189
})
188190
.strict(),
189191
metrics: MetricsWireSchema,
@@ -515,11 +517,14 @@ function parseModifiedPathsFromDiff(fileChanges: string | undefined): string[] {
515517
return [];
516518
}
517519
const paths = new Set<string>();
518-
for (const line of fileChanges.split('\n')) {
519-
if (!line.startsWith('+++ b/')) {
520+
const lines = fileChanges.split('\n');
521+
for (let index = 0; index < lines.length - 1; index++) {
522+
const oldLine = lines[index];
523+
const newLine = lines[index + 1];
524+
if (!oldLine.startsWith('--- a/') || !newLine?.startsWith('+++ b/')) {
520525
continue;
521526
}
522-
const filePath = line.slice('+++ b/'.length).trim();
527+
const filePath = newLine.slice('+++ b/'.length).trim();
523528
if (filePath && filePath !== '/dev/null') {
524529
paths.add(filePath);
525530
}
@@ -593,6 +598,26 @@ function buildFilesCreated(result: EvaluationResult, calls: readonly ToolCallRef
593598
return [...paths];
594599
}
595600

601+
function parseDeletedPathsFromDiff(fileChanges: string | undefined): string[] {
602+
if (!fileChanges) {
603+
return [];
604+
}
605+
const paths = new Set<string>();
606+
const lines = fileChanges.split('\n');
607+
for (let index = 0; index < lines.length - 1; index++) {
608+
const oldLine = lines[index];
609+
const newLine = lines[index + 1];
610+
if (!oldLine.startsWith('--- a/') || newLine !== '+++ /dev/null') {
611+
continue;
612+
}
613+
const filePath = oldLine.slice('--- a/'.length).trim();
614+
if (filePath) {
615+
paths.add(filePath);
616+
}
617+
}
618+
return [...paths];
619+
}
620+
596621
function buildWebFetches(calls: readonly ToolCallRef[]) {
597622
return calls.flatMap((call) => {
598623
if (toolCategory(call.toolCall.tool) !== 'web_fetch') {
@@ -831,6 +856,7 @@ function buildMetrics(result: EvaluationResult) {
831856
files_read: buildFileReads(calls),
832857
files_modified: buildFileModifications(result, calls),
833858
files_created: buildFilesCreated(result, calls),
859+
files_deleted: parseDeletedPathsFromDiff(result.fileChanges),
834860
web_fetches: buildWebFetches(calls),
835861
errors,
836862
errors_encountered: errors.length,
@@ -854,6 +880,7 @@ export function buildMetricsArtifact(
854880
transcriptPath?: string;
855881
gradingPath?: string;
856882
timingPath?: string;
883+
fileChangesPath?: string;
857884
generatedAt?: string;
858885
} = {},
859886
): MetricsArtifactWire {
@@ -876,6 +903,7 @@ export function buildMetricsArtifact(
876903
transcript_path: options.transcriptPath,
877904
grading_path: options.gradingPath,
878905
timing_path: options.timingPath,
906+
file_changes_path: options.fileChangesPath,
879907
}),
880908
metrics: buildMetrics(result),
881909
}),

‎packages/core/src/evaluation/result-artifact-contract.ts‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -7,7 +7,8 @@
77
* AgentV-owned artifacts belong when projected to a results ref, sidecar ref,
88
* or object store. Use pointers for large detached payload bytes, not as the
99
* discovery path for ordinary sidecars such as `metrics.json`; normal
10-
* sidecars should use explicit path fields such as `metrics_path`.
10+
* sidecars should use explicit path fields such as `metrics_path` and
11+
* `file_changes_path`.
1112
*
1213
* Git remote publishing treats the configured results branch as the
1314
* metadata/control plane and stores transcript payload bytes whose
@@ -27,6 +28,7 @@ export const AGENTV_RESULTS_REFS = {
2728

2829
export const CANONICAL_TRANSCRIPT_ARTIFACT_PATH = 'transcript.jsonl' as const;
2930
export const CANONICAL_METRICS_ARTIFACT_PATH = 'metrics.json' as const;
31+
export const CANONICAL_FILE_CHANGES_ARTIFACT_PATH = 'outputs/file_changes.diff' as const;
3032

3133
export const TRANSCRIPT_SCHEMA_VERSION = 'agentv.transcript.v1' as const;
3234
export const METRICS_SCHEMA_VERSION = 'agentv.metrics.v1' as const;

‎packages/core/src/evaluation/result-row-schema.ts‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -53,6 +53,7 @@ const RESULT_ROW_ALIASES = {
5353

5454
const NEW_SNAKE_CASE_ONLY_FIELDS = {
5555
artifactPointers: 'artifact_pointers',
56+
fileChangesPath: 'file_changes_path',
5657
} as const;
5758

5859
const TRACE_SUMMARY_ALIASES = {

0 commit comments

Comments
 (0)