Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -49,7 +49,9 @@
"test:cli": "node --import tsx --test test/*.test.ts",
"eval:engine": "tsx src/adapters/cli/index.ts eval engine",
"eval:agent": "tsx src/adapters/cli/index.ts eval agent",
"eval:ab": "tsx src/adapters/cli/index.ts eval agent --ab"
"eval:ab": "tsx src/adapters/cli/index.ts eval agent --ab",
"eval:engine:hard": "tsx src/adapters/cli/index.ts eval engine --dataset evals/dataset-hard --cases-file evals/cases/engine-hard.json --glossary evals/dataset-hard/glossary.json",
"eval:ab:hard": "tsx src/adapters/cli/index.ts eval agent --ab --dataset evals/dataset-hard --cases-file evals/cases/agent-hard.json --glossary evals/dataset-hard/glossary.json"
},
"dependencies": {
"@duckdb/node-api": "^1.5.3-r.3"
Expand Down
15 changes: 14 additions & 1 deletion src/adapters/cli/eval.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,12 @@ export interface RunEvalOptions {
ab?: boolean;
/** Agent suite: the agent's turn budget. */
steps?: number;
/** Score a different dataset folder (default: the committed evals/dataset). */
dataset?: string;
/** Score a different cases file (note: --cases is the case-id filter). */
casesFile?: string;
/** Apply a committed glossary, e.g. a hard dataset's curation. */
glossary?: string;
log?: (line: string) => void;
}

Expand All @@ -34,10 +40,17 @@ export async function runEval(options: RunEvalOptions): Promise<number> {
let report: SuiteReport;
switch (options.suite) {
case "engine":
report = await runEngineSuite();
report = await runEngineSuite({
datasetDir: options.dataset,
casesFile: options.casesFile,
glossaryFile: options.glossary,
});
break;
case "agent": {
const shared = {
datasetDir: options.dataset,
casesFile: options.casesFile,
glossaryFile: options.glossary,
only: options.only,
repeat: options.repeat,
provider: options.provider,
Expand Down
5 changes: 5 additions & 0 deletions src/adapters/cli/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,8 @@ Usage:
--no-verify (baseline with the self-critique pass off),
--arm <grounded|raw-sql>, and --ab (run both arms
interleaved and print the grounding comparison).
Both suites accept --dataset <folder>, --cases-file <path>
and --glossary <path> to score a different dataset.
querypad help Show this help

External databases (inspect · ask · enrich):
Expand Down Expand Up @@ -193,6 +195,9 @@ async function main(argv: string[]): Promise<number> {
verify: flags["no-verify"] !== true,
arm: stringFlag(flags, "arm"),
ab: flags.ab === true,
dataset: stringFlag(flags, "dataset"),
casesFile: stringFlag(flags, "cases-file"),
glossary: stringFlag(flags, "glossary"),
steps: evalSteps && Number.isFinite(evalSteps) ? evalSteps : undefined,
});
}
Expand Down
7 changes: 6 additions & 1 deletion src/evals/report.ts
Original file line number Diff line number Diff line change
Expand Up @@ -157,7 +157,12 @@ export function formatComparison(a: SuiteReport, b: SuiteReport): string {
export async function writeReport(report: SuiteReport, dir = RESULTS_DIR): Promise<string> {
await mkdir(path.resolve(dir), { recursive: true });
const arm = report.arm ? `-${report.arm}` : "";
const file = path.resolve(dir, `${report.suite}${arm}-${report.generatedAt}.json`);
// The dataset belongs in the name: two datasets would otherwise leave
// indistinguishable baselines side by side in the results directory.
const dataset = report.dataset
? `-${path.basename(report.dataset).replace(/[^A-Za-z0-9_-]/g, "_")}`
: "";
const file = path.resolve(dir, `${report.suite}${arm}${dataset}-${report.generatedAt}.json`);
await writeFile(file, JSON.stringify(report, null, 2));
return file;
}
27 changes: 15 additions & 12 deletions src/evals/run-agent.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,11 +6,10 @@ import { DATA_TOOL_DEFINITIONS } from "../core/agent/toolkit";
import type { QueryRunner } from "../core/discovery/relationships";
import { prepareDataset } from "../adapters/dataset";
import { resolveAiCredentials } from "../adapters/cli/ai-env";
import { resolveSource } from "../adapters/cli/source";
import { createNodeDb } from "../engine/duckdb/connection";
import { ARMS, ARM_IDS, type ArmId } from "./arms";
import { checkBehavior, compareRows } from "./grade";
import { EVAL_DATASET } from "./run-engine";
import { EVAL_DATASET, loadEvalDataset, type EvalDatasetOptions } from "./run-engine";
import type { AgentCase, CaseResult, SuiteConfig, SuiteReport } from "./types";

export const AGENT_CASES = "evals/cases/agent.json";
Expand Down Expand Up @@ -51,8 +50,7 @@ function buildComplete(provider?: string): AgentComplete {
});
}

export interface AgentSuiteOptions {
datasetDir?: string;
export interface AgentSuiteOptions extends EvalDatasetOptions {
casesFile?: string;
/** Restrict to these case ids. */
only?: string[];
Expand Down Expand Up @@ -80,6 +78,15 @@ export interface AgentSuiteOptions {
complete?: AgentComplete;
}

/** Dataset/cases identity recorded on every report. */
function identity(options: AgentSuiteOptions) {
return {
dataset: options.datasetDir ?? EVAL_DATASET,
casesFile: options.cases ? "<injected>" : (options.casesFile ?? AGENT_CASES),
glossary: Boolean(options.glossaryFile),
};
}

interface RunGrade {
detail: string;
toolsUsed: string[];
Expand Down Expand Up @@ -242,10 +249,7 @@ export async function runAgentSuite(options: AgentSuiteOptions = {}): Promise<Su
const repeat = Math.max(1, options.repeat ?? 1);
const db = await createNodeDb();
try {
const dataset = await prepareDataset(
{ ...resolveSource({ folder: options.datasetDir ?? EVAL_DATASET }), outDir: "/dev/null" },
db.runner
);
const dataset = await loadEvalDataset(options, db.runner);

const deps: CaseDeps = {
dataset,
Expand All @@ -268,6 +272,7 @@ export async function runAgentSuite(options: AgentSuiteOptions = {}): Promise<Su
arm: armId,
generatedAt: Date.now(),
config: describeConfig(armId, deps),
...identity(options),
...tally(results),
results,
};
Expand Down Expand Up @@ -310,10 +315,7 @@ export async function runAbSuite(
const repeat = Math.max(1, options.repeat ?? 1);
const db = await createNodeDb();
try {
const dataset = await prepareDataset(
{ ...resolveSource({ folder: options.datasetDir ?? EVAL_DATASET }), outDir: "/dev/null" },
db.runner
);
const dataset = await loadEvalDataset(options, db.runner);

const deps: CaseDeps = {
dataset,
Expand Down Expand Up @@ -343,6 +345,7 @@ export async function runAbSuite(
arm: armId,
generatedAt,
config: describeConfig(armId, deps),
...identity(options),
...tally(results),
results,
};
Expand Down
54 changes: 47 additions & 7 deletions src/evals/run-engine.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,8 @@
import { readFile } from "node:fs/promises";
import path from "node:path";
import { compileMetric } from "../core/discovery/compile-metric";
import { relationshipKey } from "../core/discovery/relationships";
import type { GlossaryEntry } from "../core/discovery/glossary";
import { relationshipKey, type QueryRunner } from "../core/discovery/relationships";
import { buildTermCatalog, formatTarget } from "../core/discovery/term-catalog";
import { resolveTerms } from "../core/discovery/term-search";
import { createNodeDb } from "../engine/duckdb/connection";
Expand All @@ -17,6 +18,46 @@ export async function loadEngineCases(file = ENGINE_CASES): Promise<EngineCase[]
return JSON.parse(await readFile(path.resolve(file), "utf8")) as EngineCase[];
}

export interface EngineSuiteOptions extends EvalDatasetOptions {
casesFile?: string;
}

/** Which dataset (and optional glossary) a suite run scores. */
export interface EvalDatasetOptions {
datasetDir?: string;
/**
* A committed glossary to apply, e.g. `evals/dataset-hard/glossary.json`. The suites read
* no `.datactx/` cache, so curation has to be passed in explicitly.
*/
glossaryFile?: string;
}

/** Read a committed glossary file (the `writeGlossary` doc shape, or a bare entry array). */
export async function loadGlossary(file: string): Promise<GlossaryEntry[]> {
const parsed = JSON.parse(await readFile(path.resolve(file), "utf8")) as
| { entries?: GlossaryEntry[] }
| GlossaryEntry[];
return Array.isArray(parsed) ? parsed : (parsed.entries ?? []);
}

/**
* Load the dataset a suite grades. Shared by the engine and agent suites so the two
* cannot drift on which dataset or curation the engine was grounded in. Cached
* artifacts are deliberately ignored: the suite grades what the engine derives now.
*/
export async function loadEvalDataset(
options: EvalDatasetOptions,
runner: QueryRunner
): Promise<PreparedDataset> {
const glossary = options.glossaryFile ? await loadGlossary(options.glossaryFile) : undefined;
return prepareDataset(
{ ...resolveSource({ folder: options.datasetDir ?? EVAL_DATASET }), outDir: "/dev/null" },
runner,
Date.now(),
{ glossary }
);
}

function gradeRelationship(testCase: EngineCase, dataset: PreparedDataset): string {
const found = dataset.relationships.find((rel) => relationshipKey(rel) === testCase.edge);
if (testCase.absent) {
Expand Down Expand Up @@ -85,17 +126,13 @@ async function gradeTerm(testCase: EngineCase, dataset: PreparedDataset): Promis
* fan-out), and term resolution. Needs no API key, so it runs in CI.
*/
export async function runEngineSuite(
options: { datasetDir?: string; casesFile?: string } = {}
options: EngineSuiteOptions = {}
): Promise<SuiteReport> {
const cases = await loadEngineCases(options.casesFile);
const db = await createNodeDb();
let dataset: PreparedDataset;
try {
// Ignore any cached artifacts: the suite must grade what the engine derives now.
dataset = await prepareDataset(
{ ...resolveSource({ folder: options.datasetDir ?? EVAL_DATASET }), outDir: "/dev/null" },
db.runner
);
dataset = await loadEvalDataset(options, db.runner);

const results: CaseResult[] = [];
for (const testCase of cases) {
Expand Down Expand Up @@ -140,6 +177,9 @@ export async function runEngineSuite(
return {
suite: "engine",
generatedAt: Date.now(),
dataset: options.datasetDir ?? EVAL_DATASET,
casesFile: options.casesFile ?? ENGINE_CASES,
glossary: Boolean(options.glossaryFile),
total: results.length,
passed,
failed,
Expand Down
5 changes: 5 additions & 0 deletions src/evals/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,11 @@ export interface SuiteReport {
arm?: string;
/** Agent suite only: the full setup this score was produced under. */
config?: SuiteConfig;
/** Which dataset and cases were scored — two datasets otherwise look identical. */
dataset?: string;
casesFile?: string;
/** Whether a glossary was applied (the grounding under test can depend on it). */
glossary?: boolean;
generatedAt: number;
total: number;
passed: number;
Expand Down
57 changes: 57 additions & 0 deletions test/evals.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,16 @@ import { checkBehavior, compareRows } from "../src/evals/grade";
import type { AgentComplete } from "../src/core/agent/loop";
import type { AgentCase } from "../src/evals/types";

async function tempJson(name: string, data: unknown): Promise<string> {
const { mkdtemp, writeFile } = await import("node:fs/promises");
const { tmpdir } = await import("node:os");
const path = await import("node:path");
const dir = await mkdtemp(path.join(tmpdir(), "querypad-eval-"));
const file = path.join(dir, name);
await writeFile(file, JSON.stringify(data));
return file;
}

const rows = (data: Record<string, unknown>[]) => ({
columns: Object.keys(data[0] ?? {}),
rows: data,
Expand Down Expand Up @@ -451,3 +461,50 @@ test("runAbSuite measures both arms and formatComparison renders the delta", asy
assert.match(out, /validity checks/);
assert.match(out, /turn budget hit on 0\//);
});

// ---- dataset parameterization -------------------------------------------------

test("suites can score a different dataset, and record which one they scored", async () => {
const { runEngineSuite } = await import("../src/evals/run-engine");

const committed = await runEngineSuite();
assert.equal(committed.dataset, "evals/dataset");
assert.equal(committed.casesFile, "evals/cases/engine.json");
assert.equal(committed.glossary, false);
assert.equal(committed.score, 1, "the default pair must stay green");

// Pointed at a different folder the same cases no longer hold, which is the
// proof that datasetDir is actually honored rather than silently ignored.
const elsewhere = await runEngineSuite({ datasetDir: "fixtures/data" });
assert.equal(elsewhere.dataset, "fixtures/data");
assert.ok(elsewhere.failed > 0, "eval cases should not pass against another dataset");
});

test("a scored glossary is recorded on the report and reaches the model", async () => {
const { runAgentSuite } = await import("../src/evals/run-agent");
const glossaryFile = await tempJson("glossary.json", {
generatedAt: 1,
entries: [
{
term: "revenue",
definition: "Money billed.",
synonyms: ["net revenue"],
mapsTo: { table: "products", column: "unit_price" },
confidence: 0.9,
},
],
applied: [],
});

const spy = spyComplete(scriptedAgent(() => COUNT_SQL));
const report = await runAgentSuite({
only: ["simple-count"],
glossaryFile,
complete: spy.complete,
});

assert.equal(report.glossary, true);
assert.equal(report.dataset, "evals/dataset");
// The glossary annotation must actually reach the grounding context the model saw.
assert.match(spy.calls[0].system, /aka net revenue/);
});