diff --git a/package.json b/package.json index 52b9eae..f932f53 100644 --- a/package.json +++ b/package.json @@ -49,7 +49,9 @@ "test:cli": "node --import tsx --test test/*.test.ts", "eval:engine": "tsx src/adapters/cli/index.ts eval engine", "eval:agent": "tsx src/adapters/cli/index.ts eval agent", - "eval:ab": "tsx src/adapters/cli/index.ts eval agent --ab" + "eval:ab": "tsx src/adapters/cli/index.ts eval agent --ab", + "eval:engine:hard": "tsx src/adapters/cli/index.ts eval engine --dataset evals/dataset-hard --cases-file evals/cases/engine-hard.json --glossary evals/dataset-hard/glossary.json", + "eval:ab:hard": "tsx src/adapters/cli/index.ts eval agent --ab --dataset evals/dataset-hard --cases-file evals/cases/agent-hard.json --glossary evals/dataset-hard/glossary.json" }, "dependencies": { "@duckdb/node-api": "^1.5.3-r.3" diff --git a/src/adapters/cli/eval.ts b/src/adapters/cli/eval.ts index a77309a..8daa8f2 100644 --- a/src/adapters/cli/eval.ts +++ b/src/adapters/cli/eval.ts @@ -20,6 +20,12 @@ export interface RunEvalOptions { ab?: boolean; /** Agent suite: the agent's turn budget. */ steps?: number; + /** Score a different dataset folder (default: the committed evals/dataset). */ + dataset?: string; + /** Score a different cases file (note: --cases is the case-id filter). */ + casesFile?: string; + /** Apply a committed glossary, e.g. a hard dataset's curation. */ + glossary?: string; log?: (line: string) => void; } @@ -34,10 +40,17 @@ export async function runEval(options: RunEvalOptions): Promise { let report: SuiteReport; switch (options.suite) { case "engine": - report = await runEngineSuite(); + report = await runEngineSuite({ + datasetDir: options.dataset, + casesFile: options.casesFile, + glossaryFile: options.glossary, + }); break; case "agent": { const shared = { + datasetDir: options.dataset, + casesFile: options.casesFile, + glossaryFile: options.glossary, only: options.only, repeat: options.repeat, provider: options.provider, diff --git a/src/adapters/cli/index.ts b/src/adapters/cli/index.ts index 61d94f7..56de615 100644 --- a/src/adapters/cli/index.ts +++ b/src/adapters/cli/index.ts @@ -34,6 +34,8 @@ Usage: --no-verify (baseline with the self-critique pass off), --arm , and --ab (run both arms interleaved and print the grounding comparison). + Both suites accept --dataset , --cases-file + and --glossary to score a different dataset. querypad help Show this help External databases (inspect · ask · enrich): @@ -193,6 +195,9 @@ async function main(argv: string[]): Promise { verify: flags["no-verify"] !== true, arm: stringFlag(flags, "arm"), ab: flags.ab === true, + dataset: stringFlag(flags, "dataset"), + casesFile: stringFlag(flags, "cases-file"), + glossary: stringFlag(flags, "glossary"), steps: evalSteps && Number.isFinite(evalSteps) ? evalSteps : undefined, }); } diff --git a/src/evals/report.ts b/src/evals/report.ts index f04faca..fb4f802 100644 --- a/src/evals/report.ts +++ b/src/evals/report.ts @@ -157,7 +157,12 @@ export function formatComparison(a: SuiteReport, b: SuiteReport): string { export async function writeReport(report: SuiteReport, dir = RESULTS_DIR): Promise { await mkdir(path.resolve(dir), { recursive: true }); const arm = report.arm ? `-${report.arm}` : ""; - const file = path.resolve(dir, `${report.suite}${arm}-${report.generatedAt}.json`); + // The dataset belongs in the name: two datasets would otherwise leave + // indistinguishable baselines side by side in the results directory. + const dataset = report.dataset + ? `-${path.basename(report.dataset).replace(/[^A-Za-z0-9_-]/g, "_")}` + : ""; + const file = path.resolve(dir, `${report.suite}${arm}${dataset}-${report.generatedAt}.json`); await writeFile(file, JSON.stringify(report, null, 2)); return file; } diff --git a/src/evals/run-agent.ts b/src/evals/run-agent.ts index 590dd7c..9a481b5 100644 --- a/src/evals/run-agent.ts +++ b/src/evals/run-agent.ts @@ -6,11 +6,10 @@ import { DATA_TOOL_DEFINITIONS } from "../core/agent/toolkit"; import type { QueryRunner } from "../core/discovery/relationships"; import { prepareDataset } from "../adapters/dataset"; import { resolveAiCredentials } from "../adapters/cli/ai-env"; -import { resolveSource } from "../adapters/cli/source"; import { createNodeDb } from "../engine/duckdb/connection"; import { ARMS, ARM_IDS, type ArmId } from "./arms"; import { checkBehavior, compareRows } from "./grade"; -import { EVAL_DATASET } from "./run-engine"; +import { EVAL_DATASET, loadEvalDataset, type EvalDatasetOptions } from "./run-engine"; import type { AgentCase, CaseResult, SuiteConfig, SuiteReport } from "./types"; export const AGENT_CASES = "evals/cases/agent.json"; @@ -51,8 +50,7 @@ function buildComplete(provider?: string): AgentComplete { }); } -export interface AgentSuiteOptions { - datasetDir?: string; +export interface AgentSuiteOptions extends EvalDatasetOptions { casesFile?: string; /** Restrict to these case ids. */ only?: string[]; @@ -80,6 +78,15 @@ export interface AgentSuiteOptions { complete?: AgentComplete; } +/** Dataset/cases identity recorded on every report. */ +function identity(options: AgentSuiteOptions) { + return { + dataset: options.datasetDir ?? EVAL_DATASET, + casesFile: options.cases ? "" : (options.casesFile ?? AGENT_CASES), + glossary: Boolean(options.glossaryFile), + }; +} + interface RunGrade { detail: string; toolsUsed: string[]; @@ -242,10 +249,7 @@ export async function runAgentSuite(options: AgentSuiteOptions = {}): Promise { + const parsed = JSON.parse(await readFile(path.resolve(file), "utf8")) as + | { entries?: GlossaryEntry[] } + | GlossaryEntry[]; + return Array.isArray(parsed) ? parsed : (parsed.entries ?? []); +} + +/** + * Load the dataset a suite grades. Shared by the engine and agent suites so the two + * cannot drift on which dataset or curation the engine was grounded in. Cached + * artifacts are deliberately ignored: the suite grades what the engine derives now. + */ +export async function loadEvalDataset( + options: EvalDatasetOptions, + runner: QueryRunner +): Promise { + const glossary = options.glossaryFile ? await loadGlossary(options.glossaryFile) : undefined; + return prepareDataset( + { ...resolveSource({ folder: options.datasetDir ?? EVAL_DATASET }), outDir: "/dev/null" }, + runner, + Date.now(), + { glossary } + ); +} + function gradeRelationship(testCase: EngineCase, dataset: PreparedDataset): string { const found = dataset.relationships.find((rel) => relationshipKey(rel) === testCase.edge); if (testCase.absent) { @@ -85,17 +126,13 @@ async function gradeTerm(testCase: EngineCase, dataset: PreparedDataset): Promis * fan-out), and term resolution. Needs no API key, so it runs in CI. */ export async function runEngineSuite( - options: { datasetDir?: string; casesFile?: string } = {} + options: EngineSuiteOptions = {} ): Promise { const cases = await loadEngineCases(options.casesFile); const db = await createNodeDb(); let dataset: PreparedDataset; try { - // Ignore any cached artifacts: the suite must grade what the engine derives now. - dataset = await prepareDataset( - { ...resolveSource({ folder: options.datasetDir ?? EVAL_DATASET }), outDir: "/dev/null" }, - db.runner - ); + dataset = await loadEvalDataset(options, db.runner); const results: CaseResult[] = []; for (const testCase of cases) { @@ -140,6 +177,9 @@ export async function runEngineSuite( return { suite: "engine", generatedAt: Date.now(), + dataset: options.datasetDir ?? EVAL_DATASET, + casesFile: options.casesFile ?? ENGINE_CASES, + glossary: Boolean(options.glossaryFile), total: results.length, passed, failed, diff --git a/src/evals/types.ts b/src/evals/types.ts index 33a298a..8575d3d 100644 --- a/src/evals/types.ts +++ b/src/evals/types.ts @@ -99,6 +99,11 @@ export interface SuiteReport { arm?: string; /** Agent suite only: the full setup this score was produced under. */ config?: SuiteConfig; + /** Which dataset and cases were scored — two datasets otherwise look identical. */ + dataset?: string; + casesFile?: string; + /** Whether a glossary was applied (the grounding under test can depend on it). */ + glossary?: boolean; generatedAt: number; total: number; passed: number; diff --git a/test/evals.test.ts b/test/evals.test.ts index 8908b23..31f98cc 100644 --- a/test/evals.test.ts +++ b/test/evals.test.ts @@ -4,6 +4,16 @@ import { checkBehavior, compareRows } from "../src/evals/grade"; import type { AgentComplete } from "../src/core/agent/loop"; import type { AgentCase } from "../src/evals/types"; +async function tempJson(name: string, data: unknown): Promise { + const { mkdtemp, writeFile } = await import("node:fs/promises"); + const { tmpdir } = await import("node:os"); + const path = await import("node:path"); + const dir = await mkdtemp(path.join(tmpdir(), "querypad-eval-")); + const file = path.join(dir, name); + await writeFile(file, JSON.stringify(data)); + return file; +} + const rows = (data: Record[]) => ({ columns: Object.keys(data[0] ?? {}), rows: data, @@ -451,3 +461,50 @@ test("runAbSuite measures both arms and formatComparison renders the delta", asy assert.match(out, /validity checks/); assert.match(out, /turn budget hit on 0\//); }); + +// ---- dataset parameterization ------------------------------------------------- + +test("suites can score a different dataset, and record which one they scored", async () => { + const { runEngineSuite } = await import("../src/evals/run-engine"); + + const committed = await runEngineSuite(); + assert.equal(committed.dataset, "evals/dataset"); + assert.equal(committed.casesFile, "evals/cases/engine.json"); + assert.equal(committed.glossary, false); + assert.equal(committed.score, 1, "the default pair must stay green"); + + // Pointed at a different folder the same cases no longer hold, which is the + // proof that datasetDir is actually honored rather than silently ignored. + const elsewhere = await runEngineSuite({ datasetDir: "fixtures/data" }); + assert.equal(elsewhere.dataset, "fixtures/data"); + assert.ok(elsewhere.failed > 0, "eval cases should not pass against another dataset"); +}); + +test("a scored glossary is recorded on the report and reaches the model", async () => { + const { runAgentSuite } = await import("../src/evals/run-agent"); + const glossaryFile = await tempJson("glossary.json", { + generatedAt: 1, + entries: [ + { + term: "revenue", + definition: "Money billed.", + synonyms: ["net revenue"], + mapsTo: { table: "products", column: "unit_price" }, + confidence: 0.9, + }, + ], + applied: [], + }); + + const spy = spyComplete(scriptedAgent(() => COUNT_SQL)); + const report = await runAgentSuite({ + only: ["simple-count"], + glossaryFile, + complete: spy.complete, + }); + + assert.equal(report.glossary, true); + assert.equal(report.dataset, "evals/dataset"); + // The glossary annotation must actually reach the grounding context the model saw. + assert.match(spy.calls[0].system, /aka net revenue/); +});