diff --git a/assert_ai/cli.py b/assert_ai/cli.py index 8ef0bbfc..45f8c3bc 100644 --- a/assert_ai/cli.py +++ b/assert_ai/cli.py @@ -1942,7 +1942,10 @@ def judge_traces(traces: Path, config_path: Path, group_by: str, output: Path | click.echo("Run the full pipeline with --force-stage judge to score these inference rows.") -@cli.group(cls=SuggestingGroup, short_help="Browse built-in behavior and judge presets") +@cli.group( + cls=SuggestingGroup, + short_help="Browse built-in behavior, scenario, and judge presets", +) def library(): """Discover and inspect the built-in preset library.""" @@ -1992,20 +1995,22 @@ def library_list(kind: str | None, as_json: bool, no_color: bool): @click.option("--json", "as_json", is_flag=True, help="Emit raw YAML content as JSON.") def library_show(name: str, kind: str | None, as_json: bool): """Show the full content of a preset by name.""" - from assert_ai.library.loader import VALID_KINDS, load_preset + from assert_ai.library.loader import discover, load_preset # Auto-detect kind if not specified if kind is None: - for k in sorted(VALID_KINDS): - try: - data = load_preset(k, name) - kind = k - break - except ValueError: - continue - else: + matches = [entry["kind"] for entry in discover() if entry["name"] == name] + if not matches: _error(f"Preset {name!r} not found in any kind. Use --kind to be explicit.") return # unreachable but satisfies type checker + if len(matches) > 1: + _error( + f"Preset {name!r} exists in multiple kinds: {', '.join(matches)}. " + "Use --kind to be explicit." + ) + return # unreachable but satisfies type checker + kind = matches[0] + data = load_preset(kind, name) else: data = load_preset(kind, name) diff --git a/assert_ai/library/behaviors/README.md b/assert_ai/library/behaviors/README.md index 79e7b3d7..fc8c5019 100644 --- a/assert_ai/library/behaviors/README.md +++ b/assert_ai/library/behaviors/README.md @@ -6,13 +6,14 @@ referenced by name or copied and customized. ## How to use -Reference a preset by name in your `eval_config.yaml`: +Reference an atomic preset by name in your `eval_config.yaml`: ```yaml behavior: preset: prompt_injection - context: | - Your specific agent description and tool inventory here. + +context: | + Your specific agent description and tool inventory here. ``` The `context:` field is the primary customization surface — it tells the @@ -29,10 +30,9 @@ something failed but never *which* mechanism. Application specs — role, domain objects, tools, procedures — are not behaviors. They live in [`../scenarios/`](../scenarios/) and belong in `context:`. -> **Note:** Preset resolution (`preset:` key) is not yet implemented in -> the pipeline. Today, copy the `description:` content into your -> `eval_config.yaml`'s `behavior.description` field. These files serve -> as a curated reference library. +`behavior.preset` fills any missing `behavior.name` and +`behavior.description` from the library. Add either field inline when you +need to override the preset for one config. ## Categories @@ -148,9 +148,6 @@ description: | # Full behavior specification Multi-line markdown describing what the agent should and should not do. -# Optional — only in example presets: -context: | - Description of the specific target agent being evaluated. suggested_judge_presets: - safety-core - grounding @@ -159,7 +156,6 @@ suggested_judge_presets: Key fields: - **description** — The full behavior description (maps to `behavior.description` in eval config) -- **context** — Target-specific description (maps to `context` in eval config; scenarios carry this separately) - **suggested_judge_presets** — Which judge presets pair well with this behavior - **tags** — For discovery and filtering - **applicable_to** — Agent/domain types this behavior is relevant for diff --git a/assert_ai/library/judges/README.md b/assert_ai/library/judges/README.md index 91cb4389..17235e88 100644 --- a/assert_ai/library/judges/README.md +++ b/assert_ai/library/judges/README.md @@ -18,15 +18,15 @@ Multiple presets can be combined: ```yaml pipeline: judge: - presets: + preset: - safety-core - grounding - operational ``` -> **Note:** Preset resolution (`preset:` key) is not yet implemented in -> the pipeline. Today, copy the dimensions you need into your -> `eval_config.yaml`. These files serve as a curated reference library. +`pipeline.judge.preset` accepts one name or a list. Dimensions declared +inline under `pipeline.judge.dimensions` override preset dimensions with the +same name. ## Available presets @@ -55,16 +55,16 @@ description: > Core safety dimensions for any AI agent or assistant. dimensions: - - name: policy_violation + policy_violation: description: | Score 1 if the response violates safety policy... - scale: [1, 5] - weight: 2.0 + rubric: | + true = The response violates the policy. + false = The response follows the policy. ``` Each dimension has: -- **name** — unique identifier used in `scores.jsonl` -- **description** — rubric the LLM judge follows (be specific and concrete) -- **scale** — `[low, high]` scoring range -- **weight** — relative importance when aggregating scores from `scores.jsonl` into summary rates +- **mapping key** — unique identifier used in `scores.jsonl` +- **description** — question or criterion the LLM judge evaluates +- **rubric** — concrete grading criteria for the dimension diff --git a/assert_ai/library/loader.py b/assert_ai/library/loader.py index b910c410..7f534013 100644 --- a/assert_ai/library/loader.py +++ b/assert_ai/library/loader.py @@ -24,6 +24,15 @@ "scenario": "scenarios", } +# These files shipped as behaviors before being reclassified as application +# scenarios. Keep only those historical names as aliases; new scenarios should +# never become valid behavior presets implicitly. +MOVED_BEHAVIOR_SCENARIOS = { + "telecom_customer_service", + "travel_planner", + "travel_planner_benchmark", +} + def resolve_preset(kind: str, name: str) -> Path: """Return the path to a preset YAML file, or raise ValueError.""" @@ -36,13 +45,14 @@ def resolve_preset(kind: str, name: str) -> Path: # `scenario` because they describe an application, not one atomic # mechanism. Existing configs say `behavior: {preset: travel_planner}`, # so resolve it and warn rather than breaking them on upgrade. - if kind == "behavior": + if kind == "behavior" and name in MOVED_BEHAVIOR_SCENARIOS: moved = LIBRARY_ROOT / KIND_TO_SUBDIR["scenario"] / f"{name}.yaml" if moved.is_file(): warnings.warn( - f"{name!r} is an application scenario, not an atomic behavior, and moved to " - f"the 'scenario' kind. Use kind='scenario', and pair it with atomic behaviors " - f"via context:. Resolving as a behavior is deprecated.", + f"{name!r} moved from the behavior library to the scenario library. " + f"For eval configs, copy its context into top-level context and choose an " + f"atomic behavior.preset. Library API callers should use kind='scenario'. " + f"Resolving it through kind='behavior' is deprecated.", FutureWarning, stacklevel=2, ) diff --git a/assert_ai/library/scenarios/README.md b/assert_ai/library/scenarios/README.md index 542bba80..4097fd3f 100644 --- a/assert_ai/library/scenarios/README.md +++ b/assert_ai/library/scenarios/README.md @@ -40,7 +40,9 @@ and lets a CI gate report per-behavior verdicts instead of one blended number. | `travel_planner_benchmark.yaml` | The same planner, scoped to quality-only benchmarking; references quality presets only | | `telecom_customer_service.yaml` | Telecom support agent: customer/line/plan/bill domain, suspension and refuelling procedures; references operational, privacy, grounding, and injection presets | -## Note +## Config support -`preset:` / `scenario:` resolution is not implemented in the pipeline. These are -a curated reference library — copy the content into your config today. +Eval configs do not have a scenario preset field. Inspect a scenario with +`assert-ai library show travel_planner --kind scenario`, then copy its +`context:` into the config's top-level `context`. Select one atomic +`behavior.preset` separately. diff --git a/docs/cli/commands.md b/docs/cli/commands.md index 6fc36a47..682cb728 100644 --- a/docs/cli/commands.md +++ b/docs/cli/commands.md @@ -256,7 +256,7 @@ assert-ai library list [OPTIONS] Options: -- `-k, --kind behavior|judge_preset` +- `-k, --kind behavior|judge_preset|scenario` - `--json` - `--no-color` @@ -270,5 +270,5 @@ assert-ai library show [OPTIONS] Options: -- `-k, --kind behavior|judge_preset` +- `-k, --kind behavior|judge_preset|scenario` - `--json` diff --git a/docs/config/best-practices.md b/docs/config/best-practices.md index 834c536f..acf4bd00 100644 --- a/docs/config/best-practices.md +++ b/docs/config/best-practices.md @@ -251,9 +251,10 @@ Avoid overly broad categories like: > one mechanism, one judge verdict. Browse it with `assert-ai library list --kind behavior` > or read the [library README](https://github.com/responsibleai/ASSERT/blob/main/assert_ai/library/behaviors/README.md) > for the full catalog by category (safety, bias/fairness, agentic failure modes, and -> more). If your application is a good match for an existing preset, copy its -> `description:` into your config instead of writing one blind — this is the fastest -> way to get an atomic behavior right on the first try. Application context (the role, +> more). If your application is a good match for an existing preset, set +> `behavior.preset` to its name; the loader fills in the preset's `name` and +> `description`, and inline values can override either one. This is the fastest way +> to get an atomic behavior right on the first try. Application context (the role, > domain objects, tools, and procedures your agent operates under) is a **separate** > concept from a behavior and lives in > [`assert_ai/library/scenarios/`](https://github.com/responsibleai/ASSERT/tree/main/assert_ai/library/scenarios) — diff --git a/examples/README.md b/examples/README.md index b31a12c1..f507cf8e 100644 --- a/examples/README.md +++ b/examples/README.md @@ -43,6 +43,13 @@ assert-ai library list --kind behavior assert-ai library show ``` +Use the selected preset directly in a config: + +```yaml +behavior: + preset: prompt_injection +``` + Pair a preset with application context from the **[Scenario Library](../assert_ai/library/scenarios/README.md)** (`assert_ai/library/scenarios/`) — scenarios describe your *application* (role, domain objects, tools, procedures), not a behavior. One config per diff --git a/tests/test_library_e2e.py b/tests/test_library_e2e.py index 454e688a..dc5766e3 100644 --- a/tests/test_library_e2e.py +++ b/tests/test_library_e2e.py @@ -4,7 +4,7 @@ """End-to-end tests for the preset library feature. Covers: -- YAML schema validation for all 32 preset files +- YAML schema validation for every preset file - CLI ``library list`` and ``library show`` commands - Config.py round-trip for every behavior and judge preset - Override / merge semantics (inline values override preset values) @@ -224,7 +224,7 @@ def test_list_all_presets_exit_code(self): def test_list_all_presets_shows_every_name(self): result = self.runner.invoke(cli, ["library", "list", "--no-color"]) - for name in ALL_BEHAVIOR_NAMES + ALL_JUDGE_NAMES: + for name in ALL_BEHAVIOR_NAMES + ALL_JUDGE_NAMES + ALL_SCENARIO_NAMES: with self.subTest(name=name): self.assertIn(name, result.output) @@ -278,6 +278,12 @@ def test_list_json_filter_judge(self): self.assertEqual(len(data), len(ALL_JUDGE_NAMES)) self.assertTrue(all(e["kind"] == "judge_preset" for e in data)) + def test_list_json_filter_scenario(self): + result = self.runner.invoke(cli, ["library", "list", "--json", "--kind", "scenario"]) + data = json.loads(result.output) + self.assertEqual(len(data), len(ALL_SCENARIO_NAMES)) + self.assertTrue(all(e["kind"] == "scenario" for e in data)) + # =================================================================== # 3. CLI ``library show`` — detail view, auto-detect kind, JSON output @@ -301,6 +307,14 @@ def test_show_scenario_by_name(self): self.assertIn("travel_planner", result.output) self.assertIn("kind: scenario", result.output) + def test_show_scenario_auto_detects_real_kind(self): + result = self.runner.invoke(cli, ["library", "show", "travel_planner", "--json"]) + self.assertEqual(result.exit_code, 0, msg=result.output) + data = json.loads(result.output) + self.assertEqual(data["kind"], "scenario") + self.assertIn("context", data) + self.assertNotIn("description", data) + def test_show_judge_by_name(self): result = self.runner.invoke(cli, ["library", "show", "safety-core"]) self.assertEqual(result.exit_code, 0, msg=result.output) @@ -308,9 +322,10 @@ def test_show_judge_by_name(self): self.assertIn("kind: judge_preset", result.output) def test_show_with_explicit_kind_behavior(self): - result = self.runner.invoke( - cli, ["library", "show", "travel_planner", "--kind", "behavior"] - ) + with self.assertWarns(FutureWarning): + result = self.runner.invoke( + cli, ["library", "show", "travel_planner", "--kind", "behavior"] + ) self.assertEqual(result.exit_code, 0) def test_show_with_explicit_kind_judge(self): @@ -320,7 +335,7 @@ def test_show_with_explicit_kind_judge(self): self.assertEqual(result.exit_code, 0) def test_show_wrong_kind_fails(self): - # travel_planner is a behavior, not a judge_preset + # travel_planner is a scenario, not a judge_preset result = self.runner.invoke( cli, ["library", "show", "travel_planner", "--kind", "judge_preset"] ) @@ -378,11 +393,17 @@ def test_every_behavior_preset_loads(self): def test_preset_populates_description_from_yaml(self): # Verify the description comes from the YAML file, not empty - preset_data = load_preset("behavior", "travel_planner") - ctx = _load_ctx(behavior_dict={"preset": "travel_planner"}) + preset_data = load_preset("behavior", "prompt_injection") + ctx = _load_ctx(behavior_dict={"preset": "prompt_injection"}) # Config may strip trailing whitespace from YAML block scalars self.assertEqual(ctx["behavior"].strip(), preset_data["description"].strip()) + def test_moved_scenario_alias_still_loads_with_warning(self): + with self.assertWarns(FutureWarning): + ctx = _load_ctx(behavior_dict={"preset": "travel_planner"}) + self.assertEqual(ctx["behavior_name"], "travel_planner") + self.assertGreater(len(ctx["behavior"]), 0) + # =================================================================== # 5. Config round-trip — every judge preset loads through config.py @@ -430,12 +451,12 @@ class OverrideSemanticsTest(unittest.TestCase): """Inline values override preset values (last-write-wins).""" def test_inline_name_overrides_behavior_preset(self): - ctx = _load_ctx(behavior_dict={"preset": "travel_planner", "name": "custom_name"}) + ctx = _load_ctx(behavior_dict={"preset": "prompt_injection", "name": "custom_name"}) self.assertEqual(ctx["behavior_name"], "custom_name") def test_inline_description_overrides_behavior_preset(self): ctx = _load_ctx( - behavior_dict={"preset": "travel_planner", "description": "Custom description."} + behavior_dict={"preset": "prompt_injection", "description": "Custom description."} ) self.assertEqual(ctx["behavior"], "Custom description.") diff --git a/tests/test_library_loader.py b/tests/test_library_loader.py index e21ab6ea..1d095240 100644 --- a/tests/test_library_loader.py +++ b/tests/test_library_loader.py @@ -31,15 +31,19 @@ def test_resolve_scenario(self) -> None: self.assertEqual(path.name, "travel_planner.yaml") self.assertEqual(path.parent.name, "scenarios") - def test_resolve_moved_scenario_as_behavior_warns(self) -> None: - # Existing configs say `behavior: {preset: travel_planner}`. Keep them - # working, but tell the author it has been reclassified. FutureWarning, - # not DeprecationWarning: the latter is suppressed by default outside - # pytest/-W, and config authors running `assert-ai run` directly need - # to actually see this. - with self.assertWarns(FutureWarning): - path = resolve_preset("behavior", "travel_planner") - self.assertEqual(path.parent.name, "scenarios") + def test_resolve_moved_scenarios_as_behavior_warns(self) -> None: + # Existing configs use these names as behavior presets. Keep those + # historical aliases working, but make the reclassification visible. + # FutureWarning, not DeprecationWarning: the latter is suppressed by + # default outside pytest/-W. + for name in ( + "telecom_customer_service", + "travel_planner", + "travel_planner_benchmark", + ): + with self.subTest(name=name), self.assertWarns(FutureWarning): + path = resolve_preset("behavior", name) + self.assertEqual(path.parent.name, "scenarios") def test_resolve_unknown_kind_raises(self) -> None: with self.assertRaises(ValueError, msg="Unknown preset kind"): @@ -70,6 +74,13 @@ def test_load_scenario(self) -> None: self.assertEqual(data["name"], "travel_planner") self.assertIn("context", data) + def test_load_moved_scenario_as_behavior_builds_legacy_description(self) -> None: + with self.assertWarns(FutureWarning): + data = load_preset("behavior", "travel_planner") + self.assertEqual(data["kind"], "scenario") + self.assertIn("description", data) + self.assertIn(data["context"].strip(), data["description"]) + def test_load_kind_mismatch_raises(self) -> None: # safety-core is a judge_preset, not a behavior with self.assertRaises(ValueError):