diff --git a/CHANGELOG.md b/CHANGELOG.md index 73c0878d..a57acd4c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,15 @@ loosely while pre-1.0 (breaking changes can land on minor bumps). ## [Unreleased] +### Added +- **Constitution delegation policies.** Optional `delegation` object on an + agent constitution (`max_depth`, `allowed_callees`, `denied_callees`, + `max_subtree_tokens`, `max_subtree_usd`). The runtime fail-closes at + `/agent`, `/parallel`, and JIT `run_ephemeral` spawn with a tool `ERR:` + (same posture as the global depth cap of 2). Unknown/malformed policy + fields throw at parse. Stock `agents/*.json` stay unrestricted unless + an operator adds a block. Presence review does not consult the gate. + See [Delegation policies](docs/concepts/delegation.md). ## [0.13.11] — 2026-09-21 - **Fleet dashboard pane (Phase C1).** The TUI consumes fleet SSE (`stream_id` + diff --git a/CMakeLists.txt b/CMakeLists.txt index a1cf8fdc..91f558fb 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1664,6 +1664,8 @@ if(INDEX_BUILD_TESTS) ${CMAKE_SOURCE_DIR}/third_party/doctest ${OPENSSL_INCLUDE_DIR} ) + target_compile_definitions(unit_constitution PRIVATE + ARBITER_AGENTS_DIR="${CMAKE_SOURCE_DIR}/agents") target_link_libraries(unit_constitution PRIVATE OpenSSL::SSL OpenSSL::Crypto @@ -1671,6 +1673,48 @@ if(INDEX_BUILD_TESTS) ) add_test(NAME unit_constitution COMMAND unit_constitution) + # Delegation policy: parse lives in unit_constitution; this binary + # drives Orchestrator spawn denial (/agent, /parallel, JIT ephemeral) + # without a live LLM (gates return ERR before send_internal). + add_executable(unit_delegation + tests/test_delegation.cpp + src/orchestrator.cpp + src/agent.cpp + src/constitution.cpp + src/commands.cpp + src/context_compaction.cpp + src/model_catalog.cpp + src/model_context.cpp + src/advisor.cpp + src/advisor_gate.cpp + src/presence.cpp + src/intent.cpp + src/atomic_file.cpp + src/message_codec.cpp + src/tui/stream_filter.cpp + src/tui/block_parser.cpp + src/workspace_root.cpp + src/workspace_map.cpp + src/ssrf_guard.cpp + src/event_routing.cpp + src/api_client.cpp + src/circuit_breaker.cpp + src/metrics.cpp + src/json.cpp + ) + target_include_directories(unit_delegation PRIVATE + ${CMAKE_SOURCE_DIR}/include + ${CMAKE_SOURCE_DIR}/third_party/doctest + ${OPENSSL_INCLUDE_DIR} + ) + target_link_libraries(unit_delegation PRIVATE + OpenSSL::SSL + OpenSSL::Crypto + CURL::libcurl + Threads::Threads + ) + add_test(NAME unit_delegation COMMAND unit_delegation) + # Per-conversation Agent history isolation (ConversationScope / #40). add_executable(unit_agent_conversation tests/test_agent_conversation.cpp diff --git a/ROADMAP.md b/ROADMAP.md index c7b52094..2a7bf1ae 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -63,7 +63,7 @@ calendar commitments. - [x] **Always-on presence-** Constitution `presence.mode: always_on`; a pair colleague looks over a peer's shoulder after each tool batch and may inject `[PRESENCE: …]` context. Fail-open; cannot halt. SSE `presence` + TUI `◎ presence`. Opposite lifetime of JIT (#208). - [x] **Fleet dashboard pane-** Live tree of depth, agent, tools, tokens; click-to-focus / `Ctrl-w f` ([docs](docs/tui/fleet.md)) - [ ] **Plan to execution observability-** Planner plans as first-class objects with progress against todos -- [ ] **Delegation policies-** Consitutions declare max depth, allowed callees, budget caps (tokens/$) +- [x] **Delegation policies-** Constitutions declare max depth, allowed callees, budget caps (tokens/$) - [ ] ~~**Workflow recipes-** Checked-in “crews” (JSON): ordered/parallel graphs of agents + shared todo board~~ - [ ] **Advisor policy packs-** Reusable gate profiles (strict / coding / research) diff --git a/docs/api/agents/create.md b/docs/api/agents/create.md index a704ad3d..4bafadf9 100644 --- a/docs/api/agents/create.md +++ b/docs/api/agents/create.md @@ -47,6 +47,7 @@ Either a bare constitution or wrapped under `agent_def`: | `memory` | object | no | Per-agent memory enrichment toggles for `/mem search` and `/mem add entry`. See schema below and [Memory enrichment](../../concepts/structured-memory.md#memory-enrichment) in the structured-memory concept. | | `intent` | object | no | Pre-dispatch classify/route. Distinct from `memory.intent_routing`. File agents default `mode: "off"`; the built-in `index` master defaults `hybrid`. See [Intent](../../concepts/intent.md). | | `presence` | object \| string | no | Always-on residency. Object form: `{mode?, watch?, interject?, model?, prompt?, max_notes_per_turn?}`. String `"always_on"` is `{mode: "always_on"}`. See [Presence](../../concepts/presence.md). | +| `delegation` | object | no | Runtime spawn gates for `/agent`, `/parallel`, and JIT ensure covers. Absent keeps today's behaviour (global depth 2, any catalog callee, no subtree budget). Unknown keys fail closed. See [Delegation policies](../../concepts/delegation.md) and the schema below. | | `personality` | string | no | Free-form personality overlay. | #### `advisor` object schema @@ -94,6 +95,18 @@ Always-on peer observation. Absent / `mode: "off"` keeps a request-scoped specia | `prompt` | string | built-in | Override the review system prompt. | | `max_notes_per_turn` | int | `1` | Cap on `CONTEXT` notes per working-agent `stream_id` (1–4). | +#### `delegation` object schema + +Runtime-enforced spawn policy. Distinct from `max_tokens` (per-turn response size). Empty / omitted fields do not tighten the stock roster. + +| Sub-field | Type | Default | Notes | +|-----------|------|---------|-------| +| `max_depth` | int | `2` | Absolute child depth this agent may spawn to (0..2). `0` = cannot spawn and cannot run as a delegated / JIT worker. | +| `allowed_callees` | array\ | `[]` | Primary allowlist of agent ids. Empty / omitted = no extra restriction beyond catalog existence. | +| `denied_callees` | array\ | `[]` | Optional denylist applied after the allowlist. | +| `max_subtree_tokens` | int | unlimited | Cap on delegated work (children + descendants) for the current top-level turn. Fail closed with `ERR:` when spent ≥ cap. | +| `max_subtree_usd` | number | unlimited | Same window; coarse model-family USD estimate, not a billing ledger. | + ```bash curl -X POST \ -H "Authorization: Bearer atr_…" \ diff --git a/docs/cli/init.md b/docs/cli/init.md index 50638246..24cd3bf1 100644 --- a/docs/cli/init.md +++ b/docs/cli/init.md @@ -32,7 +32,7 @@ Index speaks in a **conversational** register (complete sentences, collaborative The starter JSON files are the **single source of truth** for what gets written. They live in `agents/` in the source tree and are embedded into the binary at build time. `--init` writes them verbatim — pretty-printed, in source order, byte-identical to the source tree — so the file you see on disk matches what a maintainer would see in the repo. -Each file is a plain JSON document — a model id, system prompt, tool allowlist, optional advisor block, optional `presence` residency, optional cost-attribution metadata. `jules` is the bundled [always-on presence](../concepts/presence.md) example (pair colleague, not a second advisor). Edit them in place, or copy one as the basis for your own agent. Drop a new `agents/.json` into the source tree and it'll show up in `--init` automatically on the next build (no code changes required). +Each file is a plain JSON document — a model id, system prompt, tool allowlist, optional advisor block, optional `presence` residency, optional `delegation` spawn policy, optional cost-attribution metadata. `jules` is the bundled [always-on presence](../concepts/presence.md) example (pair colleague, not a second advisor). Starters omit `delegation`, so they keep the global depth cap of 2 and may call any catalog agent; add a block to tighten. Edit them in place, or copy one as the basis for your own agent. Drop a new `agents/.json` into the source tree and it'll show up in `--init` automatically on the next build (no code changes required). ## Re-seeding from defaults diff --git a/docs/concepts/architecture.md b/docs/concepts/architecture.md index fba13721..bdb1a60c 100644 --- a/docs/concepts/architecture.md +++ b/docs/concepts/architecture.md @@ -37,7 +37,7 @@ flowchart LR subgraph EXECUTION["Agent execution"] direction TB - CONSTITUTION["Constitution
model · role · rules · tool allowlist"] + CONSTITUTION["Constitution
model · role · rules · tool allowlist
delegation policy"] AGENT["Agent"] @@ -124,6 +124,7 @@ For implementation details and deeper explanations, see: - [Reconcile](reconcile.md) - [Advisor](advisor.md) - [Presence](presence.md) +- [Delegation policies](delegation.md) - [Structured memory](structured-memory.md) - [MCP](mcp.md) - [A2A](a2a.md) diff --git a/docs/concepts/data-model.md b/docs/concepts/data-model.md index 9ea0034c..3b684504 100644 --- a/docs/concepts/data-model.md +++ b/docs/concepts/data-model.md @@ -80,6 +80,7 @@ Deleting a folder unfiles its conversations (`folder_id` cleared) rather than ca | `advisor` | object? | Structured advisor config: `{model, prompt?, mode?, max_redirects?, malformed_halts?}`. `mode: "consult"` (default) makes `/advise` available; `mode: "gate"` additionally enforces a runtime gate at the executor's terminating turn. See [advisor](advisor.md). | | `intent` | object? | Ingress classify/route: `{mode?, min_confidence?, apply_routing?, model?}`. Distinct from `memory.intent_routing`. See [intent](intent.md). | | `presence` | object? | Pair-colleague residency: `{mode?, watch?, interject?, model?, prompt?, max_notes_per_turn?}`. `mode: "always_on"` looks over a matching peer's shoulder after each tool batch and may inject a `[PRESENCE: …]` note. See [presence](presence.md). | +| `delegation` | object? | Runtime spawn gates for `/agent`, `/parallel`, and JIT ensure covers: `{max_depth?, allowed_callees?, denied_callees?, max_subtree_tokens?, max_subtree_usd?}`. Absent = global depth cap 2, any catalog callee, no subtree budget. Distinct from `max_tokens`. See [delegation policies](delegation.md). | | `advisor_model` | string? | **Legacy** shorthand for `advisor.model` with `mode: "consult"`. New configs should use `advisor`. | | `personality` | string? | Free-form personality overlay. | | `created_at` | integer | Epoch seconds. Stored agents only; absent for the built-in `index`. | diff --git a/docs/concepts/delegation.md b/docs/concepts/delegation.md new file mode 100644 index 00000000..75149382 --- /dev/null +++ b/docs/concepts/delegation.md @@ -0,0 +1,88 @@ +# Delegation policies + +Constitutions can declare **runtime** limits on who an agent may spawn, how deep the pipeline may go, and how much delegated work may cost. The orchestrator enforces them at `/agent`, `/parallel`, and JIT ensure cover spawn. The model does not get a vote — a violation is a tool `ERR:`, the same posture as the global depth cap of 2. + +This is not a new orchestrator. Conversational [`POST /v1/orchestrate`](../api/orchestrate.md) and reconcile observe/ensure stay on their existing loops; policy is a gate on spawn. + +## Why a runtime gate + +`capabilities` already decides whether an agent may emit `/agent` at all. That is a verb allowlist. Delegation policy is the *callee* allowlist (and budget) sitting next to it: + +- Index can be told it may not send work to `forge`. +- A specialist can be told it may not re-delegate (max depth 1), or may not run as a child at all (max depth 0). +- A token or spend cap on the delegated subtree stops further spawn once it is exhausted. + +Prompt text cannot do this. The runtime already refuses `depth >= 2` with `ERR: delegation depth limit reached (max 2 levels)` regardless of what the model wrote. Per-constitution policy uses that same fail-closed path. + +Presence is the opposite: it is fail-open and **cannot halt**. Presence review does not consult this gate. + +## Configuration + +The `delegation` object lives on a `Constitution`. Absent / omitted fields keep today's behaviour (global depth 2, any catalog callee, no subtree budget). Stock `agents/*.json` starters do not set the block. + +```jsonc +"delegation": { + "max_depth": 1, + "allowed_callees": ["scout", "vera"], + "denied_callees": ["forge"], + "max_subtree_tokens": 8000, + "max_subtree_usd": 0.50 +} +``` + +| Field | Type | Default | Notes | +|-------|------|---------|-------| +| `max_depth` | int 0..2 | `2` | Absolute pipeline depth this agent may spawn *to* (the child's depth). `0` = cannot spawn, and cannot itself run as a delegated or JIT worker. Must not exceed the global cap of 2. | +| `allowed_callees` | array\ | `[]` | **Primary allowlist.** Non-empty: callee id must be in the list. Empty / omitted = no extra restriction beyond catalog existence. | +| `denied_callees` | array\ | `[]` | Optional denylist applied *after* the allowlist. Empty / omitted = nobody extra is forbidden. Use this to forbid one id (`forge`) without listing the rest of the roster. | +| `max_subtree_tokens` | int ≥ 0 | unlimited | Cap on delegated work (children + their descendants) for the current top-level turn. Distinct from constitution `max_tokens` (response size). `0` = no spawn. Further spawn returns `ERR:` when spent ≥ cap; the in-flight child is not killed mid-turn. | +| `max_subtree_usd` | number ≥ 0 | unlimited | Same window as the token cap, using a coarse model-family USD estimate (Haiku / Sonnet / Opus / GPT / local=0; unknown hosted ≈ Sonnet). Not a billing ledger. | + +Unknown keys, wrong types, `max_depth` outside 0..2, or invalid agent ids **throw** at `Constitution::from_json` (admit fail closed). A malformed policy file is skipped by `load_agents`; `POST /v1/agents` returns 400. + +Both lists may be set: the callee must pass the allowlist (if any) and must not be on the denylist. + +## What the runtime checks + +On every spawn the caller is the agent that emitted `/agent` / `/parallel` (or `index` for JIT `run_ephemeral`). The child depth is `caller_depth + 1`. + +1. Global cap: `depth >= 2` still returns the historical `ERR: delegation depth limit reached (max 2 levels)` / `ERR: /parallel cannot delegate past depth 2`. +2. Self-invoke and `index` as callee — unchanged. +3. Catalog existence — unchanged (`ERR: no agent '…'`). +4. **Caller `max_depth`:** child depth must be ≤ the caller's effective cap. +5. **Caller `allowed_callees` / `denied_callees`.** +6. **Callee `max_depth`:** the child must be allowed to *run* at that depth (a `max_depth: 0` agent cannot be a depth-1 worker). +7. **Caller subtree budget** (tokens and/or USD) against spend recorded for this top-level turn. + +A child cannot re-delegate past *its* own `max_depth`. Index cannot send work to a forbidden callee. JIT ensure covers go through `run_ephemeral` at depth 1 with `index` as the caller, so a constitution on index that forbids `forge` (or a cover constitution with `max_depth: 0`) is honoured the same way as `/agent`. Nested `/agent` from a JIT clone uses that clone's constitution. + +The gate runs **before** the child LLM call. Presence review does not. + +## `ERR:` shapes + +``` +ERR: delegation depth limit reached (max 2 levels) +ERR: delegation depth limit reached (constitution max_depth 1) +ERR: callee 'forge' is not permitted by constitution.delegation.allowed_callees +ERR: callee 'forge' is forbidden by constitution.delegation.denied_callees +ERR: agent 'scout' cannot run at depth 1 (constitution max_depth 0) +ERR: delegation token budget exceeded (max_subtree_tokens 8000) +ERR: delegation spend budget exceeded (max_subtree_usd 0.5) +``` + +Do not retry the same spawn; the runtime will refuse it again. When a policy is present the system prompt also grows a `DELEGATION POLICY` block so the model sees the same numbers — that is hint, not enforcement. + +## What this is not + +- Not workflow recipes (ordered crews). +- Not advisor policy packs. +- Not plan-to-execution observability. +- Not a second fleet dashboard. A denied spawn is an `ERR:` tool result (`tool_call` `ok: false` on the existing stream). + +## See also + +- [Writ](writ.md) — `/agent` / `/parallel` verbs. +- [Reconcile](reconcile.md) — JIT ensure covers (`run_ephemeral`). +- [Fleet streaming](fleet-streaming.md) — depth 0 / 1 / 2. +- [Presence](presence.md) — fail-open; cannot halt. +- [`POST /v1/agents`](../api/agents/create.md) — constitution schema. diff --git a/docs/concepts/fleet-streaming.md b/docs/concepts/fleet-streaming.md index 5150da15..e1798bd2 100644 --- a/docs/concepts/fleet-streaming.md +++ b/docs/concepts/fleet-streaming.md @@ -12,7 +12,7 @@ Open a UI slot on `stream_start`, route every subsequent event with matching `st | 1 | A delegated sub-agent (via `/agent` or `/parallel`). | | 2 | A sub-sub-agent (delegation by a depth-1 agent). | -The depth cap is 2; attempts to delegate past depth 2 surface to the requesting agent as an `ERR:` tool result. +The depth cap is 2; attempts to delegate past depth 2 surface to the requesting agent as an `ERR:` tool result. A constitution may tighten that further with `delegation.max_depth` (0..2) and restrict callees / subtree budget — see [Delegation policies](delegation.md). Those gates also return `ERR:` and do not invent a third depth. ## When `/parallel` is in play @@ -59,7 +59,7 @@ done ok=true ## Parallel safety rails - **Same `agent_id` reused in `/parallel` is allowed.** Each child runs on an ephemeral `Agent` instance built from the canonical agent's `Constitution`, so siblings have independent `history_` vectors and don't race. (This was a constraint pre-2026-04 but is now lifted.) -- **Depth cap.** A depth-2 turn cannot `/parallel`; attempts return an ERR tool result. +- **Depth cap.** A depth-2 turn cannot `/parallel`; attempts return an ERR tool result. A constitution `delegation.max_depth` of 0 or 1 refuses earlier, with a constitution-scoped ERR. - **Each parallel child gets its own dedup cache.** Sibling threads fetching the same URL both fetch — accept the duplicate over a `std::map` data race. - **SSE writes are serialized.** A shared mutex on the wire-writer means events interleave cleanly even when N threads emit at once. diff --git a/docs/concepts/index.md b/docs/concepts/index.md index 2d26d97e..2ce7cfbd 100644 --- a/docs/concepts/index.md +++ b/docs/concepts/index.md @@ -10,6 +10,7 @@ Start here if you want the model before the reference material. | [Voice](voice.md) | Spoken register + `channel: "voice"` for Intercom-style bridges; PA memory habit | | [Advisor](advisor.md) | Structural supervision gates (`CONTINUE` / `REDIRECT` / `HALT`) | | [Presence](presence.md) | Pair colleague at a peer's shoulder — useful context mid-turn | +| [Delegation policies](delegation.md) | Per-constitution spawn gates: max depth, callees, subtree budget | | [Intent](intent.md) | Pre-dispatch classify/route (heuristic + optional LLM) | | [Reconcile](reconcile.md) | Desired end state → workspace contract, tests, rollback | | [SSE events](sse-events.md) | The stream contract shared by TUI and HTTP | diff --git a/docs/concepts/presence.md b/docs/concepts/presence.md index e8f7d578..cee2cc8b 100644 --- a/docs/concepts/presence.md +++ b/docs/concepts/presence.md @@ -65,7 +65,7 @@ The review is history-less — one snapshot in, one signal out — matching the one or two sentences the working agent should see now ``` -`SILENT` is the default. Surrounding prose is tolerated. Missing ``, an unknown token, or `CONTEXT` without `` is **malformed Silent** (fail-open). Presence cannot `HALT` or `REDIRECT` — that remains the advisor's job. +`SILENT` is the default. Surrounding prose is tolerated. Missing ``, an unknown token, or `CONTEXT` without `` is **malformed Silent** (fail-open). Presence cannot `HALT` or `REDIRECT` — that remains the advisor's job. Constitution `delegation` policy does not apply to presence review. ## Runtime control flow diff --git a/docs/concepts/reconcile.md b/docs/concepts/reconcile.md index 121d37f0..1d48750b 100644 --- a/docs/concepts/reconcile.md +++ b/docs/concepts/reconcile.md @@ -130,6 +130,8 @@ Same SSE fabric as orchestrate. Persist + replay via `request_status` / `request | `reconcile.done` | Structured result (`status`, `contract`, `delta`, `evidence`, `waves`). | | `done` | Terminal aggregate (`ok` true only when `status=satisfied`). | +JIT covers honour the covering agent's constitution **and** index's `delegation` policy: a constitution that forbids calling `forge` or `max_depth: 0` fail-closes `run_ephemeral` with `error_type: delegation_policy` before the clone's LLM turn. Nested `/agent` from a clone uses that clone's policy. See [Delegation policies](delegation.md). + TUI / `--send` do not call this path yet. Use [`POST /v1/reconcile`](../api/reconcile.md) or [`@arbiter/sdk`](../../sdk/ts/README.md). ## See also @@ -137,6 +139,7 @@ TUI / `--send` do not call this path yet. Use [`POST /v1/reconcile`](../api/reco - [`POST /v1/reconcile`](../api/reconcile.md) - [Intent](intent.md) — classify/route, not reconcile - [Presence](presence.md) — always-on residency; opposite of JIT +- [Delegation policies](delegation.md) — spawn gates on JIT covers - [Sandbox](sandbox.md) - [Durable execution](durable-execution.md) - ROADMAP Phase 5 diff --git a/docs/concepts/sse-events.md b/docs/concepts/sse-events.md index 329fca14..c4bcad19 100644 --- a/docs/concepts/sse-events.md +++ b/docs/concepts/sse-events.md @@ -18,7 +18,7 @@ Every event on the `/v1/orchestrate` stream has an `event:` line and a `data:` l | `stream_start` | Opens each turn. Fires for master + every delegated or parallel child. | `agent`, `stream_id`, `depth` (0 = master, 1 = delegated, 2 = sub-sub). | | `agent_start` | Just before each turn's outbound LLM request. | `agent`, `stream_id`, `depth`. | | `text` | Each clean (tool-call lines filtered out) delta from the model. Partial prose lines are emitted as they arrive; `/cmd` lines are held until a newline confirms them. After a delegation writ is parsed, a `→ delegating: …` status line is also emitted. | `agent`, `stream_id`, `depth` (master only — sub-agent text events only have `agent` + `stream_id`), `delta`. | -| `tool_call` | After each `/cmd` (fetch, search, browse, write, agent, parallel, mem, advise, exec) finishes. | `tool`, `ok`, `stream_id`, `depth`, `agent`. | +| `tool_call` | After each `/cmd` (fetch, search, browse, write, agent, parallel, mem, advise, exec) finishes. | `tool`, `ok`, `stream_id`, `depth`, `agent`. Denied `/agent` / `/parallel` spawns (depth cap, constitution `delegation`) still emit `ok: false` with the `ERR:` in the tool envelope. | | `file` | Each time the agent emits a `/write` block; content is captured in-memory and forwarded here instead of written to disk. | `path`, `size`, `encoding` (always `"utf-8"`), `content`, `stream_id`, `depth`, `agent`. | | `sub_agent_response` | After a delegated turn completes (depth > 0). The full turn body in one payload — useful for consumers that don't want to reconstruct from deltas. | `agent`, `stream_id`, `depth`, `content`. | | `token_usage` | After each turn completes. | `agent`, `stream_id`, `depth`, `model`, `input_tokens`, `output_tokens`, `cache_read_tokens?`, `cache_create_tokens?`. | @@ -64,6 +64,7 @@ Spec-compatible A2A clients hit [`POST /v1/a2a/agents/:id`](../api/a2a/dispatch. - [TUI fleet dashboard](../tui/fleet.md) - [Advisor](advisor.md) — gate signal grammar, modes, redirect budget. - [Presence](presence.md) — always-on peer review and mid-turn injection. +- [Delegation policies](delegation.md) — constitution spawn gates (`ERR:` on `/agent` / `/parallel` / JIT). - [Intent](intent.md) — pre-dispatch classify/route. - [Reconcile](reconcile.md) — desired-end-state contract, tests, rollback. - [Voice](voice.md) — spoken constitution register and `channel: "voice"` for TTS bridges. diff --git a/docs/concepts/writ.md b/docs/concepts/writ.md index 41ad07df..65e8aac7 100644 --- a/docs/concepts/writ.md +++ b/docs/concepts/writ.md @@ -54,8 +54,9 @@ Every agent's [Constitution](../api/agents/create.md) declares a `capabilities` - **Rules** (soft, prompt-level) tell the model *when* to use a writ. - **Capabilities** (hard, runtime-level) decide whether a writ the model emits will do anything. +- **Delegation policy** (hard, runtime-level) decides *which agents* a permitted `/agent` / `/parallel` may spawn, how deep, and on what subtree budget. See [Delegation policies](delegation.md). -You can't social-engineer a `/exec` out of an agent that doesn't have it in its warrant. A research agent emits the same writ syntax as a backend agent; only the dispatched subset differs. +You can't social-engineer a `/exec` out of an agent that doesn't have it in its warrant. You also can't `/agent forge` past a constitution that forbids that callee — the runtime returns `ERR:` the same way it refuses `depth >= 2`. A research agent emits the same writ syntax as a backend agent; only the dispatched subset (and the callee/budget gate) differs. ## Image content in tool results diff --git a/include/constitution.h b/include/constitution.h index cd287886..ac042ede 100644 --- a/include/constitution.h +++ b/include/constitution.h @@ -141,6 +141,51 @@ struct Constitution { // docs/concepts/presence.md. PresenceConfig presence; + // Runtime-enforced spawn gates for `/agent`, `/parallel`, and JIT + // ensure covers. Distinct from `max_tokens` (per-turn response size). + // Absent / default = today's behaviour (global depth cap 2, any + // catalog callee, no subtree budget). The runtime fail-closes with + // an `ERR:` tool result — do not rely on the model to obey this text. + // See docs/concepts/delegation.md. + struct DelegationPolicy { + static constexpr int kGlobalMaxDepth = 2; + + // Absolute pipeline depth this agent may spawn to (child depth). + // 0 = cannot spawn (and cannot itself run as a delegated / JIT + // worker). Omitted → kGlobalMaxDepth. Must not exceed 2. + std::optional max_depth; + + // Primary allowlist of agent ids this constitution may `/agent` + // or `/parallel` (and that JIT covers may instantiate when this + // constitution is the caller). Empty / omitted = no extra + // restriction beyond catalog existence. + std::vector allowed_callees; + + // Optional denylist applied after the allowlist. Empty / omitted + // = nobody extra is forbidden. Use this to forbid one callee + // (e.g. `forge`) without listing the rest of the roster. + std::vector denied_callees; + + // Caps on delegated work (children + their descendants) for the + // current top-level turn. Omitted = unlimited. 0 = no further + // spawn allowed. Fail closed when spent >= cap. + std::optional max_subtree_tokens; + std::optional max_subtree_usd; + + int effective_max_depth() const { + int d = max_depth.value_or(kGlobalMaxDepth); + if (d < 0) return 0; + if (d > kGlobalMaxDepth) return kGlobalMaxDepth; + return d; + } + + bool is_default() const { + return !max_depth && allowed_callees.empty() && + denied_callees.empty() && !max_subtree_tokens && + !max_subtree_usd; + } + } delegation; + // --- System prompt pieces --- std::string goal; // what this agent is trying to accomplish std::vector rules; // explicit behavioral constraints @@ -181,4 +226,24 @@ Constitution master_constitution(); std::string brevity_to_string(Brevity b); Brevity brevity_from_string(const std::string& s); +// Spend so far on delegated work attributed to `caller_id` this turn. +struct DelegationSpend { + int tokens = 0; + double usd = 0.0; +}; + +// Crude USD estimate for constitution spend caps. Same family rates the +// TUI sidebar uses (Haiku / Sonnet / Opus / GPT / local=0); unknown hosted +// models fall back to Sonnet-equivalent. Not a billing ledger. +double delegation_estimate_usd(std::string_view model, int in_tokens, int out_tokens); + +// Runtime spawn gate. Empty = allowed. Otherwise a tool `ERR:` line +// (already prefixed). `callee` may be null when the catalog row has not +// been loaded yet — callee-side max_depth is skipped in that case. +std::string delegation_spawn_error(const Constitution& caller, + const Constitution* callee, + const std::string& callee_id, + int child_depth, + const DelegationSpend& spent); + } // namespace arbiter diff --git a/include/orchestrator.h b/include/orchestrator.h index c9c79c6a..495682f8 100644 --- a/include/orchestrator.h +++ b/include/orchestrator.h @@ -433,6 +433,21 @@ class Orchestrator { const std::string& message, const std::string& original_query = ""); + // Runtime spawn gate used by `/agent`, `/parallel`, and JIT + // `run_ephemeral`. Empty = allowed. Non-empty is a tool `ERR:` line. + // Looks up the callee constitution when the id is registered (so a + // child's own max_depth is honoured). Does not call the model. + std::string check_delegation_spawn(const std::string& caller_id, + const Constitution& caller_cfg, + const std::string& callee_id, + int child_depth) const; + + // Attribute delegated-work spend to `caller_id` for this top-level + // turn. Tests seed this to exercise budget denial without an LLM. + void record_delegation_spend(const std::string& caller_id, + int tokens, double usd); + DelegationSpend delegation_spend_for(const std::string& caller_id) const; + // True after cancel() until the next send()/send_streaming() exits. // Survives ApiClient::stream()/complete() clearing their own cancelled // flag at call entry — used so an admin kill-switch during pre-send @@ -552,6 +567,13 @@ class Orchestrator { int next_stream_id(); void fire_history_checkpoint(); + // Delegated-work spend this top-level turn, keyed by caller agent id. + // Cleared at send() / send_streaming() / run_ephemeral entry. + mutable std::mutex delegation_spend_mu_; + std::map delegation_spend_; + void clear_delegation_spend(); + const Constitution* find_constitution(const std::string& id) const; + // True when this turn should stop: sticky cancel, hard-cancel, or the // thread-local RequestCancelScope token. Rechecked at each dispatch // iteration so a kill during tools does not start another LLM call. @@ -600,11 +622,13 @@ class Orchestrator { const std::string& original_query); // Build an AgentInvoker lambda for use in command dispatch. - // depth is the current nesting level; invoker refuses beyond depth 2. + // depth is the current nesting level; invoker refuses beyond depth 2 + // and honours the caller's constitution.delegation policy. // shared_cache and original_query propagate through the delegation chain. AgentInvoker make_invoker(const std::string& caller_id, int depth, std::map* shared_cache, - const std::string& original_query); + const std::string& original_query, + const Constitution& caller_cfg); // Build a ParallelInvoker for /parallel fan-out. Each child runs on its // own std::thread at depth+1 with a fresh dedup cache (shared caches @@ -614,7 +638,8 @@ class Orchestrator { // before the returned vector is filled — /parallel blocks the calling // turn until every child completes. ParallelInvoker make_parallel_invoker(const std::string& caller_id, int depth, - const std::string& original_query); + const std::string& original_query, + const Constitution& caller_cfg); // Consult always-on watchers whose `watch` globs match `working_agent_id` // and prepend any CONTEXT notes. Fail-open; budget-capped. `notes_used` diff --git a/src/constitution.cpp b/src/constitution.cpp index d85f425c..5a949a6d 100644 --- a/src/constitution.cpp +++ b/src/constitution.cpp @@ -2,7 +2,9 @@ #include "constitution.h" #include "api_client.h" // is_weak_executor #include "json.h" +#include #include +#include #include #include #include @@ -68,6 +70,177 @@ Brevity brevity_from_string(const std::string& s) { return Brevity::Full; } +namespace { + +bool starts_with_ci(std::string_view hay, std::string_view needle) { + if (hay.size() < needle.size()) return false; + for (size_t i = 0; i < needle.size(); ++i) { + if (std::tolower(static_cast(hay[i])) + != std::tolower(static_cast(needle[i]))) + return false; + } + return true; +} + +bool json_is_int(const JsonValue& v) { + if (!v.is_number()) return false; + double d = v.as_number(); + if (!std::isfinite(d)) return false; + return d == std::floor(d); +} + +bool valid_callee_id(std::string_view id) { + if (id.empty() || id.size() > 64) return false; + for (char c : id) { + if (!std::isalnum(static_cast(c)) && c != '_' && c != '-') + return false; + } + return true; +} + +std::vector parse_callee_ids(const JsonValue& arr, const char* field) { + if (!arr.is_array()) + throw std::runtime_error(std::string("constitution: delegation.") + field + + " must be an array of agent ids"); + std::vector out; + for (auto& v : arr.as_array()) { + if (!v || !v->is_string()) + throw std::runtime_error(std::string("constitution: delegation.") + field + + " entries must be strings"); + const std::string& id = v->as_string(); + if (!valid_callee_id(id)) + throw std::runtime_error(std::string("constitution: delegation.") + field + + " contains invalid agent id '" + id + "'"); + out.push_back(id); + } + return out; +} + +Constitution::DelegationPolicy parse_delegation(const JsonValue& obj) { + static const std::set kKnown = { + "max_depth", "allowed_callees", "denied_callees", + "max_subtree_tokens", "max_subtree_usd" + }; + for (const auto& [key, val] : obj.as_object()) { + if (!kKnown.count(key)) + throw std::runtime_error( + "constitution: delegation unknown field '" + key + "'"); + (void)val; + } + + Constitution::DelegationPolicy p; + + if (auto v = obj.get("max_depth")) { + if (!json_is_int(*v)) + throw std::runtime_error( + "constitution: delegation.max_depth must be an integer 0..2"); + int d = static_cast(v->as_number()); + if (d < 0 || d > Constitution::DelegationPolicy::kGlobalMaxDepth) + throw std::runtime_error( + "constitution: delegation.max_depth must be an integer 0..2"); + p.max_depth = d; + } + if (auto v = obj.get("allowed_callees")) + p.allowed_callees = parse_callee_ids(*v, "allowed_callees"); + if (auto v = obj.get("denied_callees")) + p.denied_callees = parse_callee_ids(*v, "denied_callees"); + if (auto v = obj.get("max_subtree_tokens")) { + if (!json_is_int(*v) || v->as_number() < 0) + throw std::runtime_error( + "constitution: delegation.max_subtree_tokens must be a " + "non-negative integer"); + p.max_subtree_tokens = static_cast(v->as_number()); + } + if (auto v = obj.get("max_subtree_usd")) { + if (!v->is_number() || !std::isfinite(v->as_number()) || v->as_number() < 0) + throw std::runtime_error( + "constitution: delegation.max_subtree_usd must be a " + "non-negative number"); + p.max_subtree_usd = v->as_number(); + } + return p; +} + +} // namespace + +double delegation_estimate_usd(std::string_view model, int in_tokens, int out_tokens) { + if (in_tokens <= 0 && out_tokens <= 0) return 0.0; + double in_per_m = 3.0; + double out_per_m = 15.0; + if (starts_with_ci(model, "ollama/") || starts_with_ci(model, "local/")) { + in_per_m = 0.0; + out_per_m = 0.0; + } else if (model.find("haiku") != std::string_view::npos) { + in_per_m = 0.8; + out_per_m = 4.0; + } else if (model.find("opus") != std::string_view::npos) { + in_per_m = 15.0; + out_per_m = 75.0; + } else if (model.find("sonnet") != std::string_view::npos) { + in_per_m = 3.0; + out_per_m = 15.0; + } else if (model.find("gpt-4o-mini") != std::string_view::npos) { + in_per_m = 0.15; + out_per_m = 0.6; + } else if (model.find("gpt-4o") != std::string_view::npos || + model.find("gpt-") != std::string_view::npos) { + in_per_m = 2.5; + out_per_m = 10.0; + } + return (static_cast(std::max(0, in_tokens)) / 1'000'000.0) * in_per_m + + (static_cast(std::max(0, out_tokens)) / 1'000'000.0) * out_per_m; +} + +std::string delegation_spawn_error(const Constitution& caller, + const Constitution* callee, + const std::string& callee_id, + int child_depth, + const DelegationSpend& spent) { + const auto& pol = caller.delegation; + const int cap = pol.effective_max_depth(); + if (child_depth > cap) { + if (cap >= Constitution::DelegationPolicy::kGlobalMaxDepth) + return "ERR: delegation depth limit reached (max 2 levels)"; + return "ERR: delegation depth limit reached (constitution max_depth " + + std::to_string(cap) + ")"; + } + if (!pol.allowed_callees.empty()) { + bool ok = false; + for (const auto& id : pol.allowed_callees) { + if (id == callee_id) { ok = true; break; } + } + if (!ok) + return "ERR: callee '" + callee_id + + "' is not permitted by constitution.delegation.allowed_callees"; + } + for (const auto& id : pol.denied_callees) { + if (id == callee_id) + return "ERR: callee '" + callee_id + + "' is forbidden by constitution.delegation.denied_callees"; + } + if (callee) { + int callee_cap = callee->delegation.effective_max_depth(); + if (child_depth > callee_cap) { + return "ERR: agent '" + callee_id + + "' cannot run at depth " + std::to_string(child_depth) + + " (constitution max_depth " + std::to_string(callee_cap) + ")"; + } + } + if (pol.max_subtree_tokens && + spent.tokens >= *pol.max_subtree_tokens) { + return "ERR: delegation token budget exceeded (max_subtree_tokens " + + std::to_string(*pol.max_subtree_tokens) + ")"; + } + if (pol.max_subtree_usd && + spent.usd >= *pol.max_subtree_usd) { + std::ostringstream os; + os << "ERR: delegation spend budget exceeded (max_subtree_usd " + << *pol.max_subtree_usd << ")"; + return os.str(); + } + return {}; +} + // ─── Voice + brevity ───────────────────────────────────────────────────────── // Specialists keep a compressed register (token-efficient field reports). // Index uses a conversational register — users talk to the orchestrator, not @@ -1062,6 +1235,40 @@ std::string Constitution::build_system_prompt() const { ss << "- " << r << "\n"; } + // Layer 3a: runtime delegation policy. Stock agents omit this so + // their prompt is unchanged. When present, tell the model the + // runtime will ERR rather than hoping it remembers the JSON. + if (!delegation.is_default()) { + ss << "\nDELEGATION POLICY (runtime-enforced — violating it returns ERR):\n"; + ss << "- Pipeline depth cap: " << delegation.effective_max_depth() + << " (global max " + << Constitution::DelegationPolicy::kGlobalMaxDepth << ").\n"; + if (!delegation.allowed_callees.empty()) { + ss << "- You may spawn only: "; + for (size_t i = 0; i < delegation.allowed_callees.size(); ++i) { + if (i) ss << ", "; + ss << delegation.allowed_callees[i]; + } + ss << ".\n"; + } + if (!delegation.denied_callees.empty()) { + ss << "- You must not spawn: "; + for (size_t i = 0; i < delegation.denied_callees.size(); ++i) { + if (i) ss << ", "; + ss << delegation.denied_callees[i]; + } + ss << ".\n"; + } + if (delegation.max_subtree_tokens) + ss << "- Delegated subtree token cap: " + << *delegation.max_subtree_tokens << ".\n"; + if (delegation.max_subtree_usd) + ss << "- Delegated subtree spend cap: $" + << *delegation.max_subtree_usd << ".\n"; + ss << "- Do not retry a spawn that returned ERR: the runtime will " + "refuse it again.\n"; + } + // Layer 3b: always-on presence. The dedicated review prompt is what // the runtime actually calls; this block tells a directly-addressed // presence agent what its residency means so it does not start acting @@ -1393,6 +1600,34 @@ std::string Constitution::to_json() const { m["intent"] = ic; } + // Delegation policy — only emit when it tightens the default (global + // depth 2, no callee list, no subtree budget) so stock agents stay + // compact. + if (!delegation.is_default()) { + auto dc = jobj(); + auto& dco = dc->as_object_mut(); + if (delegation.max_depth) + dco["max_depth"] = jnum(static_cast(*delegation.max_depth)); + if (!delegation.allowed_callees.empty()) { + auto a = jarr(); + for (auto& id : delegation.allowed_callees) + a->as_array_mut().push_back(jstr(id)); + dco["allowed_callees"] = a; + } + if (!delegation.denied_callees.empty()) { + auto a = jarr(); + for (auto& id : delegation.denied_callees) + a->as_array_mut().push_back(jstr(id)); + dco["denied_callees"] = a; + } + if (delegation.max_subtree_tokens) + dco["max_subtree_tokens"] = + jnum(static_cast(*delegation.max_subtree_tokens)); + if (delegation.max_subtree_usd) + dco["max_subtree_usd"] = jnum(*delegation.max_subtree_usd); + m["delegation"] = dc; + } + return json_serialize(*obj); } @@ -1544,6 +1779,17 @@ Constitution Constitution::from_json(const std::string& json_str) { } } + // Delegation policy. Absent → defaults (global depth 2, no extra + // callee/budget gates). Present but malformed → throw (admit fail + // closed). Unknown keys are rejected so a typo cannot silently + // disable a gate. + auto delegation_val = root->get("delegation"); + if (delegation_val) { + if (!delegation_val->is_object()) + throw std::runtime_error("constitution: delegation must be an object"); + c.delegation = parse_delegation(*delegation_val); + } + auto rules_val = root->get("rules"); if (rules_val && rules_val->is_array()) { for (auto& r : rules_val->as_array()) { diff --git a/src/orchestrator.cpp b/src/orchestrator.cpp index 1d26edf8..285b3af2 100644 --- a/src/orchestrator.cpp +++ b/src/orchestrator.cpp @@ -166,16 +166,72 @@ void Orchestrator::load_agents(const std::string& dir) { } } +void Orchestrator::clear_delegation_spend() { + std::lock_guard lk(delegation_spend_mu_); + delegation_spend_.clear(); +} + +void Orchestrator::record_delegation_spend(const std::string& caller_id, + int tokens, double usd) { + if (caller_id.empty()) return; + std::lock_guard lk(delegation_spend_mu_); + auto& s = delegation_spend_[caller_id]; + s.tokens += std::max(0, tokens); + s.usd += std::max(0.0, usd); +} + +DelegationSpend Orchestrator::delegation_spend_for(const std::string& caller_id) const { + std::lock_guard lk(delegation_spend_mu_); + auto it = delegation_spend_.find(caller_id); + if (it == delegation_spend_.end()) return {}; + return it->second; +} + +const Constitution* Orchestrator::find_constitution(const std::string& id) const { + if (id == "index") return &index_master_->config(); + std::lock_guard lock(agents_mutex_); + auto it = agents_.find(id); + if (it == agents_.end()) return nullptr; + return &it->second->config(); +} + +std::string Orchestrator::check_delegation_spawn( + const std::string& caller_id, + const Constitution& caller_cfg, + const std::string& callee_id, + int child_depth) const { + const Constitution* callee = find_constitution(callee_id); + return delegation_spawn_error(caller_cfg, callee, callee_id, child_depth, + delegation_spend_for(caller_id)); +} + +namespace { + +void credit_caller_spend(Orchestrator& orch, + const std::string& caller_id, + const std::string& callee_id, + const std::string& callee_model, + const ApiResponse& resp) { + const int tok = std::max(0, resp.input_tokens) + std::max(0, resp.output_tokens); + const double usd = delegation_estimate_usd( + callee_model, resp.input_tokens, resp.output_tokens); + DelegationSpend nested = orch.delegation_spend_for(callee_id); + orch.record_delegation_spend(caller_id, tok + nested.tokens, usd + nested.usd); +} + +} // namespace + // Build an AgentInvoker that runs a sub-agent through the full dispatch loop. AgentInvoker Orchestrator::make_invoker(const std::string& caller_id, int depth, std::map* shared_cache, - const std::string& original_query) { - if (depth >= 2) { + const std::string& original_query, + const Constitution& caller_cfg) { + if (depth >= Constitution::DelegationPolicy::kGlobalMaxDepth) { return [](const std::string&, const std::string&) -> std::string { return "ERR: delegation depth limit reached (max 2 levels)"; }; } - return [this, caller_id, depth, shared_cache, original_query]( + return [this, caller_id, depth, shared_cache, original_query, caller_cfg]( const std::string& sub_id, const std::string& sub_msg) -> std::string { if (sub_id == caller_id) return "ERR: agent cannot invoke itself"; if (sub_id == "index") return "ERR: index cannot be delegated to"; @@ -184,6 +240,10 @@ AgentInvoker Orchestrator::make_invoker(const std::string& caller_id, int depth, if (!agents_.count(sub_id)) return "ERR: no agent '" + sub_id + "'"; } + if (auto err = check_delegation_spawn(caller_id, caller_cfg, sub_id, + depth + 1); !err.empty()) { + return err; + } // Inject delegation context so sub-agent knows the user's goal // and its position in the pipeline. @@ -257,21 +317,29 @@ AgentInvoker Orchestrator::make_invoker(const std::string& caller_id, int depth, // Shared cache propagates so sub-agents don't re-fetch URLs. auto resp = send_internal(sub_id, enriched_msg, depth + 1, shared_cache, original_query); + std::string model; + { + std::lock_guard lk(agents_mutex_); + auto it = agents_.find(sub_id); + if (it != agents_.end()) model = it->second->config().model; + } + credit_caller_spend(*this, caller_id, sub_id, model, resp); return resp.ok ? resp.content : "ERR: " + resp.error; }; } ParallelInvoker Orchestrator::make_parallel_invoker(const std::string& caller_id, int depth, - const std::string& original_query) { - if (depth >= 2) { + const std::string& original_query, + const Constitution& caller_cfg) { + if (depth >= Constitution::DelegationPolicy::kGlobalMaxDepth) { return [](const std::vector>& kids) { return std::vector( kids.size(), "ERR: /parallel cannot delegate past depth 2"); }; } - return [this, caller_id, depth, original_query]( + return [this, caller_id, depth, original_query, caller_cfg]( const std::vector>& kids) -> std::vector { if (kids.size() > kMaxParallelChildren) { @@ -341,7 +409,7 @@ ParallelInvoker Orchestrator::make_parallel_invoker(const std::string& caller_id const std::string sub_id = kids[i].first; const std::string sub_msg = kids[i].second; threads.emplace_back([this, i, sub_id, sub_msg, caller_id, depth, - original_query, &results, &child_clients, + original_query, caller_cfg, &results, &child_clients, pane_binder, conv_key, parent_token]() { if (pane_binder) pane_binder(); ConversationScope scope(conv_key); @@ -371,6 +439,11 @@ ParallelInvoker Orchestrator::make_parallel_invoker(const std::string& caller_id // returned cfg_copy is owned solely by this thread. cfg_copy = it->second->config(); } + if (auto err = check_delegation_spawn(caller_id, caller_cfg, sub_id, + depth + 1); !err.empty()) { + results[i] = err; + return; + } // Match make_invoker's delegation-context prelude so the // sub-agent has the same framing whether it was called @@ -444,6 +517,8 @@ ParallelInvoker Orchestrator::make_parallel_invoker(const std::string& caller_id try { auto resp = run_dispatch(ephemeral, sub_id, enriched_msg, depth + 1, &local_cache, orig_q); + credit_caller_spend(*this, caller_id, sub_id, + ephemeral.config().model, resp); results[i] = resp.ok ? resp.content : "ERR: " + resp.error; } catch (const std::exception& e) { results[i] = std::string("ERR: ") + e.what(); @@ -488,9 +563,27 @@ ApiResponse Orchestrator::run_ephemeral(const std::string& agent_id, return err; } + // JIT covers are spawned at depth 1. Honour the index caller's + // constitution (stock index has no extra policy) and the cover's own + // max_depth so a constitution that forbids `forge` or depth>1 is + // fail-closed here the same way /agent is. + { + std::string err = delegation_spawn_error( + index_master_->config(), &cfg, agent_id, /*child_depth=*/1, + delegation_spend_for("index")); + if (!err.empty()) { + ApiResponse r; + r.ok = false; + r.error = err; + r.error_type = "delegation_policy"; + return r; + } + } + // Subset validation is the supervisor on this path. Advisor gate // CONTINUE/REDIRECT/HALT and presence residency must not hitch a ride - // on a JIT clone. + // on a JIT clone. Delegation policy stays — nested /agent from the + // clone still fail-closes. cfg.advisor.mode = "off"; cfg.presence.mode = "off"; cfg.intent.mode = "off"; @@ -521,6 +614,7 @@ ApiResponse Orchestrator::run_ephemeral(const std::string& agent_id, std::string orig_q = original_query.empty() ? message : original_query; out = run_dispatch(ephemeral, agent_id, message, /*depth=*/1, &local_cache, orig_q); + credit_caller_spend(*this, "index", agent_id, ephemeral.config().model, out); // Drop clone residency explicitly; destructor would too. ephemeral.reset_all_histories(); } catch (const std::exception& e) { @@ -675,6 +769,9 @@ std::string Orchestrator::collect_presence_notes( const std::string& tool_summary, int stream_id, std::map& notes_used) { + // Presence is fail-open and cannot halt. Do not consult + // constitution.delegation here — a spawn policy must not turn a + // pair-colleague review into an ERR / HALT. struct Watcher { std::string id; @@ -1071,9 +1168,11 @@ ApiResponse Orchestrator::run_dispatch(Agent& agent, } catch (...) { /* never let todo probe break dispatch */ } } - auto invoker = make_invoker(agent_id, depth, shared_cache, orig_q); + auto invoker = make_invoker(agent_id, depth, shared_cache, orig_q, + agent.config()); auto advisor_invoker = make_advisor_invoker(agent_id); - auto parallel_invoker = make_parallel_invoker(agent_id, depth, orig_q); + auto parallel_invoker = make_parallel_invoker(agent_id, depth, orig_q, + agent.config()); // Gate-mode advisor wiring. Built lazily — if the agent's advisor // config is anything other than mode == "gate", the gate is never @@ -1428,6 +1527,7 @@ ApiResponse Orchestrator::send(const std::string& agent_id, r.error = "cancelled"; return r; } + clear_delegation_spend(); return send_internal(agent_id, message, 0, nullptr, original_query); } @@ -1520,6 +1620,7 @@ ApiResponse Orchestrator::send_streaming(const std::string& agent_id, r.error = "cancelled"; return r; } + clear_delegation_spend(); Agent* agent_ptr; std::vector current_parts; @@ -1752,9 +1853,11 @@ ApiResponse Orchestrator::send_streaming(const std::string& agent_id, end_iteration(cmds); std::map shared_cache; - auto invoker = make_invoker(dispatch_id, 0, &shared_cache, orig_q); + auto invoker = make_invoker(dispatch_id, 0, &shared_cache, orig_q, + agent_ptr->config()); auto advisor_invoker = make_advisor_invoker(dispatch_id); - auto parallel_invoker = make_parallel_invoker(dispatch_id, 0, orig_q); + auto parallel_invoker = make_parallel_invoker(dispatch_id, 0, orig_q, + agent_ptr->config()); // Gate-mode wiring (master / top-level). Same construction as // run_dispatch — see the longer comment there for the reasoning. @@ -2434,9 +2537,11 @@ std::string Orchestrator::execute_slash_command(const std::string& line, } std::map dedup_cache; - auto invoker = make_invoker(agent_id, 0, &dedup_cache, ""); + auto invoker = make_invoker(agent_id, 0, &dedup_cache, "", + agent_ptr->config()); auto advisor_invoker = make_advisor_invoker(agent_id); - auto parallel_invoker = make_parallel_invoker(agent_id, 0, ""); + auto parallel_invoker = make_parallel_invoker(agent_id, 0, "", + agent_ptr->config()); CommandCancelScope cancel_scope([this] { return turn_is_cancelled(); }); return execute_agent_commands(cmds, agent_id, memory_dir_, diff --git a/tests/test_constitution.cpp b/tests/test_constitution.cpp index e7f2fea3..6f5710d1 100644 --- a/tests/test_constitution.cpp +++ b/tests/test_constitution.cpp @@ -12,6 +12,10 @@ #include "constitution.h" #include "presence.h" +#include +#include +#include + using namespace arbiter; // Helper: construct a minimal Constitution and run the composer. Anchored @@ -582,3 +586,148 @@ TEST_CASE("file_backed_agent_id prefers stem for Title Case display names") { c.name = "research"; CHECK(file_backed_agent_id(c, "scout") == "research"); } + +TEST_CASE("delegation: absent yields unrestricted defaults") { + auto c = Constitution::from_json(R"({"name":"scout","model":"claude-sonnet-4-6"})"); + CHECK(c.delegation.is_default()); + CHECK_FALSE(c.delegation.max_depth.has_value()); + CHECK(c.delegation.effective_max_depth() == + Constitution::DelegationPolicy::kGlobalMaxDepth); + CHECK(c.delegation.allowed_callees.empty()); + CHECK(c.delegation.denied_callees.empty()); + CHECK_FALSE(c.delegation.max_subtree_tokens.has_value()); + CHECK_FALSE(c.delegation.max_subtree_usd.has_value()); + CHECK(c.to_json().find("\"delegation\"") == std::string::npos); + CHECK(c.build_system_prompt().find("DELEGATION POLICY") == std::string::npos); +} + +TEST_CASE("delegation: round-trip allowlist, denylist, depth, budgets") { + std::string js = R"({ + "name": "lead", + "model": "claude-sonnet-4-6", + "delegation": { + "max_depth": 1, + "allowed_callees": ["scout", "vera"], + "denied_callees": ["forge"], + "max_subtree_tokens": 8000, + "max_subtree_usd": 0.5 + } + })"; + auto c = Constitution::from_json(js); + CHECK(c.delegation.max_depth == 1); + CHECK(c.delegation.effective_max_depth() == 1); + CHECK(c.delegation.allowed_callees == std::vector({"scout", "vera"})); + CHECK(c.delegation.denied_callees == std::vector({"forge"})); + CHECK(c.delegation.max_subtree_tokens == 8000); + CHECK(c.delegation.max_subtree_usd == doctest::Approx(0.5)); + auto again = Constitution::from_json(c.to_json()); + CHECK(again.delegation.max_depth == 1); + CHECK(again.delegation.allowed_callees == c.delegation.allowed_callees); + CHECK(again.delegation.denied_callees == c.delegation.denied_callees); + CHECK(again.delegation.max_subtree_tokens == 8000); + CHECK(again.delegation.max_subtree_usd == doctest::Approx(0.5)); + CHECK(c.build_system_prompt().find("DELEGATION POLICY") != std::string::npos); + CHECK(c.build_system_prompt().find("must not spawn: forge") != std::string::npos); +} + +TEST_CASE("delegation: unknown and malformed fields fail closed") { + auto throws = [](const std::string& js) { + CHECK_THROWS_AS(Constitution::from_json(js), std::runtime_error); + }; + throws(R"({"name":"x","delegation":"deny"})"); + throws(R"({"name":"x","delegation":[]})"); + throws(R"({"name":"x","delegation":{"max_depth":3}})"); + throws(R"({"name":"x","delegation":{"max_depth":-1}})"); + throws(R"({"name":"x","delegation":{"max_depth":1.5}})"); + throws(R"({"name":"x","delegation":{"max_depth":"1"}})"); + throws(R"({"name":"x","delegation":{"allowed_callees":"scout"}})"); + throws(R"({"name":"x","delegation":{"allowed_callees":[1]}})"); + throws(R"({"name":"x","delegation":{"allowed_callees":[""]}})"); + throws(R"({"name":"x","delegation":{"allowed_callees":["bad id"]}})"); + throws(R"({"name":"x","delegation":{"denied_callees":{"forge":true}}})"); + throws(R"({"name":"x","delegation":{"max_subtree_tokens":-5}})"); + throws(R"({"name":"x","delegation":{"max_subtree_usd":-0.1}})"); + throws(R"({"name":"x","delegation":{"max_tokens":100}})"); + throws(R"({"name":"x","delegation":{"allow_list":["scout"]}})"); +} + +TEST_CASE("delegation_spawn_error: defaults allow depth 1-2") { + Constitution caller; + CHECK(delegation_spawn_error(caller, nullptr, "forge", 1, {}).empty()); + CHECK(delegation_spawn_error(caller, nullptr, "forge", 2, {}).empty()); + CHECK(delegation_spawn_error(caller, nullptr, "forge", 3, {}).find("ERR:") == 0); +} + +TEST_CASE("delegation_spawn_error: max_depth 0/1 and forbidden callee") { + Constitution caller; + caller.delegation.max_depth = 0; + auto err = delegation_spawn_error(caller, nullptr, "forge", 1, {}); + CHECK(err.find("ERR:") == 0); + CHECK(err.find("max_depth 0") != std::string::npos); + + caller.delegation.max_depth = 1; + CHECK(delegation_spawn_error(caller, nullptr, "forge", 1, {}).empty()); + err = delegation_spawn_error(caller, nullptr, "forge", 2, {}); + CHECK(err.find("max_depth 1") != std::string::npos); + + caller.delegation = {}; + caller.delegation.denied_callees = {"forge"}; + err = delegation_spawn_error(caller, nullptr, "forge", 1, {}); + CHECK(err.find("denied_callees") != std::string::npos); + CHECK(delegation_spawn_error(caller, nullptr, "scout", 1, {}).empty()); + + caller.delegation = {}; + caller.delegation.allowed_callees = {"scout", "vera"}; + CHECK(delegation_spawn_error(caller, nullptr, "scout", 1, {}).empty()); + err = delegation_spawn_error(caller, nullptr, "forge", 1, {}); + CHECK(err.find("allowed_callees") != std::string::npos); +} + +TEST_CASE("delegation_spawn_error: callee max_depth and budgets") { + Constitution caller; + Constitution child; + child.delegation.max_depth = 0; + auto err = delegation_spawn_error(caller, &child, "scout", 1, {}); + CHECK(err.find("cannot run at depth 1") != std::string::npos); + + child.delegation.max_depth = 1; + CHECK(delegation_spawn_error(caller, &child, "scout", 1, {}).empty()); + err = delegation_spawn_error(caller, &child, "scout", 2, {}); + CHECK(err.find("cannot run at depth 2") != std::string::npos); + + caller.delegation.max_subtree_tokens = 1000; + CHECK(delegation_spawn_error(caller, nullptr, "forge", 1, {999, 0}).empty()); + err = delegation_spawn_error(caller, nullptr, "forge", 1, {1000, 0}); + CHECK(err.find("token budget") != std::string::npos); + + caller.delegation = {}; + caller.delegation.max_subtree_usd = 0.25; + CHECK(delegation_spawn_error(caller, nullptr, "forge", 1, {0, 0.24}).empty()); + err = delegation_spawn_error(caller, nullptr, "forge", 1, {0, 0.25}); + CHECK(err.find("spend budget") != std::string::npos); +} + +TEST_CASE("stock agents parse without a delegation block") { + // Starters must keep today's behavior unless an operator adds policy. +#ifdef ARBITER_AGENTS_DIR + const std::string dir = ARBITER_AGENTS_DIR; +#else + const std::string dir = "agents"; +#endif + const char* files[] = { + "scout.json", "forge.json", "vera.json", "quill.json", "nexus.json", + "loom.json", "compass.json", "beacon.json", "echo.json", "jules.json" + }; + int loaded = 0; + for (auto* f : files) { + std::string path = dir + "/" + f; + try { + auto c = Constitution::from_file(path); + CHECK(c.delegation.is_default()); + ++loaded; + } catch (const std::exception& e) { + FAIL("failed to load " << path << ": " << e.what()); + } + } + CHECK(loaded == 10); +} diff --git a/tests/test_delegation.cpp b/tests/test_delegation.cpp new file mode 100644 index 00000000..bd146455 --- /dev/null +++ b/tests/test_delegation.cpp @@ -0,0 +1,147 @@ +// tests/test_delegation.cpp — Runtime spawn gates for constitution +// delegation policy. No live LLM: denials happen before send_internal. +#define DOCTEST_CONFIG_IMPLEMENT_WITH_MAIN +#include "doctest.h" + +#include "orchestrator.h" +#include "constitution.h" + +#include + +using namespace arbiter; + +static Constitution specialist(const std::string& name) { + Constitution c; + c.name = name; + c.role = "specialist"; + c.model = "anthropic/claude-sonnet-5"; + c.capabilities = {"/agent", "/parallel"}; + return c; +} + +static void add_roster(Orchestrator& orch) { + orch.create_agent("forge", specialist("forge")); + orch.create_agent("scout", specialist("scout")); + orch.create_agent("vera", specialist("vera")); +} + +TEST_CASE("check_delegation_spawn: stock index allows catalog callees") { + Orchestrator orch({}); + add_roster(orch); + const auto& idx = orch.get_constitution("index"); + CHECK(idx.delegation.is_default()); + CHECK(orch.check_delegation_spawn("index", idx, "forge", 1).empty()); + CHECK(orch.check_delegation_spawn("index", idx, "forge", 2).empty()); +} + +TEST_CASE("/agent: forbidden callee (denied_callees forge)") { + Orchestrator orch({}); + add_roster(orch); + orch.get_agent("index").config_mut().delegation.denied_callees = {"forge"}; + + std::string out = orch.execute_slash_command("/agent forge do the deploy", "index"); + CHECK(out.find("ERR:") != std::string::npos); + CHECK(out.find("forge") != std::string::npos); + CHECK(out.find("denied_callees") != std::string::npos); + + CHECK(orch.check_delegation_spawn( + "index", orch.get_constitution("index"), "scout", 1).empty()); +} + +TEST_CASE("/agent: allowlist that omits forge") { + Orchestrator orch({}); + add_roster(orch); + orch.get_agent("index").config_mut().delegation.allowed_callees = {"scout", "vera"}; + + std::string out = orch.execute_slash_command("/agent forge hello", "index"); + CHECK(out.find("ERR:") != std::string::npos); + CHECK(out.find("allowed_callees") != std::string::npos); +} + +TEST_CASE("/agent and /parallel: max_depth 0") { + Orchestrator orch({}); + add_roster(orch); + orch.get_agent("index").config_mut().delegation.max_depth = 0; + + std::string out = orch.execute_slash_command("/agent scout hi", "index"); + CHECK(out.find("ERR:") != std::string::npos); + CHECK(out.find("max_depth 0") != std::string::npos); + + out = orch.execute_slash_command( + "/parallel\n/agent scout a\n/agent vera b\n/endparallel", "index"); + CHECK(out.find("ERR:") != std::string::npos); + CHECK(out.find("max_depth 0") != std::string::npos); +} + +TEST_CASE("/parallel: forbidden callee") { + Orchestrator orch({}); + add_roster(orch); + orch.get_agent("index").config_mut().delegation.denied_callees = {"forge"}; + + std::string out = orch.execute_slash_command( + "/parallel\n/agent forge ship it\n/endparallel", "index"); + CHECK(out.find("ERR:") != std::string::npos); + CHECK(out.find("denied_callees") != std::string::npos); +} + +TEST_CASE("max_depth 1: depth-0 spawn ok, depth-2 spawn denied") { + Orchestrator orch({}); + add_roster(orch); + orch.get_agent("scout").config_mut().delegation.max_depth = 1; + const auto& scout = orch.get_constitution("scout"); + + CHECK(orch.check_delegation_spawn("scout", scout, "vera", 1).empty()); + std::string err = orch.check_delegation_spawn("scout", scout, "vera", 2); + CHECK(err.find("ERR:") == 0); + CHECK(err.find("max_depth 1") != std::string::npos); + + // Index (max_depth default 2) can still send work to scout at depth 1. + CHECK(orch.check_delegation_spawn("index", orch.get_constitution("index"), + "scout", 1).empty()); +} + +TEST_CASE("budget exceeded stops further spawn") { + Orchestrator orch({}); + add_roster(orch); + orch.get_agent("index").config_mut().delegation.max_subtree_tokens = 500; + orch.record_delegation_spend("index", 500, 0); + + std::string out = orch.execute_slash_command("/agent scout hi", "index"); + CHECK(out.find("token budget exceeded") != std::string::npos); + + orch.get_agent("index").config_mut().delegation.max_subtree_tokens.reset(); + orch.get_agent("index").config_mut().delegation.max_subtree_usd = 0.01; + orch.record_delegation_spend("index", 0, 0.01); + out = orch.execute_slash_command("/agent scout hi", "index"); + CHECK(out.find("spend budget exceeded") != std::string::npos); +} + +TEST_CASE("JIT run_ephemeral honours caller deny-forge and callee max_depth") { + Orchestrator orch({}); + add_roster(orch); + Constitution cfg = specialist("forge"); + + orch.get_agent("index").config_mut().delegation.denied_callees = {"forge"}; + ApiResponse denied = orch.run_ephemeral("forge", cfg, "close the clause"); + CHECK_FALSE(denied.ok); + CHECK(denied.error_type == "delegation_policy"); + CHECK(denied.error.find("denied_callees") != std::string::npos); + + orch.get_agent("index").config_mut().delegation = {}; + cfg.delegation.max_depth = 0; + ApiResponse too_deep = orch.run_ephemeral("forge", cfg, "close the clause"); + CHECK_FALSE(too_deep.ok); + CHECK(too_deep.error_type == "delegation_policy"); + CHECK(too_deep.error.find("cannot run at depth 1") != std::string::npos); +} + +TEST_CASE("presence-style always_on constitution still has default delegation") { + // Guard: adding delegation must not change presence defaults or imply + // that presence consults the spawn gate. + Constitution c = Constitution::from_json(R"({ + "name": "jules", + "presence": "always_on" + })"); + CHECK(c.presence.mode == "always_on"); + CHECK(c.delegation.is_default()); +}