From 77931cf2db395bde2c39b0975cf22b152040da65 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" Date: Tue, 29 Sep 2026 22:02:18 +0000 Subject: [PATCH] chore: sync public mirror from internal --- .repository-projection.json | 6 +- CHANGELOG.md | 36 + Cargo.lock | 2 +- package-lock.json | 4 +- package.json | 2 +- packages/maestro-rs/Cargo.toml | 2 +- vendor/dex-loop/Cargo.toml | 59 + vendor/dex-loop/README.md | 70 + vendor/dex-loop/src/budget.rs | 123 ++ vendor/dex-loop/src/compaction.rs | 87 + vendor/dex-loop/src/context.rs | 684 +++++++ vendor/dex-loop/src/engine.rs | 1197 ++++++++++++ vendor/dex-loop/src/event.rs | 779 ++++++++ vendor/dex-loop/src/lib.rs | 46 + vendor/dex-loop/src/ports.rs | 227 +++ vendor/dex-loop/src/rehydrate.rs | 41 + vendor/dex-loop/src/sanitize.rs | 172 ++ vendor/dex-loop/tests/authorized_tools.rs | 89 + vendor/dex-loop/tests/client_tool_replay.rs | 387 ++++ vendor/dex-loop/tests/scenarios.rs | 1821 +++++++++++++++++++ vendor/dex-loop/tests/sim.rs | 448 +++++ vendor/dex-loop/tests/sim/fakes.rs | 615 +++++++ vendor/dex-loop/tests/sim/invariants.rs | 444 +++++ vendor/dex-loop/tests/sim/scenario.rs | 623 +++++++ vendor/dex-loop/tests/support/mod.rs | 850 +++++++++ vendor/dex-loop/tests/tool_deadline.rs | 211 +++ vendor/dex-loop/tests/wall_budget.rs | 125 ++ 27 files changed, 9142 insertions(+), 8 deletions(-) create mode 100644 vendor/dex-loop/Cargo.toml create mode 100644 vendor/dex-loop/README.md create mode 100644 vendor/dex-loop/src/budget.rs create mode 100644 vendor/dex-loop/src/compaction.rs create mode 100644 vendor/dex-loop/src/context.rs create mode 100644 vendor/dex-loop/src/engine.rs create mode 100644 vendor/dex-loop/src/event.rs create mode 100644 vendor/dex-loop/src/lib.rs create mode 100644 vendor/dex-loop/src/ports.rs create mode 100644 vendor/dex-loop/src/rehydrate.rs create mode 100644 vendor/dex-loop/src/sanitize.rs create mode 100644 vendor/dex-loop/tests/authorized_tools.rs create mode 100644 vendor/dex-loop/tests/client_tool_replay.rs create mode 100644 vendor/dex-loop/tests/scenarios.rs create mode 100644 vendor/dex-loop/tests/sim.rs create mode 100644 vendor/dex-loop/tests/sim/fakes.rs create mode 100644 vendor/dex-loop/tests/sim/invariants.rs create mode 100644 vendor/dex-loop/tests/sim/scenario.rs create mode 100644 vendor/dex-loop/tests/support/mod.rs create mode 100644 vendor/dex-loop/tests/tool_deadline.rs create mode 100644 vendor/dex-loop/tests/wall_budget.rs diff --git a/.repository-projection.json b/.repository-projection.json index c45b486cc..9bfae1fee 100644 --- a/.repository-projection.json +++ b/.repository-projection.json @@ -3,11 +3,11 @@ "projection": "deixic-code", "projectionSchemaVersion": 1, "sourceRepository": "dx-corp/mono", - "sourceSha": "16426c3c3b3dc59f6817c9a2836fd714d17a5533", + "sourceSha": "2283b0b186fcab14fb2da2b58203f59522ccfcc0", "destinationRepository": "dx-corp/code", - "priorProjectedBase": "5bfbbeb575801b605b42755a5a1154849c56a75a", + "priorProjectedBase": "9ee0b79f0f789db33a877ebc134d67500a04d573", "definitionDigest": "82936441c776e3e8edb5d215a75007ec9714a233f489d460075d79d5ef5ba32f", "toolDigest": "c244d99199a7ae3eb8ff644a99462163c23b0bb6a83ef50af01efbdca0b81d04", - "contentDigest": "42be9fd04b572f4a3cf4bb0854bcac5ae498a5d48dc99cf8d129ade8a4ab0536", + "contentDigest": "92cd365a6ecde13575d354a172d727b2d98690a8901e4091f5a39a684ed4029c", "publicationEligible": true } diff --git a/CHANGELOG.md b/CHANGELOG.md index be9196caa..24033e1a6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -51,6 +51,42 @@ versioning when releases are cut. and keep scheduled public runs inert so public publishing stays downstream of the internal source-of-truth release. +## [0.10.109] - 2026-09-29 + +### Added + +- Dispatch a Dex-approved call without parking for its own resume (#11373). + +### Changed + +- Add bounded gateway command load proof (#11386). +- Channel-edge-slack: correlate every Slack event log line (event_id, retry_num, trace_id) (#11385). +- Platform-api: add dex_turn_submit_seconds and phase histograms (#11379). +- Model-gateway: drop unbounded labels from provider latency histogram, extend buckets to 120s (#11378). +- Tool-executor: latency histograms for computer.start and tool calls (#11375). +- Stabilize the parallel cold-provision timing check (#11393). +- Platform-worker: continue the Slack trace through reply projection and delivery; slack_delivery_* logs (#11389). + +### Fixed + +- Clear two deny-level clippy lints on main (#11392). +- Park dex approval for human-held API keys (#11390). +- Declare tools the history calls but the request does not offer (#11388). +- Route public projection off saturated PR runners (#11387). +- Repair dex-loop build and provider SSE framing (#11383). +- Resident-process PUT counts as activity (#11384). +- Record sandbox audit binding production applier (#11382). +- Execute Dex gateway commands on placement workers (#11381). +- Remove stash-conflict marker that left deixic_operating_threads_tests.rs unparseable (#11380). +- Preflight public mirror App access before build (#11374). +- Own the sandbox idle policy; sandboxwich TTL is a safety net (#11376). +- Grant delivery worker the ack-obligation table (#11377). +- Unblock Maestro release validation (#11397). +- Stop saturated workers scanning queued jobs (#11396). +- Don't surface React Query CancelledError as a thread load error (#11395). +- Log why a Gemini stream failed and the shape of a rejected history (#11394). +- Close the learning-outbox test left open by a stash marker (#11391). + ## [0.10.108] - 2026-09-29 ### Fixed diff --git a/Cargo.lock b/Cargo.lock index 40b989ec7..438e5814d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3752,7 +3752,7 @@ checksum = "dae608c151f68243f2b000364e1f7b186d9c29845f7d2d85bd31b9ad77ad552b" [[package]] name = "maestro" -version = "0.10.108" +version = "0.10.109" dependencies = [ "anyhow", "ctor", diff --git a/package-lock.json b/package-lock.json index 65bbc1f2d..3f02d53fe 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@evalops/maestro", - "version": "0.10.108", + "version": "0.10.109", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@evalops/maestro", - "version": "0.10.108", + "version": "0.10.109", "license": "BUSL-1.1", "bin": { "deixic-code": "bin/deixic-code", diff --git a/package.json b/package.json index 9f8bf5f55..69b40960b 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "@evalops/deixic-code", "description": "Deixic Code — native Rust coding agent, CLI, TUI, and runtime gateway", - "version": "0.10.108", + "version": "0.10.109", "private": false, "type": "module", "bin": { diff --git a/packages/maestro-rs/Cargo.toml b/packages/maestro-rs/Cargo.toml index 86d3fee35..96c545165 100644 --- a/packages/maestro-rs/Cargo.toml +++ b/packages/maestro-rs/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "maestro" -version = "0.10.108" +version = "0.10.109" edition = "2021" license = "MIT" description = "Canonical native Rust CLI for Deixic Code" diff --git a/vendor/dex-loop/Cargo.toml b/vendor/dex-loop/Cargo.toml new file mode 100644 index 000000000..9bcba04cf --- /dev/null +++ b/vendor/dex-loop/Cargo.toml @@ -0,0 +1,59 @@ +[package] +name = "dex-loop" +version = "0.1.0" +edition = "2024" +license = "UNLICENSED" +publish = false +rust-version = "1.95" + +[dependencies] +futures-util = "0.3" +hex = "0.4" +serde = { version = "1.0", features = ["derive"] } +serde_json = "1.0" +sha2 = "0.10" +thiserror = "2.0" +# Only the timer, to race the model stream against `budget.wall` in +# `Engine::model_step` and each tool call against its deadline: the crate +# still does no I/O of its own. +tokio = { version = "1.48", features = ["time"] } +tokio-util = "0.7" + +[dev-dependencies] +proptest = "1.9" +rand = "0.8" +tokio = { version = "1.48", features = ["macros", "rt", "sync", "test-util", "time"] } + +[lints.rust] +non_ascii_idents = "deny" +unsafe_code = "forbid" +unexpected_cfgs = { level = "warn", check-cfg = ["cfg(kani)"] } +unused_lifetimes = "warn" + +[lints.clippy] +await_holding_lock = "deny" +dbg_macro = "deny" +disallowed_methods = "deny" +empty_drop = "deny" +exit = "deny" +filetype_is_file = "deny" +fn_to_numeric_cast_any = "deny" +lossy_float_literal = "deny" +mem_forget = "deny" +mutex_atomic = "deny" +rc_buffer = "deny" +rest_pat_in_fully_bound_structs = "deny" +string_add = "deny" +string_lit_as_bytes = "deny" +todo = "deny" +unchecked_time_subtraction = "deny" +undocumented_unsafe_blocks = "deny" +verbose_file_reads = "deny" + +# Trim debug info in local dev/test builds: full `debug = true` is the +# default and was measured at ~49% of artifact bytes. Line tables keep +# backtraces and panic locations useful. CI overrides both profiles to 0 +# via CARGO_PROFILE_DEV_DEBUG / CARGO_PROFILE_TEST_DEBUG, which take +# precedence over these manifest values. + +[workspace] diff --git a/vendor/dex-loop/README.md b/vendor/dex-loop/README.md new file mode 100644 index 000000000..69241b8ff --- /dev/null +++ b/vendor/dex-loop/README.md @@ -0,0 +1,70 @@ +# dex-loop + +The one Dex agent loop. Web, Slack, Teams and every later surface run the same +engine over the same event log; a surface only writes ingress events and +renders the stream. + +## The loop + +```text +host appends UserMessage { principal, text } +ctx = rehydrate(thread, log) // warm or after a crash: same path +loop { + read control events (steer, interrupt, approval decisions, answers) + StepStarted -> stream the model -> text to the log as it arrives + ModelStepCompleted { text, calls } // commit point, before any policy + no calls? Final, return Done + per call, in the model's order: + policy (under the call's principal) -> approval? park, return Parked + read-only: join the parallel wave // results enter history in call order + mutation: Effects claim -> ToolStarted -> run -> record -> ToolFinished +} +``` + +`Engine::run(&mut ctx, &cancel)` returns `Done`, `Parked(approval)`, +`Asked(call)`, `Interrupted`, or `Failed`. After an approval or answer the +host appends the decision and calls `run` again; the pending calls come from +the log, not from memory. + +## Ports + +| Port | Job | +| --- | --- | +| `Log` | Append events (fenced by the thread lease), coalesce text into `TextDelta` rows, read control events after a cursor. | +| `Model` | Stream one response for the history and the offered tool schemas. | +| `Tools` | Catalog, `search`, `policy` (governance, grants, guardrails, guardian) and `run`. | +| `Effects` | The durable ledger for mutations: claim before dispatch, record after. A mutation is dispatched at most once per call id; an unknown outcome is never replayed. | +| `Sanitizer` | Customer-safe text: replaces tool names and internal names before any delta is written. `Lexicon` is the built-in implementation. | +| `Compactor` | Optional. `Threshold` summarizes old history when it grows past a size. | + +The engine offers `tools.search` plus core tools on every step; tools that +search matches are exposed from the next step and recorded as `ToolsExposed`. + +## Events + +Ingress (hosts write): `UserMessage`, `Steer`, `Interrupt`, `ApprovalDecided`, +`Answer`, `ToolProgress`. Each ingress event carries its principal. + +Engine: `StepStarted`, `TextDelta`, `Usage`, `ModelStepCompleted`, +`ModelAttemptAbandoned`, `ToolStarted`, `ToolsExposed`, `ToolFinished` +(`succeeded | failed | running | unknown`), `ApprovalRequested`, `Question`, +`Compaction`, `Final`, `Error`, `Interrupted`. + +Interrupt cancels the model stream and running reads; a mutation that has +started completes, and the turn stops before the next effect. + +## What this crate will never contain + +- Surface logic: no Slack, Teams, web or renderer code, and no per-surface + behavior in the loop. +- Coding concepts: no coding task kinds, validators or workflows. Coding is + tools run through this same loop. +- Provider or service clients: no HTTP, SQL, model SDKs or `tokio::spawn`. + Hosts implement the ports. + +## Checks + +```bash +cargo test --manifest-path rust/Cargo.toml -p dex-loop +cargo clippy --manifest-path rust/Cargo.toml -p dex-loop --all-targets -- -D warnings +``` diff --git a/vendor/dex-loop/src/budget.rs b/vendor/dex-loop/src/budget.rs new file mode 100644 index 000000000..74a9ceea3 --- /dev/null +++ b/vendor/dex-loop/src/budget.rs @@ -0,0 +1,123 @@ +//! Per-turn limits. + +use std::fmt; +use std::time::Duration; + +use crate::event::Usage; + +/// Limits for one turn. The engine checks them before every model call. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Budget { + /// Model calls per turn that may offer tools. Once a turn has made this + /// many calls the engine makes one more, with no tools offered, so the + /// model writes its answer from what it has. If that call still asks for + /// a tool, the turn fails with `BudgetExhausted`. + pub max_steps: u32, + /// Input plus output tokens per turn. + pub max_tokens: u64, + pub max_cost_micros: u64, + /// Time spent inside one `Engine::run` call. Time parked on an approval + /// or a question does not count. + pub wall: Duration, +} + +impl Default for Budget { + /// 60 steps and 25 minutes. Token and cost caps come from org policy, so + /// the default leaves them open. + fn default() -> Self { + Self { + max_steps: 60, + max_tokens: u64::MAX, + max_cost_micros: u64::MAX, + wall: Duration::from_secs(25 * 60), + } + } +} + +/// The limit a turn ran out of. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum BudgetAxis { + Steps, + Tokens, + Cost, + Wall, +} + +impl fmt::Display for BudgetAxis { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + BudgetAxis::Steps => "steps", + BudgetAxis::Tokens => "tokens", + BudgetAxis::Cost => "cost", + BudgetAxis::Wall => "wall", + }) + } +} + +impl Budget { + /// Whether the next model call is the answer-only call: `steps` calls + /// have already been made and the tool-offering allowance is spent. + pub fn answer_only(&self, steps: u32) -> bool { + steps >= self.max_steps + } + + /// The first exhausted axis, if any. `steps` counts model calls made so + /// far. Steps are exhausted only after the answer-only call + /// (`max_steps + 1`), which `answer_only` allows for. The other axes are + /// exhausted as soon as they are reached, answer-only call included. + pub fn exhausted(&self, steps: u32, usage: Usage, elapsed: Duration) -> Option { + if steps > self.max_steps { + Some(BudgetAxis::Steps) + } else if usage.tokens() >= self.max_tokens { + Some(BudgetAxis::Tokens) + } else if usage.cost_micros >= self.max_cost_micros { + Some(BudgetAxis::Cost) + } else if elapsed >= self.wall { + Some(BudgetAxis::Wall) + } else { + None + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn each_axis_is_reported() { + let budget = Budget { + max_steps: 2, + max_tokens: 100, + max_cost_micros: 50, + wall: Duration::from_secs(1), + }; + let usage = |tokens, cost| Usage { + input_tokens: tokens, + output_tokens: 0, + cost_micros: cost, + }; + let short = Duration::from_millis(1); + assert_eq!(budget.exhausted(1, usage(10, 1), short), None); + // The answer-only call after `max_steps` is still allowed. + assert_eq!(budget.exhausted(2, usage(10, 1), short), None); + assert!(!budget.answer_only(1)); + assert!(budget.answer_only(2)); + assert_eq!( + budget.exhausted(3, usage(10, 1), short), + Some(BudgetAxis::Steps) + ); + assert_eq!( + budget.exhausted(1, usage(100, 1), short), + Some(BudgetAxis::Tokens) + ); + assert_eq!( + budget.exhausted(1, usage(10, 50), short), + Some(BudgetAxis::Cost) + ); + assert_eq!( + budget.exhausted(1, usage(10, 1), Duration::from_secs(1)), + Some(BudgetAxis::Wall) + ); + } +} diff --git a/vendor/dex-loop/src/compaction.rs b/vendor/dex-loop/src/compaction.rs new file mode 100644 index 000000000..9a0447f9b --- /dev/null +++ b/vendor/dex-loop/src/compaction.rs @@ -0,0 +1,87 @@ +//! Keeping the model's context bounded. + +use std::future::Future; + +use crate::context::{Context, Entry, Message}; +use crate::event::Cursor; + +/// A planned compaction: history up to and including `covers_to` becomes +/// `summary`. The engine appends it as `Event::Compaction`, so rehydration +/// applies it the same way. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Compaction { + pub covers_to: Cursor, + pub summary: String, +} + +/// Decides, before each model call, whether to compact. +pub trait Compactor: Send + Sync { + fn plan(&self, ctx: &Context) -> impl Future> + Send; +} + +/// Never compacts. +#[derive(Clone, Copy, Debug, Default)] +pub struct NoCompaction; + +impl Compactor for NoCompaction { + async fn plan(&self, _ctx: &Context) -> Option { + None + } +} + +/// Writes the summary for a prefix of history (usually one model call). +pub trait Summarize: Send + Sync { + /// `None` skips this compaction; the engine continues uncompacted. + fn summarize(&self, entries: &[Entry]) -> impl Future> + Send; +} + +/// Compacts when history exceeds `max_bytes`, keeping at least the last +/// `keep_recent` entries verbatim. +#[derive(Clone, Debug)] +pub struct Threshold { + max_bytes: usize, + keep_recent: usize, + summarizer: S, +} + +impl Threshold { + pub fn new(max_bytes: usize, keep_recent: usize, summarizer: S) -> Self { + Self { + max_bytes, + keep_recent, + summarizer, + } + } +} + +impl Compactor for Threshold { + async fn plan(&self, ctx: &Context) -> Option { + let history = ctx.history(); + let size: usize = history.iter().map(|entry| entry.message.size()).sum(); + if size <= self.max_bytes { + return None; + } + let cut = cut_point(history, self.keep_recent)?; + let summary = self.summarizer.summarize(&history[..cut]).await?; + Some(Compaction { + covers_to: history[cut - 1].cursor, + summary, + }) + } +} + +/// The latest index `cut` with at least `keep_recent` entries after it where +/// history can be split: the kept part must not start with a tool result +/// (it belongs to the assistant message before it), and the cut must fall +/// between two cursors so the compaction event can name it. +fn cut_point(history: &[Entry], keep_recent: usize) -> Option { + let latest = history.len().checked_sub(keep_recent)?; + (1..=latest).rev().find(|&cut| { + let first_kept = history.get(cut); + let only_summary = cut == 1 && matches!(history[0].message, Message::Summary { .. }); + let splits_cursor = first_kept.is_some_and(|kept| kept.cursor == history[cut - 1].cursor); + let orphans_result = + first_kept.is_some_and(|kept| matches!(kept.message, Message::Tool { .. })); + !only_summary && !splits_cursor && !orphans_result + }) +} diff --git a/vendor/dex-loop/src/context.rs b/vendor/dex-loop/src/context.rs new file mode 100644 index 000000000..04cb03682 --- /dev/null +++ b/vendor/dex-loop/src/context.rs @@ -0,0 +1,684 @@ +//! The in-memory state of one thread, derived from its log. +//! +//! `Context::observe` is the only state transition. The engine calls it for +//! every event it appends and every control event it reads; `rehydrate` calls +//! it for every event in the log. The same events therefore always produce +//! the same context, whether the actor stayed warm or restarted. + +use crate::event::{ + ApprovalId, ApprovalMode, ArtifactRef, CallId, ClientToolSpec, Cursor, Event, MessageId, + Outcome, Output, PrincipalId, ProposedCall, ProviderReasoning, ThreadId, ToolName, ToolResult, + TurnId, Usage, +}; + +const NOT_RUN_NEW_TURN: &str = "not run: a new turn started first"; +const UNKNOWN_NEW_TURN: &str = "outcome unknown: a new turn started before the result was recorded"; + +/// One message in the model's view of the thread. +#[derive(Clone, Debug, PartialEq)] +pub enum Message { + User { + turn: TurnId, + message_id: Option, + principal: PrincipalId, + text: String, + attachments: Vec, + }, + Assistant { + text: String, + calls: Vec, + /// The step's provider continuation state, as the `Model` port + /// wrote it; `None` for steps logged without one. + reasoning: Option, + }, + Tool { + call: CallId, + name: ToolName, + outcome: Outcome, + output: Output, + }, + /// A compaction summary of everything before it. + Summary { text: String }, +} + +impl Message { + /// A rough size in bytes, for compaction thresholds. + pub fn size(&self) -> usize { + match self { + Message::User { text, .. } | Message::Summary { text } => text.len(), + Message::Assistant { text, calls, .. } => { + text.len() + + calls + .iter() + .map(|call| call.tool.as_str().len() + call.args.to_string().len()) + .sum::() + } + Message::Tool { output, .. } => match output { + Output::Text(text) => text.len(), + Output::Ref(reference) => reference.as_str().len(), + }, + } + } +} + +/// A history message and the cursor of the event that placed it. Cursors are +/// non-decreasing along the history. +#[derive(Clone, Debug, PartialEq)] +pub struct Entry { + pub cursor: Cursor, + pub message: Message, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum Status { + Idle, + Running, + Done, + Interrupted, + Failed, +} + +/// Where one call of the open step stands. +#[derive(Clone, Debug, PartialEq)] +pub(crate) enum CallState { + /// Not yet dispatched. + Todo, + /// `ToolStarted` is in the log; no `ToolFinished` yet. + Started, + Parked { + approval: ApprovalId, + decision: Option, + }, + Asked { + answer: Option, + }, + /// `ClientToolRequested` is in the log; no `ClientToolResult` yet. + AwaitingClient { + result: Option, + /// Unix milliseconds after which the engine gives up on the wait, + /// from `ClientToolRequested::deadline_ms`. Carried on the state + /// itself, not read fresh from the log, so it is exactly the value + /// the call was originally requested with, on a warm engine and on + /// every rehydrate alike. + deadline_ms: i64, + }, + Done(ToolResult), +} + +/// A recorded approval decision, checked again on resume. +#[derive(Clone, Debug, PartialEq)] +pub(crate) struct Decision { + pub(crate) approved: bool, + pub(crate) args_digest: String, +} + +/// The calls of the last `ModelStepCompleted` until every one has a result. +#[derive(Clone, Debug, PartialEq)] +pub(crate) struct OpenStep { + pub(crate) calls: Vec, + pub(crate) states: Vec, +} + +/// Everything the engine knows about a thread. +#[derive(Clone, Debug, PartialEq)] +pub struct Context { + thread: ThreadId, + turn: Option, + acting: Option, + status: Status, + history: Vec, + step: u32, + usage: Usage, + cursor: Cursor, + control: Cursor, + steers: Vec<(Cursor, PrincipalId, String)>, + interrupt_requested: bool, + attempt: Option, + open_step: Option, + exposed: Vec, + client_tools: Vec, + authorized_tools: Vec, + approval_mode: ApprovalMode, + /// `UserMessage`s for a later turn that arrived (by log cursor) while an + /// earlier turn was still running. FIFO: applied one at a time, each when + /// the turn ahead of it reaches a terminal status, so the still-running + /// turn's own later events (its `ModelStepCompleted`, `ToolFinished`, + /// `Final`) are never attributed to the turn that raced in ahead of them. + pending_turns: Vec, + authorized_principal: Option, +} + +#[derive(Clone, Debug, PartialEq)] +struct PendingTurn { + turn: TurnId, + message_id: Option, + principal: PrincipalId, + text: String, + attachments: Vec, + /// Resolved when this turn actually starts, so a message that raced in + /// does not replace the still-running turn's client tools. + client_tools: Vec, + authorized_tools: Vec, + approval_mode: ApprovalMode, +} + +impl Context { + /// An empty thread. + pub fn new(thread: ThreadId) -> Self { + Self { + thread, + turn: None, + acting: None, + status: Status::Idle, + history: Vec::new(), + step: 0, + usage: Usage::default(), + cursor: Cursor::START, + control: Cursor::START, + steers: Vec::new(), + interrupt_requested: false, + attempt: None, + open_step: None, + exposed: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: ApprovalMode::Interactive, + authorized_principal: None, + pending_turns: Vec::new(), + } + } + + pub fn thread(&self) -> &ThreadId { + &self.thread + } + + /// The current (or last) turn. + pub fn turn(&self) -> Option<&TurnId> { + self.turn.as_ref() + } + + /// The author of the most recent user input in history. Calls proposed + /// by the next model step act under this principal. + pub fn acting_principal(&self) -> Option<&PrincipalId> { + self.acting.as_ref() + } + + /// Tools `tools.search` exposed in this turn. + pub fn exposed_tools(&self) -> &[ToolName] { + &self.exposed + } + + /// Tools this turn's client session declared on `Send`, exactly as + /// logged (the client's own claims, unresolved). The host turns these + /// into catalog entries per turn (see `dex_tools::client::declare`), + /// tagged to the declaring session: the engine never offers them + /// directly, so a host that ignores this returns to offering nothing + /// client-declared, never a stale or duplicated set. + pub fn client_tools(&self) -> &[ClientToolSpec] { + &self.client_tools + } + + /// The original admitted principal, unchanged by steering. + pub fn authorized_principal(&self) -> Option<&PrincipalId> { + self.authorized_principal.as_ref() + } + + pub fn authorized_tools(&self) -> &[ToolName] { + &self.authorized_tools + } + + /// Who can answer this turn's approvals, as logged on its `UserMessage`. + pub fn approval_mode(&self) -> ApprovalMode { + self.approval_mode + } + + /// The model's view of the thread. + pub fn history(&self) -> &[Entry] { + &self.history + } + + /// Model calls started in the current turn. + pub fn step(&self) -> u32 { + self.step + } + + /// Model spend in the current turn. + pub fn usage(&self) -> Usage { + self.usage + } + + /// The last event observed. + pub fn cursor(&self) -> Cursor { + self.cursor + } + + /// The last control event observed; `Log::control_since` reads after it. + pub fn control_cursor(&self) -> Cursor { + self.control + } + + /// Raises the floor `control_cursor()` can never fall below, regardless + /// of which control events this `Context` actually observes. + /// + /// `rehydrate` calls this with the first cursor of the log suffix it + /// replays, before observing anything: a suffix chosen to start after + /// some already-accounted-for point (a compaction boundary, most + /// commonly) can easily contain zero control-kind events even though + /// real ones exist earlier in the full log. Without a floor, + /// `control_cursor()` would default to `Cursor::START` in that case, and + /// the engine's first `Log::control_since(control_cursor())` call would + /// re-fetch every control event the suffix deliberately left out -- + /// including an old, already-resolved `Interrupt` for a turn that ended + /// before this one even started, observed again with the *new* turn + /// `Running`, which interrupts it. The kernel contract this enforces: + /// no control event with a cursor before the rehydrate point is ever + /// re-applied, whether directly (during this replay) or indirectly + /// (fetched again afterward because this replay under-reported where it + /// left off). + pub(crate) fn advance_control_floor(&mut self, floor: Cursor) { + self.control = self.control.max(floor); + } + + /// Puts the control cursor back after the engine appended a control event + /// of its own, so a control event from another writer that landed just + /// before it is still read. + pub(crate) fn rewind_control(&mut self, to: Cursor) { + self.control = to; + } + + pub(crate) fn status(&self) -> Status { + self.status + } + + pub(crate) fn interrupt_requested(&self) -> bool { + self.interrupt_requested + } + + pub(crate) fn has_queued_steers(&self) -> bool { + !self.steers.is_empty() + } + + pub(crate) fn open_step(&self) -> Option<&OpenStep> { + self.open_step.as_ref() + } + + /// The step of a model attempt that started but never completed. + pub(crate) fn open_attempt(&self) -> Option { + self.attempt + } + + /// Applies one log event. Events must be observed in log order. + pub fn observe(&mut self, cursor: Cursor, event: &Event) { + self.cursor = self.cursor.max(cursor); + if event.is_control() { + self.control = self.control.max(cursor); + } + match event { + Event::UserMessage { + turn, + message_id, + principal, + text, + attachments, + client_tools, + authorized_tools, + approval_mode, + } => { + let next = PendingTurn { + turn: turn.clone(), + message_id: message_id.clone(), + principal: principal.clone(), + text: text.clone(), + attachments: attachments.clone(), + client_tools: client_tools.clone(), + authorized_tools: authorized_tools.clone(), + approval_mode: *approval_mode, + }; + if self.status == Status::Running { + // This message's cursor landed while the current turn was + // still in flight (it was sent before that turn ended). + // The engine that ran the current turn never read it + // mid-flight, and the current turn's own later events + // (still to come, at cursors after this one) belong to + // the turn already running, not to this one. Queue it and + // start it once the running turn reaches a terminal + // status, so the log's cursor order can never misattach + // one turn's events to another. + self.pending_turns.push(next); + } else { + self.begin_turn(cursor, next); + } + } + Event::Steer { principal, text } => { + // Only queue a steer once this replay has actually observed + // a turn start. A `Steer` for a turn whose own `UserMessage` + // is before the rehydrate point (excluded from `events`, + // most often by a compaction boundary that lands between + // that turn's start and one of its own not-yet-flushed + // steers) has no turn in this replay to belong to; queuing + // it anyway would only surface later, at the *next* + // `begin_turn`'s unconditional flush, as a stray message + // misattributed to whatever turn happens to start next -- + // possibly one with nothing to do with the principal who + // sent it. `self.turn` is exactly this replay's marker of + // "a turn has started here": `begin_turn` is the only place + // that sets it, and a steer legitimately queued mid-turn + // always arrives after its own turn's `UserMessage`. + if self.turn.is_some() { + self.steers.push((cursor, principal.clone(), text.clone())); + } + } + Event::Interrupt { .. } => { + if self.status == Status::Running { + self.interrupt_requested = true; + } + } + Event::ApprovalDecided { + call, + approval, + args_digest, + approved, + .. + } => { + if let Some(CallState::Parked { + approval: parked, + decision, + }) = self.state_mut(call) + && parked == approval + && decision.is_none() + { + *decision = Some(Decision { + approved: *approved, + args_digest: args_digest.clone(), + }); + } + } + Event::Answer { call, text, .. } => { + if let Some(CallState::Asked { answer }) = self.state_mut(call) + && answer.is_none() + { + *answer = Some(text.clone()); + } + } + Event::ClientToolResult { + call, + outcome, + output, + .. + } => { + if let Some(CallState::AwaitingClient { result, .. }) = self.state_mut(call) + && result.is_none() + { + *result = Some(ToolResult { + outcome: *outcome, + output: Output::Text(output.clone()), + receipt: None, + }); + } + } + Event::StepStarted { + step, + control_through, + } => { + self.step = *step; + self.attempt = Some(*step); + self.flush_steers(cursor, *control_through); + } + // Text reaches history through `ModelStepCompleted`. + Event::TextDelta { .. } | Event::ToolProgress { .. } => {} + Event::Usage(usage) => self.usage += *usage, + Event::ModelStepCompleted { + text, + calls, + reasoning, + .. + } => { + self.attempt = None; + self.push( + cursor, + Message::Assistant { + text: text.clone(), + calls: calls.clone(), + reasoning: reasoning.clone(), + }, + ); + if !calls.is_empty() { + self.open_step = Some(OpenStep { + calls: calls.clone(), + states: vec![CallState::Todo; calls.len()], + }); + } + } + Event::ModelAttemptAbandoned { .. } => self.attempt = None, + Event::ToolStarted { call, .. } => { + if let Some(state) = self.state_mut(call) + && !matches!(state, CallState::Done(_)) + { + *state = CallState::Started; + } + } + Event::ToolsExposed { tools, .. } => { + for tool in tools { + if !self.exposed.contains(tool) { + self.exposed.push(tool.clone()); + } + } + } + Event::ToolFinished { + call, + outcome, + output, + receipt, + } => { + if let Some(state) = self.state_mut(call) { + *state = CallState::Done(ToolResult { + outcome: *outcome, + output: output.clone(), + receipt: receipt.clone(), + }); + } + self.close_step_if_resolved(cursor); + } + Event::ApprovalRequested { call, approval, .. } => { + if let Some(state) = self.state_mut(call) { + *state = CallState::Parked { + approval: approval.clone(), + decision: None, + }; + } + } + Event::Question { call, .. } => { + if let Some(state) = self.state_mut(call) { + *state = CallState::Asked { answer: None }; + } + } + Event::ClientToolRequested { + call, deadline_ms, .. + } => { + if let Some(state) = self.state_mut(call) { + *state = CallState::AwaitingClient { + result: None, + deadline_ms: *deadline_ms, + }; + } + } + Event::Compaction { + covers_to_cursor, + summary, + } => { + self.history + .retain(|entry| entry.cursor > *covers_to_cursor); + self.history.insert( + 0, + Entry { + cursor: *covers_to_cursor, + message: Message::Summary { + text: summary.clone(), + }, + }, + ); + } + Event::Final { .. } => { + self.status = Status::Done; + self.begin_next_pending_turn(cursor); + } + Event::Error { .. } => { + self.attempt = None; + self.status = Status::Failed; + self.begin_next_pending_turn(cursor); + } + Event::Interrupted => { + self.attempt = None; + self.status = Status::Interrupted; + self.interrupt_requested = false; + self.begin_next_pending_turn(cursor); + } + } + } + + /// The common `UserMessage` transition: applied immediately when no turn + /// is running, or deferred through `pending_turns` and applied here once + /// the turn ahead of it ends. + fn begin_turn(&mut self, cursor: Cursor, next: PendingTurn) { + self.close_abandoned_step(cursor); + // Flush before overwriting `self.turn`: any steer still queued here + // is a straggler of the turn that was running (never picked up by + // that turn's own `StepStarted` flush before it ended), so it is + // attributed to that turn, not the one about to start. + self.flush_steers(cursor, Cursor(i64::MAX)); + let PendingTurn { + turn, + message_id, + principal, + text, + attachments, + client_tools, + authorized_tools, + approval_mode, + } = next; + self.turn = Some(turn.clone()); + self.acting = Some(principal.clone()); + self.status = Status::Running; + self.step = 0; + self.usage = Usage::default(); + self.attempt = None; + self.interrupt_requested = false; + self.exposed.clear(); + self.client_tools = client_tools; + self.authorized_tools = authorized_tools; + self.approval_mode = approval_mode; + self.authorized_principal = Some(principal.clone()); + self.push( + cursor, + Message::User { + turn, + message_id, + principal, + text, + attachments, + }, + ); + } + + /// Starts the oldest queued turn, if any, at `cursor`: the cursor of the + /// terminal event (`Final`, `Error` or `Interrupted`) that just ended the + /// turn ahead of it. + fn begin_next_pending_turn(&mut self, cursor: Cursor) { + if self.pending_turns.is_empty() { + return; + } + let next = self.pending_turns.remove(0); + self.begin_turn(cursor, next); + } + + fn push(&mut self, cursor: Cursor, message: Message) { + self.history.push(Entry { cursor, message }); + } + + fn state_mut(&mut self, call: &CallId) -> Option<&mut CallState> { + let step = self.open_step.as_mut()?; + let index = step.calls.iter().position(|c| &c.id == call)?; + step.states.get_mut(index) + } + + /// Steers the engine had read by `through` become user messages, placed + /// at `cursor`. A steer carries no attachments of its own and no source + /// message id: it is control text, not a `UserMessage` the host + /// constructed from an inbound message. + fn flush_steers(&mut self, cursor: Cursor, through: Cursor) { + let (ready, waiting): (Vec<_>, Vec<_>) = std::mem::take(&mut self.steers) + .into_iter() + .partition(|(at, _, _)| *at <= through); + self.steers = waiting; + if ready.is_empty() { + return; + } + // A `Steer` is only ever queued into `self.steers` while a turn is + // already `self.turn` (see `Event::Steer` in `observe`), so this is + // always `Some` by the time anything reaches `ready`; the fallback + // exists only so this can never panic if that invariant is ever + // violated. + let turn = self + .turn + .clone() + .unwrap_or_else(|| TurnId::new(String::new())); + for (_, principal, text) in ready { + self.acting = Some(principal.clone()); + self.push( + cursor, + Message::User { + turn: turn.clone(), + message_id: None, + principal, + text, + attachments: Vec::new(), + }, + ); + } + } + + /// When every call of the open step has a result, the results enter + /// history in the model's call order. + fn close_step_if_resolved(&mut self, cursor: Cursor) { + let resolved = self.open_step.as_ref().is_some_and(|step| { + step.states + .iter() + .all(|state| matches!(state, CallState::Done(_))) + }); + if !resolved { + return; + } + let Some(step) = self.open_step.take() else { + return; + }; + for (call, state) in step.calls.into_iter().zip(step.states) { + if let CallState::Done(result) = state { + self.push( + cursor, + Message::Tool { + call: call.id, + name: call.tool, + outcome: result.outcome, + output: result.output, + }, + ); + } + } + } + + /// A new turn over an unfinished step: every call needs a result for the + /// model's history to stay well formed. + fn close_abandoned_step(&mut self, cursor: Cursor) { + if let Some(step) = &mut self.open_step { + for state in &mut step.states { + let result = match state { + CallState::Done(_) => continue, + CallState::Started => ToolResult::unknown(UNKNOWN_NEW_TURN), + _ => ToolResult::error(NOT_RUN_NEW_TURN), + }; + *state = CallState::Done(result); + } + } + self.close_step_if_resolved(cursor); + } +} diff --git a/vendor/dex-loop/src/engine.rs b/vendor/dex-loop/src/engine.rs new file mode 100644 index 000000000..ee514ba7a --- /dev/null +++ b/vendor/dex-loop/src/engine.rs @@ -0,0 +1,1197 @@ +//! The loop. +//! +//! Per step: `StepStarted` → model stream (text to the log as it arrives) → +//! `ModelStepCompleted` (the commit point: every proposed call, with full +//! arguments) → per call, in order: policy → approval → `ToolStarted` → +//! effect → `ToolFinished`. The turn ends when a step proposes no calls. + +use std::pin::pin; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + +use futures_util::StreamExt; +use futures_util::stream::FuturesUnordered; +use tokio_util::sync::CancellationToken; + +use crate::budget::{Budget, BudgetAxis}; +use crate::compaction::{Compactor, NoCompaction}; +use crate::context::{CallState, Context, Decision, Status}; +use crate::event::{ + ApprovalId, ApprovalMode, CallId, ErrorCode, Event, HEADLESS_AUTO_APPROVER, Outcome, + PrincipalId, ProposedCall, ToolName, ToolResult, TurnId, +}; +use crate::ports::{ + Claim, Effects, ExecutorKind, Fenced, GovernanceClass, Log, Model, ModelChunk, ModelError, + ToolSpec, Tools, Verdict, +}; +use crate::sanitize::{DeltaFilter, Sanitizer}; + +/// The engine-owned discovery tool. Always offered to the model. +pub const TOOLS_SEARCH: &str = "tools.search"; + +const NOT_RUN_INTERRUPTED: &str = "not run: the turn was interrupted"; +const UNKNOWN_INTERRUPTED: &str = + "outcome unknown: the turn was interrupted before the result was recorded"; +/// A call that already started (its tool vanished from the catalog, or the +/// ledger still shows it running) must never be told "unknown tool" or shown +/// a `Running` outcome that nothing will update: both invite the model to +/// retry, which mints a new `CallId` and can run the mutation twice. +const UNKNOWN_NO_RETRY: &str = "outcome unknown: this call already started; do not retry it without first checking whether it took effect"; +const APPROVER_DECLINED: &str = "denied: the approver declined this call"; +const APPROVAL_MISMATCH: &str = "denied: the approval does not match this call's arguments"; +const MISSING_QUESTION: &str = "invalid call: args.question must be a non-empty string"; +const MISSING_QUERY: &str = "invalid call: args.query must be a non-empty string"; +const CLIENT_TIMED_OUT_READ: &str = + "not completed: the client did not report a result in time; it is safe to try again"; +const CLIENT_TIMED_OUT_MUTATION: &str = + "outcome unknown: the client did not report a result in time; check before trying again"; + +/// How long a call waits for `Event::ClientToolResult` before the engine +/// gives up on it. Matches `dex_tools::Timeouts::default().client`; a host +/// that wants a different wait calls `Engine::with_client_timeout`. +const DEFAULT_CLIENT_TIMEOUT: Duration = Duration::from_secs(120); + +/// How long one `Tools::run` may take before the engine stops waiting and +/// finishes the call itself: `Failed` for a read (safe to retry), `Unknown` +/// for a mutation (recorded in the ledger, so a resume adopts the same +/// answer). The engine's own outer bound, whatever the executor: a tool port +/// that never returns (a stranded remote dispatch, a hung in-process HTTP +/// call) used to hold `Engine::run` open indefinitely, because `budget.wall` +/// is only checked between steps. The deadline is also clamped to the wall +/// budget's remaining time, so a turn never outlives `budget.wall` by more +/// than the time it takes to append the result. A host that wants a +/// different bound calls `Engine::with_tool_call_deadline`. +pub const DEFAULT_TOOL_CALL_DEADLINE: Duration = Duration::from_secs(5 * 60); +const DEADLINE_READ: &str = + "not completed: the call did not finish within Dex's time limit; it is safe to try again"; +const DEADLINE_MUTATION: &str = "outcome unknown: the call did not finish within Dex's time limit; check whether it took effect before trying again"; + +/// Why `Engine::run` returned. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Exit { + /// The model answered without tool calls. + Done, + /// Waiting for `Event::ApprovalDecided`; call `run` again after it lands. + Parked(ApprovalId), + /// Waiting for `Event::Answer` to this call; call `run` again after it lands. + Asked(CallId), + /// Waiting for `Event::ClientToolResult` for this call; call `run` again + /// after it lands. + AwaitingClientTool(CallId), + Interrupted, + /// `Event::Error` was appended. + Failed, +} + +/// What `dispatch_client_tool` did with the call it was given. +enum ClientToolOutcome { + /// Resolved without effect (denied, mismatched digest): the loop over + /// calls continues. + Continue, + /// The pending wave was flushed and an interrupt arrived: the loop over + /// calls stops early, same as the other `flush`-checking branches. + Break, + /// The turn parked: return this from `dispatch` at once. + Exit(Exit), +} + +/// The current client call and any decision recorded before this run. +struct ClientCall<'a> { + proposal: &'a ProposedCall, + decision: Option, +} + +/// One result of racing the model stream against `cancel` and the wall +/// budget in `model_step`. +enum StreamStep { + Chunk(Result), + /// The stream ended on its own (a truncated or otherwise finite stream). + Ended, + Cancelled, + /// `budget.wall` elapsed while waiting for the next chunk: the stream + /// itself never errored or ended, so nothing else would have caught this. + WallExceeded, +} + +/// The agent loop. Holds the host's ports and no state of its own, so any +/// replica can run any thread from its log. +pub struct Engine { + log: L, + model: M, + tools: T, + effects: E, + sanitizer: S, + compactor: C, + budget: Budget, + search: ToolSpec, + client_timeout: Duration, + tool_call_deadline: Duration, +} + +impl Engine +where + L: Log, + M: Model, + T: Tools, + E: Effects, + S: Sanitizer, +{ + pub fn new(log: L, model: M, tools: T, effects: E, sanitizer: S, budget: Budget) -> Self { + Self { + log, + model, + tools, + effects, + sanitizer, + compactor: NoCompaction, + budget, + search: search_spec(), + client_timeout: DEFAULT_CLIENT_TIMEOUT, + tool_call_deadline: DEFAULT_TOOL_CALL_DEADLINE, + } + } +} + +impl Engine +where + L: Log, + M: Model, + T: Tools, + E: Effects, + S: Sanitizer, + C: Compactor, +{ + pub fn with_compactor(self, compactor: C2) -> Engine { + Engine { + log: self.log, + model: self.model, + tools: self.tools, + effects: self.effects, + sanitizer: self.sanitizer, + compactor, + budget: self.budget, + search: self.search, + client_timeout: self.client_timeout, + tool_call_deadline: self.tool_call_deadline, + } + } + + /// How long a `Client`-executor call waits for `Event::ClientToolResult` + /// before the engine gives up on it and finishes the call as timed out. + /// Defaults to `dex_tools::Timeouts::default().client` (120s). + pub fn with_client_timeout(mut self, client_timeout: Duration) -> Self { + self.client_timeout = client_timeout; + self + } + + /// How long one `Tools::run` may take before the engine finishes the + /// call as timed out (see [`DEFAULT_TOOL_CALL_DEADLINE`]). + pub fn with_tool_call_deadline(mut self, tool_call_deadline: Duration) -> Self { + self.tool_call_deadline = tool_call_deadline; + self + } + + /// The time one call started now may run: the per-call deadline, or the + /// wall budget's remaining time if that is shorter. + fn call_deadline(&self, run_started: Instant) -> Duration { + self.tool_call_deadline + .min(self.budget.wall.saturating_sub(run_started.elapsed())) + } + + /// Runs the current turn from wherever `ctx` stands until it finishes, + /// parks, or is interrupted. `ctx` is usually fresh from `rehydrate`; the + /// same call resumes after a restart, an approval, or an answer. + /// + /// The host appends `Interrupt` and then cancels `cancel`. `Err(Fenced)` + /// means a write was refused and nothing more was appended. + pub async fn run(&self, ctx: &mut Context, cancel: &CancellationToken) -> Result { + let started = Instant::now(); + loop { + self.read_control(ctx).await?; + match ctx.status() { + Status::Idle | Status::Done => return Ok(Exit::Done), + Status::Interrupted => return Ok(Exit::Interrupted), + Status::Failed => return Ok(Exit::Failed), + Status::Running => {} + } + if let Some(step) = ctx.open_attempt() { + // A crash mid-stream: the attempt's text never committed. + self.emit(ctx, vec![Event::ModelAttemptAbandoned { step }]) + .await?; + } + if ctx.interrupt_requested() || cancel.is_cancelled() { + return self.interrupt(ctx).await; + } + if ctx.open_step().is_some() { + if let Some(exit) = self.dispatch(ctx, cancel, started).await? { + return Ok(exit); + } + continue; + } + if let Some(axis) = self + .budget + .exhausted(ctx.step(), ctx.usage(), started.elapsed()) + { + let message = self.budget_message(ctx, axis); + self.emit( + ctx, + vec![Event::Error { + code: ErrorCode::BudgetExhausted, + message, + }], + ) + .await?; + return Ok(Exit::Failed); + } + if let Some(plan) = self.compactor.plan(ctx).await { + self.emit( + ctx, + vec![Event::Compaction { + covers_to_cursor: plan.covers_to, + summary: plan.summary, + }], + ) + .await?; + } + if let Some(exit) = self.model_step(ctx, cancel, started).await? { + return Ok(exit); + } + } + } + + async fn read_control(&self, ctx: &mut Context) -> Result<(), Fenced> { + for (cursor, event) in self.log.control_since(ctx.control_cursor()).await? { + ctx.observe(cursor, &event); + } + Ok(()) + } + + /// One model attempt, ending in `ModelStepCompleted` (and `Final` when + /// the turn is done), or `ModelAttemptAbandoned` + `Error` on failure + /// (including a stream that never completes: `budget.wall` bounds the + /// whole `Engine::run` call, streaming included, not just the time + /// between steps). + async fn model_step( + &self, + ctx: &mut Context, + cancel: &CancellationToken, + started: Instant, + ) -> Result, Fenced> { + let step = ctx.step().saturating_add(1); + self.emit( + ctx, + vec![Event::StepStarted { + step, + control_through: ctx.control_cursor(), + }], + ) + .await?; + // Read after `StepStarted`: it places queued steers, whose authors + // become the acting principal. + let (Some(turn), Some(principal)) = (ctx.turn().cloned(), ctx.acting_principal().cloned()) + else { + return Ok(Some(Exit::Done)); + }; + + let mut filter = self.sanitizer.filter(); + let mut text = String::new(); + let mut calls = Vec::new(); + let mut failure = None; + let mut wall_exceeded = false; + // Usage arrives mid-stream but is never written on its own: it is + // batched into whichever terminal event ends this attempt + // (`ModelStepCompleted` or `ModelAttemptAbandoned`), so one attempt + // costs one append instead of one per usage chunk plus one for the + // terminal event. `Context::observe` sums `Usage` regardless of + // position in the batch, so budget accounting is unaffected. + let mut pending_usage = Vec::new(); + // The step's provider continuation state, committed with the step + // (the model sends it only after a clean terminal). A second chunk + // replaces the first: one step has one. + let mut reasoning = None; + { + // The answer-only call offers nothing, not even `tools.search`. + let owned = if self.budget.answer_only(step.saturating_sub(1)) { + Vec::new() + } else { + self.offered(ctx) + }; + let offered: Vec<&ToolSpec> = owned.iter().collect(); + let mut stream = pin!(self.model.stream(ctx, &offered)); + // Dropping the stream on cancel is safe: a model call has no + // effects to wait for. A stream that never yields another chunk + // and is never cancelled would otherwise hang here forever, past + // `budget.wall`: race every chunk against the wall deadline too, + // not only against `cancel`. + loop { + let remaining = self.budget.wall.saturating_sub(started.elapsed()); + let outcome = tokio::select! { + biased; + () = cancel.cancelled() => StreamStep::Cancelled, + () = tokio::time::sleep(remaining) => StreamStep::WallExceeded, + item = stream.next() => match item { + Some(chunk) => StreamStep::Chunk(chunk), + None => StreamStep::Ended, + }, + }; + match outcome { + StreamStep::Chunk(Ok(ModelChunk::Text(delta))) => { + let safe = filter.push(&delta); + if !safe.is_empty() { + text.push_str(&safe); + self.log.append_text(safe).await?; + } + } + StreamStep::Chunk(Ok(ModelChunk::ToolCall { name, args })) => { + let id = call_id(&turn, step, calls.len()); + calls.push(ProposedCall::new(id, name, args, principal.clone())); + } + StreamStep::Chunk(Ok(ModelChunk::Usage(usage))) => { + pending_usage.push(Event::Usage(usage)); + } + StreamStep::Chunk(Ok(ModelChunk::Reasoning(state))) => { + reasoning = Some(state); + } + StreamStep::Chunk(Err(error)) => { + failure = Some(error.message); + break; + } + StreamStep::Ended | StreamStep::Cancelled => break, + StreamStep::WallExceeded => { + wall_exceeded = true; + break; + } + } + } + if failure.is_none() && !wall_exceeded { + let tail = filter.finish(); + if !tail.is_empty() { + text.push_str(&tail); + self.log.append_text(tail).await?; + } + } + } + + if wall_exceeded { + let message = self.budget_message(ctx, BudgetAxis::Wall); + let mut events = pending_usage; + events.push(Event::ModelAttemptAbandoned { step }); + events.push(Event::Error { + code: ErrorCode::BudgetExhausted, + message, + }); + self.emit(ctx, events).await?; + return Ok(Some(Exit::Failed)); + } + if let Some(message) = failure { + let mut events = pending_usage; + events.push(Event::ModelAttemptAbandoned { step }); + events.push(Event::Error { + code: ErrorCode::ModelFailed, + message, + }); + self.emit(ctx, events).await?; + return Ok(Some(Exit::Failed)); + } + if cancel.is_cancelled() { + // Keep what the customer saw; calls from a cut stream never run, + // so their continuation state is not kept either. + let mut events = pending_usage; + events.push(Event::ModelStepCompleted { + step, + text, + calls: Vec::new(), + reasoning: None, + }); + self.emit(ctx, events).await?; + return self.interrupt(ctx).await.map(Some); + } + if !calls.is_empty() && self.budget.answer_only(step.saturating_sub(1)) { + // Asked for a tool on the call that offered none. Nothing can run + // it, so the turn ends here instead of looping. + let message = self.budget_message(ctx, BudgetAxis::Steps); + let mut events = pending_usage; + events.push(Event::ModelAttemptAbandoned { step }); + events.push(Event::Error { + code: ErrorCode::BudgetExhausted, + message, + }); + self.emit(ctx, events).await?; + return Ok(Some(Exit::Failed)); + } + if calls.is_empty() { + // A steer that arrived during the answer continues the turn. + self.read_control(ctx).await?; + let continues = ctx.has_queued_steers() && !ctx.interrupt_requested(); + let mut events = pending_usage; + events.push(Event::ModelStepCompleted { + step, + text: text.clone(), + calls: Vec::new(), + reasoning, + }); + if !continues { + events.push(Event::Final { text }); + } + self.emit(ctx, events).await?; + return Ok((!continues).then_some(Exit::Done)); + } + let mut events = pending_usage; + events.push(Event::ModelStepCompleted { + step, + text, + calls, + reasoning, + }); + self.emit(ctx, events).await?; + Ok(None) + } + + /// Dispatches the open step's calls in the model's order. Allowed + /// read-only calls collect into a wave that runs in parallel; anything + /// else runs the pending wave first, so effects keep the model's order. + async fn dispatch( + &self, + ctx: &mut Context, + cancel: &CancellationToken, + run_started: Instant, + ) -> Result, Fenced> { + let Some(step) = ctx.open_step() else { + return Ok(None); + }; + let calls = step.calls.clone(); + let mut wave: Vec = Vec::new(); + for (index, call) in calls.iter().enumerate() { + // Interrupt stops at the next effect boundary. + if cancel.is_cancelled() { + break; + } + let state = ctx + .open_step() + .and_then(|step| step.states.get(index)) + .cloned(); + let decision = match state { + None | Some(CallState::Done(_)) => continue, + Some(CallState::Asked { answer: None }) => { + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + return Ok(Some(Exit::Asked(call.id.clone()))); + } + Some(CallState::Asked { + answer: Some(answer), + }) => { + self.finish(ctx, call, ToolResult::text(answer)).await?; + continue; + } + Some(CallState::AwaitingClient { + result: None, + deadline_ms, + }) => { + if now_ms() >= deadline_ms { + self.timeout_client_call(ctx, call).await?; + continue; + } + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + return Ok(Some(Exit::AwaitingClientTool(call.id.clone()))); + } + Some(CallState::AwaitingClient { + result: Some(result), + .. + }) => { + self.finish_client_result(ctx, call, result).await?; + continue; + } + Some(CallState::Parked { + approval, + decision: None, + }) => { + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + return Ok(Some(Exit::Parked(approval))); + } + Some(CallState::Parked { + decision: Some(decision), + .. + }) => Some(decision), + Some(CallState::Started) => { + // Started before a restart and never finished. Reads run + // again; mutations resolve through the ledger, which + // never dispatches a claimed call a second time. + if call.tool.as_str() == TOOLS_SEARCH { + self.search_tools(ctx, call).await?; + continue; + } + match self.offered_spec(ctx, &call.tool) { + Some(spec) if spec.read_only => wave.push(index), + Some(_) => { + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + self.run_mutation(ctx, call, run_started).await?; + } + // The tool is no longer offered (deploy, grant + // revoke), but this call already started: it may + // have run. Resolve through the ledger instead of + // telling the model to retry with a different tool, + // which would dispatch a new call id for the same + // mutation. + None => self.resolve_started(ctx, call).await?, + } + continue; + } + Some(CallState::Todo) => None, + }; + + if call.tool.as_str() == TOOLS_SEARCH { + self.search_tools(ctx, call).await?; + continue; + } + let Some(spec) = self.offered_spec(ctx, &call.tool) else { + self.finish(ctx, call, unknown_tool(&call.tool)).await?; + continue; + }; + // Client-executor tools are not in `self.tools`'s catalog and + // carry their own governance (set by the host's allowlist when + // it resolved the client's declaration), so they skip + // `Tools::policy`: that port models the registry's grants, + // guardrails and guardian, none of which apply to a tool an + // ephemeral client session declared for this turn only. + if spec.executor == ExecutorKind::Client { + match self + .dispatch_client_tool( + ctx, + &calls, + &mut wave, + cancel, + run_started, + ClientCall { + proposal: call, + decision, + }, + ) + .await? + { + ClientToolOutcome::Continue => continue, + ClientToolOutcome::Break => break, + ClientToolOutcome::Exit(exit) => return Ok(Some(exit)), + } + } + // Current policy first, even for a decided call: a revoked + // grant or changed policy denies it. + let verdict = match self.tools.policy(ctx, call).await { + Verdict::Deny(reason) => { + self.finish(ctx, call, ToolResult::error(format!("denied: {reason}"))) + .await?; + continue; + } + verdict => verdict, + }; + if let Some(decision) = decision.as_ref() { + if decision.args_digest != call.args_digest { + self.finish(ctx, call, ToolResult::error(APPROVAL_MISMATCH)) + .await?; + continue; + } + if !decision.approved { + self.finish(ctx, call, ToolResult::error(APPROVER_DECLINED)) + .await?; + continue; + } + } + if let (None, Verdict::NeedsApproval { approval, summary }) = (decision, verdict) { + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + if !self + .request_approval(ctx, call, approval.clone(), summary) + .await? + { + return Ok(Some(Exit::Parked(approval))); + } + } + + if spec.executor == ExecutorKind::User { + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + let Some(question) = non_empty_str(call, "question") else { + self.finish(ctx, call, ToolResult::error(MISSING_QUESTION)) + .await?; + continue; + }; + self.emit( + ctx, + vec![Event::Question { + call: call.id.clone(), + text: question.to_owned(), + }], + ) + .await?; + return Ok(Some(Exit::Asked(call.id.clone()))); + } + if spec.read_only { + wave.push(index); + } else { + if self + .flush(ctx, &calls, &mut wave, cancel, run_started) + .await? + { + break; + } + self.run_mutation(ctx, call, run_started).await?; + } + } + self.run_wave(ctx, &calls, wave, cancel, run_started) + .await?; + if cancel.is_cancelled() { + return self.interrupt(ctx).await.map(Some); + } + Ok(None) + } + + /// One call to a `Client`-executor tool: approval (if the host's + /// allowlist marked it a mutation), then `ClientToolRequested`, mirroring + /// how the main `dispatch` loop handles `NeedsApproval` and `User`. + async fn dispatch_client_tool( + &self, + ctx: &mut Context, + calls: &[ProposedCall], + wave: &mut Vec, + cancel: &CancellationToken, + run_started: Instant, + client_call: ClientCall<'_>, + ) -> Result { + let ClientCall { + proposal: call, + decision, + } = client_call; + let Some(spec) = self.offered_spec(ctx, &call.tool) else { + self.finish(ctx, call, unknown_tool(&call.tool)).await?; + return Ok(ClientToolOutcome::Continue); + }; + if let Some(decision) = decision.as_ref() { + if decision.args_digest != call.args_digest { + self.finish(ctx, call, ToolResult::error(APPROVAL_MISMATCH)) + .await?; + return Ok(ClientToolOutcome::Continue); + } + if !decision.approved { + self.finish(ctx, call, ToolResult::error(APPROVER_DECLINED)) + .await?; + return Ok(ClientToolOutcome::Continue); + } + } else if spec.governance == GovernanceClass::Approval { + if self.flush(ctx, calls, wave, cancel, run_started).await? { + return Ok(ClientToolOutcome::Break); + } + let approval = ApprovalId::new(format!("client-{}", call.id)); + let summary = format!("Run {} in your browser", spec.label); + if !self + .request_approval(ctx, call, approval.clone(), summary) + .await? + { + return Ok(ClientToolOutcome::Exit(Exit::Parked(approval))); + } + } + if self.flush(ctx, calls, wave, cancel, run_started).await? { + return Ok(ClientToolOutcome::Break); + } + let deadline_ms = now_ms().saturating_add(self.client_timeout_millis()); + self.emit( + ctx, + vec![Event::ClientToolRequested { + call: call.id.clone(), + tool: call.tool.clone(), + args: call.args.clone(), + label: spec.label.clone(), + principal: call.principal.clone(), + target_session: call.principal.to_string(), + deadline_ms, + }], + ) + .await?; + Ok(ClientToolOutcome::Exit(Exit::AwaitingClientTool( + call.id.clone(), + ))) + } + + /// Runs the pending wave before a call that must not overlap it. Returns + /// true when an interrupt arrived meanwhile: the next effect must not + /// start. + async fn flush( + &self, + ctx: &mut Context, + calls: &[ProposedCall], + wave: &mut Vec, + cancel: &CancellationToken, + run_started: Instant, + ) -> Result { + self.run_wave(ctx, calls, std::mem::take(wave), cancel, run_started) + .await?; + Ok(cancel.is_cancelled()) + } + + /// Runs read-only calls concurrently. Each `ToolFinished` is appended as + /// its call returns; history receives the results in call order when the + /// step closes. + async fn run_wave( + &self, + ctx: &mut Context, + calls: &[ProposedCall], + wave: Vec, + cancel: &CancellationToken, + run_started: Instant, + ) -> Result<(), Fenced> { + if wave.is_empty() || cancel.is_cancelled() { + return Ok(()); + } + let starts: Vec = wave + .iter() + .filter_map(|&index| { + let call = &calls[index]; + let spec = self.offered_spec(ctx, &call.tool)?; + Some(started(call, &spec)) + }) + .collect(); + self.emit(ctx, starts).await?; + let thread = ctx.thread().clone(); + let thread = &thread; + // One deadline for the wave: its reads run concurrently, so each + // gets the full time. A read that overruns is dropped and finished + // `Failed`; a read has no effect to wait for, so retrying is safe. + let deadline = self.call_deadline(run_started); + let mut running: FuturesUnordered<_> = wave + .iter() + .map(|&index| { + let call = &calls[index]; + async move { + let run = self.tools.run(thread, call, cancel); + let result = match tokio::time::timeout(deadline, run).await { + Ok(result) => result, + Err(_elapsed) => ToolResult::error(DEADLINE_READ), + }; + (call, result) + } + }) + .collect(); + // On `Fenced` the remaining reads are dropped: a stale owner must not + // append, and the new owner runs them again. + while let Some((call, result)) = running.next().await { + self.finish(ctx, call, result).await?; + } + Ok(()) + } + + /// Claim, dispatch, record. A claimed call is never dispatched again: + /// its recorded outcome is adopted instead, with `Running` settled to + /// `Unknown` first — nothing ever revisits a `Running` report, so + /// showing it as final would leave the model unable to tell whether to + /// retry. Interrupt does not cancel a mutation that has started. + async fn run_mutation( + &self, + ctx: &mut Context, + call: &ProposedCall, + run_started: Instant, + ) -> Result<(), Fenced> { + let Some(spec) = self.offered_spec(ctx, &call.tool) else { + // Unreachable from today's two call sites, which already + // checked `offered_spec` before calling in; kept correct in + // case a future caller skips that check. A vanished tool for a + // call that already started must resolve through the ledger, + // never as "unknown tool". + return self.resolve_started(ctx, call).await; + }; + match self.effects.claim(call).await? { + Claim::Existing(result) => { + let result = self.settle_claim(&call.id, result).await?; + self.finish(ctx, call, result).await + } + Claim::Granted => { + self.emit(ctx, vec![started(call, &spec)]).await?; + let never = CancellationToken::new(); + // A mutation that overruns its deadline is dropped, not + // cancelled: the effect may still land. `Unknown` is recorded + // under the claim, so a later resume of this call adopts it + // instead of dispatching the mutation a second time. + let run = self.tools.run(ctx.thread(), call, &never); + let result = match tokio::time::timeout(self.call_deadline(run_started), run).await + { + Ok(result) => result, + Err(_elapsed) => ToolResult::unknown(DEADLINE_MUTATION), + }; + self.effects.record(&call.id, &result).await?; + self.finish(ctx, call, result).await + } + } + } + + /// A `Started` call resolved without dispatching anything: its tool + /// vanished from the offered catalog (deploy, grant revoke). Never + /// "unknown tool" for a call that already started — the model would be + /// told to retry, minting a new `CallId` for a mutation that may have + /// already run. A prior ledger claim is settled the same way + /// `run_mutation` settles one; a call the ledger never saw (it may have + /// been a read) is claimed now and recorded `Unknown`, so a later resume + /// gets the same answer instead of claiming it again. + async fn resolve_started(&self, ctx: &mut Context, call: &ProposedCall) -> Result<(), Fenced> { + let result = match self.effects.claim(call).await? { + Claim::Existing(result) => self.settle_claim(&call.id, result).await?, + Claim::Granted => { + let result = ToolResult::unknown(UNKNOWN_NO_RETRY); + self.effects.record(&call.id, &result).await?; + result + } + }; + self.finish(ctx, call, result).await + } + + /// Settles a claimed call's recorded outcome for the model: `Running` + /// becomes `Unknown` (and the ledger is updated to match, so a later + /// resume adopts the same answer); every other outcome passes through. + async fn settle_claim( + &self, + call_id: &CallId, + result: ToolResult, + ) -> Result { + if result.outcome != Outcome::Running { + return Ok(result); + } + let result = ToolResult::unknown(UNKNOWN_NO_RETRY); + self.effects.record(call_id, &result).await?; + Ok(result) + } + + /// `tools.search`: matched catalog tools are offered from the next step. + async fn search_tools(&self, ctx: &mut Context, call: &ProposedCall) -> Result<(), Fenced> { + self.emit(ctx, vec![started(call, &self.search)]).await?; + let Some(query) = non_empty_str(call, "query") else { + return self + .finish(ctx, call, ToolResult::error(MISSING_QUERY)) + .await; + }; + let matches: Vec = self + .tools + .search(&call.principal, query) + .await + .iter() + .filter_map(|name| self.tools.spec(name)) + .cloned() + .collect(); + let listing = if matches.is_empty() { + format!("no tools match {query:?}") + } else { + matches + .iter() + .map(|spec| format!("{}: {}", spec.name, spec.label)) + .collect::>() + .join("\n") + }; + let mut events = Vec::with_capacity(2); + if !matches.is_empty() { + events.push(Event::ToolsExposed { + call: call.id.clone(), + tools: matches.into_iter().map(|spec| spec.name).collect(), + }); + } + events.push(finished(call, ToolResult::text(listing))); + self.emit(ctx, events).await + } + + /// Every call without a result gets one, then `Interrupted`. + async fn interrupt(&self, ctx: &mut Context) -> Result { + let mut events: Vec = ctx + .open_step() + .map(|step| { + step.calls + .iter() + .zip(&step.states) + .filter_map(|(call, state)| match state { + CallState::Done(_) => None, + CallState::Started => { + Some(finished(call, ToolResult::unknown(UNKNOWN_INTERRUPTED))) + } + _ => Some(finished(call, ToolResult::error(NOT_RUN_INTERRUPTED))), + }) + .collect() + }) + .unwrap_or_default(); + events.push(Event::Interrupted); + self.emit(ctx, events).await?; + Ok(Exit::Interrupted) + } + + /// The specs the model sees this step: `tools.search`, core tools, and + /// tools exposed earlier in the turn. A turn's client-declared tools + /// reach the model only if the host's `Tools::catalog()` already + /// includes them (see `Context::client_tools`, and + /// `dex_tools::client::declare` for dex-runtime's host); the engine + /// does not merge them in itself, so they are never offered twice. + fn offered(&self, ctx: &Context) -> Vec { + std::iter::once(self.search.clone()) + .chain( + self.tools + .catalog() + .iter() + .filter(|spec| spec.core || ctx.exposed_tools().contains(&spec.name)) + .cloned(), + ) + .collect() + } + + /// A tool the model was offered. Calls to anything else are unknown. + fn offered_spec(&self, ctx: &Context, name: &ToolName) -> Option { + if name == &self.search.name { + return Some(self.search.clone()); + } + self.tools + .spec(name) + .filter(|spec| spec.core || ctx.exposed_tools().contains(name)) + .cloned() + } + + fn budget_message(&self, ctx: &Context, axis: BudgetAxis) -> String { + let budget = &self.budget; + match axis { + BudgetAxis::Steps => format!( + "step budget exhausted: {} steps and the answer-only step after them", + budget.max_steps + ), + BudgetAxis::Tokens => format!( + "token budget exhausted: {} of {} tokens", + ctx.usage().tokens(), + budget.max_tokens + ), + BudgetAxis::Cost => format!( + "cost budget exhausted: {} of {} micros", + ctx.usage().cost_micros, + budget.max_cost_micros + ), + BudgetAxis::Wall => format!("wall budget exhausted: {:?}", budget.wall), + } + } + + async fn finish( + &self, + ctx: &mut Context, + call: &ProposedCall, + result: ToolResult, + ) -> Result<(), Fenced> { + self.emit(ctx, vec![finished(call, result)]).await + } + + fn client_timeout_millis(&self) -> i64 { + i64::try_from(self.client_timeout.as_millis()).unwrap_or(i64::MAX) + } + + /// An `AwaitingClient` call whose `Event::ClientToolResult` is already in + /// the log: wraps `raw` through the host (untrusted-content marking, + /// output storage) and, for a mutation, through the effect ledger -- + /// exactly the shaping a live dispatch of any other executor gets, and + /// exactly once per call id, whether this is the first time the result + /// is seen or a replay after a restart lands on the same state. Without + /// this, a crash between `ClientToolResult` and `ToolFinished` would + /// finish the call with the client's raw, unwrapped text. + async fn finish_client_result( + &self, + ctx: &mut Context, + call: &ProposedCall, + raw: ToolResult, + ) -> Result<(), Fenced> { + match self.offered_spec(ctx, &call.tool) { + Some(spec) if !spec.read_only => { + let result = match self.effects.claim(call).await? { + Claim::Existing(existing) => self.settle_claim(&call.id, existing).await?, + Claim::Granted => { + let wrapped = self.tools.wrap_client_result(ctx.thread(), call, raw).await; + self.effects.record(&call.id, &wrapped).await?; + wrapped + } + }; + self.finish(ctx, call, result).await + } + Some(_) => { + let wrapped = self.tools.wrap_client_result(ctx.thread(), call, raw).await; + self.finish(ctx, call, wrapped).await + } + // The tool vanished from the offered catalog (deploy, grant + // revoke) between the request and the result: finish with what + // the client reported rather than lose it to "unknown tool". + None => self.finish(ctx, call, raw).await, + } + } + + /// An `AwaitingClient` call whose deadline has passed with no + /// `Event::ClientToolResult` in the log: finishes it as timed out, + /// through the same ledger guard as `finish_client_result` for a + /// mutation, so a later replay that lands on the same expired deadline + /// adopts the one recorded outcome instead of manufacturing (and + /// ledgering) a second one. + async fn timeout_client_call( + &self, + ctx: &mut Context, + call: &ProposedCall, + ) -> Result<(), Fenced> { + match self.offered_spec(ctx, &call.tool) { + Some(spec) if !spec.read_only => { + let result = match self.effects.claim(call).await? { + Claim::Existing(existing) => self.settle_claim(&call.id, existing).await?, + Claim::Granted => { + let timed_out = ToolResult::unknown(CLIENT_TIMED_OUT_MUTATION); + self.effects.record(&call.id, &timed_out).await?; + timed_out + } + }; + self.finish(ctx, call, result).await + } + _ => { + self.finish(ctx, call, ToolResult::error(CLIENT_TIMED_OUT_READ)) + .await + } + } + } + + /// Appends and observes. + /// Logs an `ApprovalRequested` for a call policy said must ask. Returns + /// `true` when the call is already decided and dispatch continues. + /// + /// An `Interactive` turn parks (returns `false`) until a human's + /// `ApprovalDecided` arrives. A `Headless` turn has no human, so the + /// request and an approving `ApprovalDecided` from + /// `HEADLESS_AUTO_APPROVER` are appended together: the audit trail is the + /// same pair a human approval would leave. Only reached for a + /// `NeedsApproval` verdict; `Deny` was handled before this point and stays + /// denied. + async fn request_approval( + &self, + ctx: &mut Context, + call: &ProposedCall, + approval: ApprovalId, + summary: String, + ) -> Result { + let mut events = vec![Event::ApprovalRequested { + call: call.id.clone(), + approval: approval.clone(), + args_digest: call.args_digest.clone(), + summary, + }]; + let headless = ctx.approval_mode() == ApprovalMode::Headless; + if headless { + events.push(Event::ApprovalDecided { + call: call.id.clone(), + approval, + args_digest: call.args_digest.clone(), + approved: true, + principal: PrincipalId::new(HEADLESS_AUTO_APPROVER), + }); + } + // The engine wrote the decision itself: it must not move the control + // cursor past a `Steer` or `Interrupt` that landed in between. + let control = ctx.control_cursor(); + self.emit(ctx, events).await?; + if headless { + ctx.rewind_control(control); + } + Ok(headless) + } + + async fn emit(&self, ctx: &mut Context, events: Vec) -> Result<(), Fenced> { + if events.is_empty() { + return Ok(()); + } + let cursors = self.log.append(&events).await?; + if cursors.len() != events.len() { + return Err(Fenced::new(format!( + "log returned {} cursors for {} events", + cursors.len(), + events.len() + ))); + } + for (cursor, event) in cursors.into_iter().zip(&events) { + ctx.observe(cursor, event); + } + Ok(()) + } +} + +fn search_spec() -> ToolSpec { + ToolSpec { + name: ToolName::new(TOOLS_SEARCH), + label: "Finding the right tools".into(), + schema: serde_json::json!({ + "type": "object", + "properties": {"query": {"type": "string", "description": "What you need to do"}}, + "required": ["query"], + }), + read_only: true, + core: true, + governance: GovernanceClass::Plain, + executor: ExecutorKind::InProcess, + } +} + +fn call_id(turn: &TurnId, step: u32, index: usize) -> CallId { + CallId(format!("{turn}-{step}-{index}")) +} + +/// Unix milliseconds, for comparing against `ClientToolRequested::deadline_ms`. +/// A clock that cannot read (`UNIX_EPOCH` in the future) or a duration wider +/// than `i64` counts as "now is very late": both saturate toward always +/// expiring a deadline rather than panicking or parking forever. +fn now_ms() -> i64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|since_epoch| i64::try_from(since_epoch.as_millis()).unwrap_or(i64::MAX)) + .unwrap_or(i64::MAX) +} + +fn non_empty_str<'a>(call: &'a ProposedCall, key: &str) -> Option<&'a str> { + call.args + .get(key) + .and_then(serde_json::Value::as_str) + .filter(|value| !value.trim().is_empty()) +} + +fn unknown_tool(name: &ToolName) -> ToolResult { + ToolResult::error(format!( + "unknown tool: {name}; use {TOOLS_SEARCH} to find tools" + )) +} + +fn started(call: &ProposedCall, spec: &ToolSpec) -> Event { + Event::ToolStarted { + call: call.id.clone(), + tool: call.tool.clone(), + label: spec.label.clone(), + principal: call.principal.clone(), + } +} + +fn finished(call: &ProposedCall, result: ToolResult) -> Event { + Event::ToolFinished { + call: call.id.clone(), + outcome: result.outcome, + output: result.output, + receipt: result.receipt, + } +} diff --git a/vendor/dex-loop/src/event.rs b/vendor/dex-loop/src/event.rs new file mode 100644 index 000000000..5c7816ff6 --- /dev/null +++ b/vendor/dex-loop/src/event.rs @@ -0,0 +1,779 @@ +//! The event log vocabulary: every row a thread's log holds, and the values +//! those rows carry. + +use std::fmt; +use std::ops::AddAssign; + +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; + +macro_rules! string_id { + ($(#[$meta:meta])* $name:ident) => { + $(#[$meta])* + #[derive(Clone, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] + #[serde(transparent)] + pub struct $name(pub String); + + impl $name { + pub fn new(value: impl Into) -> Self { + Self(value.into()) + } + + pub fn as_str(&self) -> &str { + &self.0 + } + } + + impl fmt::Display for $name { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(&self.0) + } + } + }; +} + +string_id!( + /// One user turn. Host-assigned; must match `[A-Za-z0-9_-]+` because call + /// ids are derived from it and model providers restrict tool-call ids. + TurnId +); +string_id!( + /// The source message a `UserMessage` was appended for. Host-constructed + /// and validated by the host: an opaque, bounded string (1 to 256 + /// bytes) with no ASCII control characters or whitespace, but otherwise + /// unrestricted, since real ids (e.g. platform-api's + /// `message:human:`) are not limited to `TurnId`'s narrower + /// `[A-Za-z0-9_-]` charset. The value is carried byte for byte, never + /// normalized. Optional because log rows written before this field + /// existed carry none; `#[serde(default)]` on + /// `Event::UserMessage::message_id` makes those rows deserialize to + /// `None` rather than fail. + MessageId +); +string_id!( + /// One tool call. `"{turn}-{step}-{index}"`, assigned by the engine. + /// Unique only within its thread: `TurnId` is caller-chosen and not + /// guaranteed unique across threads, so the same `CallId` string can + /// occur in two different threads. A downstream idempotency key built + /// from this alone can collide; see `Tools::run`. + CallId +); +string_id!( + /// A registry tool name. Internal: surfaces show `ToolSpec::label`. + ToolName +); +string_id!( + /// One approval request, issued by `Tools::policy`. + ApprovalId +); +string_id!( + /// The principal who decided an approval. + PrincipalId +); +string_id!( + /// A reference into tenant-scoped storage holding a tool's output. + OutputRef +); +string_id!( + /// A governance receipt written for a tool call. + ReceiptId +); +string_id!( + /// A reference to a user-supplied attachment in tenant-scoped storage. + ArtifactRef +); + +/// The tenant scope of one thread. Every log read and write carries all three. +#[derive(Clone, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] +pub struct ThreadId { + pub org: String, + pub workspace: String, + pub thread: String, +} + +/// Position of an event in a thread's log. Strictly increasing per thread. +#[derive( + Clone, Copy, Debug, Default, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize, +)] +#[serde(transparent)] +pub struct Cursor(pub i64); + +impl Cursor { + /// Before the first event. + pub const START: Cursor = Cursor(0); +} + +/// Opaque provider continuation state for one model step: what a provider +/// requires the client to send back unmodified with that step's calls on the +/// next request (Gemini function-call thought signatures, Anthropic signed +/// thinking blocks, OpenAI encrypted reasoning items). +/// +/// dex-loop stores it with the step and hands it back in history; it never +/// reads `payload`. Only the `Model` port that wrote it interprets it. +/// +/// `payload` can hold model-written text about tenant data (Anthropic +/// thinking summaries), so `Debug` prints its size, never its content; keep +/// it out of tracing, metrics and catch-all metadata. +#[derive(Clone, PartialEq, Serialize, Deserialize)] +pub struct ProviderReasoning { + /// The payload's shape, e.g. `google.gemini.v1`, + /// `anthropic.messages.v1`, `openai.responses.v1`. + pub format: String, + /// The provider model id that produced it. + pub model: String, + /// Opaque provider payload. Never edited by dex-loop. + pub payload: serde_json::Value, +} + +impl fmt::Debug for ProviderReasoning { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("ProviderReasoning") + .field("format", &self.format) + .field("model", &self.model) + .field( + "payload", + &format_args!("<{} bytes>", self.payload.to_string().len()), + ) + .finish() + } +} + +/// Model spend reported by one model response. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct Usage { + pub input_tokens: u64, + pub output_tokens: u64, + pub cost_micros: u64, +} + +impl Usage { + pub fn tokens(&self) -> u64 { + self.input_tokens.saturating_add(self.output_tokens) + } +} + +impl AddAssign for Usage { + fn add_assign(&mut self, other: Usage) { + self.input_tokens = self.input_tokens.saturating_add(other.input_tokens); + self.output_tokens = self.output_tokens.saturating_add(other.output_tokens); + self.cost_micros = self.cost_micros.saturating_add(other.cost_micros); + } +} + +/// The default for `ClientToolRequested::deadline_ms` on a row written +/// before that field existed: never expires, rather than timing out every +/// call still parked from before this field shipped. +fn never_expires() -> i64 { + i64::MAX +} + +/// Hex SHA-256 of serialized tool arguments. +pub fn args_digest(args: &serde_json::Value) -> String { + // Serializing a `serde_json::Value` into a Vec cannot fail: every key is + // a string and there is no I/O. + let bytes = serde_json::to_vec(args).unwrap_or_default(); + hex::encode(Sha256::digest(bytes)) +} + +/// A tool call the model proposed. Durable from `ModelStepCompleted` on: +/// park, resume and replay all read it from the log, never from memory. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct ProposedCall { + pub id: CallId, + pub tool: ToolName, + pub args: serde_json::Value, + /// An approval authorizes only a call with this digest. + pub args_digest: String, + /// The principal whose authority the call runs under: the author of the + /// most recent user input the step acts on. + pub principal: PrincipalId, +} + +impl ProposedCall { + pub fn new( + id: CallId, + tool: ToolName, + args: serde_json::Value, + principal: PrincipalId, + ) -> Self { + let args_digest = args_digest(&args); + Self { + id, + tool, + args, + args_digest, + principal, + } + } +} + +/// One tool a client session declared it can execute locally, resolved by +/// the host against its per-surface allowlist before this reaches the log: +/// `read_only` here is the host's decision, never the client's own claim. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct ClientToolSpec { + pub name: ToolName, + pub schema: serde_json::Value, + pub read_only: bool, + /// The only tool text a surface may show. + pub label: String, +} + +/// The durable outcome of a call. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Outcome { + Succeeded, + Failed, + /// Dispatched; completion is not yet known. + Running, + /// The effect may or may not have happened. Never replayed automatically. + Unknown, +} + +/// What a tool call produced, as the model will see it. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(tag = "kind", content = "value", rename_all = "snake_case")] +pub enum Output { + /// Tool output stored in tenant-scoped storage. The `Model` port resolves + /// it when it renders history; the log never holds the bytes. + Ref(OutputRef), + /// A short note: a denial, an unknown tool, a user's answer, an outcome + /// the engine could not recover. + Text(String), +} + +/// The outcome of one tool call. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct ToolResult { + pub outcome: Outcome, + pub output: Output, + pub receipt: Option, +} + +impl ToolResult { + /// A successful call whose output lives in storage. + pub fn stored(output: OutputRef, receipt: Option) -> Self { + Self { + outcome: Outcome::Succeeded, + output: Output::Ref(output), + receipt, + } + } + + /// A successful call with a short inline result. + pub fn text(text: impl Into) -> Self { + Self { + outcome: Outcome::Succeeded, + output: Output::Text(text.into()), + receipt: None, + } + } + + /// A failed call. The model sees `message`. + pub fn error(message: impl Into) -> Self { + Self { + outcome: Outcome::Failed, + output: Output::Text(message.into()), + receipt: None, + } + } + + /// A call whose effect may or may not have happened. + pub fn unknown(message: impl Into) -> Self { + Self { + outcome: Outcome::Unknown, + output: Output::Text(message.into()), + receipt: None, + } + } +} + +/// Why a turn stopped with `Event::Error`. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ErrorCode { + BudgetExhausted, + ModelFailed, +} + +impl ErrorCode { + pub fn as_str(self) -> &'static str { + match self { + ErrorCode::BudgetExhausted => "budget_exhausted", + ErrorCode::ModelFailed => "model_failed", + } + } +} + +/// Who can answer a turn's approval requests. +/// +/// `Headless` turns come from callers with no human to click Approve (service +/// accounts, workloads, agents, synthetic canaries, API automation). For them +/// the engine resolves a policy `NeedsApproval` verdict itself, recording the +/// request and the decision in the log. A `Deny` verdict is never affected. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ApprovalMode { + /// A human decides: the turn parks on `ApprovalRequested`. + #[default] + Interactive, + /// No human: policy approves what would otherwise ask. + Headless, +} + +impl ApprovalMode { + pub fn as_str(self) -> &'static str { + match self { + ApprovalMode::Interactive => "interactive", + ApprovalMode::Headless => "headless", + } + } +} + +/// The principal recorded on an `ApprovalDecided` the engine wrote itself +/// for a `Headless` turn. +pub const HEADLESS_AUTO_APPROVER: &str = "policy:headless_auto_approve"; + +/// One row in a thread's log. Hosts append the ingress events (`UserMessage`, +/// `Steer`, `Interrupt`, `ApprovalDecided`, `Answer`, and optionally +/// `ToolProgress`); the engine appends everything else. +/// +/// Surfaces render `TextDelta`, labels, summaries and questions. Tool names in +/// `ModelStepCompleted`, `ToolStarted` and `ToolsExposed` are internal and +/// for replay only. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +#[serde(tag = "type", rename_all = "snake_case")] +pub enum Event { + /// Starts a turn under `principal`. + UserMessage { + turn: TurnId, + /// The source message this turn was sent for, when the host has one. + /// `#[serde(default)] so log rows written before this field existed + /// still deserialize, to `None`. + #[serde(default)] + message_id: Option, + principal: PrincipalId, + text: String, + attachments: Vec, + /// Tools this turn's client session can execute locally, already + /// resolved by the host against its allowlist. `#[serde(default)]` + /// so log rows written before this field existed still deserialize. + #[serde(default)] + client_tools: Vec, + /// Owner tools admitted by the authenticated host for this principal. + /// Older log rows carry no authority. Public clients cannot set this. + #[serde(default)] + authorized_tools: Vec, + /// Whether a human can answer this turn's approval requests. Older + /// log rows carry none and replay as `Interactive`, the behaviour + /// they were written under. + #[serde(default)] + approval_mode: ApprovalMode, + }, + /// Control: becomes a user message from `principal` before the next model + /// call. Calls the model then proposes act under `principal`. + Steer { + principal: PrincipalId, + text: String, + }, + /// Control: cancels the model stream and running read-only calls, and + /// stops before the next effect. A started mutation completes. + Interrupt { + principal: PrincipalId, + }, + /// Control: the decision on a parked call. Approval is necessary, not + /// sufficient: policy runs again on resume, then `args_digest` must match. + ApprovalDecided { + call: CallId, + approval: ApprovalId, + args_digest: String, + approved: bool, + principal: PrincipalId, + }, + /// Control: the user's reply to a `Question`. + Answer { + call: CallId, + principal: PrincipalId, + text: String, + }, + /// Control: a client session's outcome for one `ClientToolRequested` + /// call. Only `Outcome::Succeeded` or `Outcome::Failed` are accepted + /// from a client; the host validates that before appending this event. + ClientToolResult { + call: CallId, + principal: PrincipalId, + outcome: Outcome, + output: String, + }, + + /// A model attempt begins. Steers at or before `control_through` were + /// placed in history before this call. + StepStarted { + step: u32, + control_through: Cursor, + }, + /// Customer-safe model text (already through the `Sanitizer`). The `Log` + /// coalesces deltas, so one row may hold many model chunks. + TextDelta { + text: String, + }, + Usage(Usage), + /// The commit point of a model attempt: its full text and every proposed + /// call with full arguments, appended before any policy check or + /// execution. An interrupted stream completes with the text so far and no + /// calls. + ModelStepCompleted { + step: u32, + text: String, + calls: Vec, + /// The step's provider continuation state (`ModelChunk::Reasoning`), + /// returned in history on `Message::Assistant`. Internal: never + /// rendered or sent to a surface. `#[serde(default)]` so log rows + /// written before this field existed still deserialize, to `None`. + #[serde(default, skip_serializing_if = "Option::is_none")] + reasoning: Option, + }, + /// A model attempt with no `ModelStepCompleted` (a crash mid-stream or a + /// model failure). Its streamed text is dropped from model context; + /// renderers remove it. The call is issued again as a new step. + ModelAttemptAbandoned { + step: u32, + }, + + ToolStarted { + call: CallId, + tool: ToolName, + label: String, + principal: PrincipalId, + }, + /// Host-appended progress for a running call, e.g. "Starting computer". + ToolProgress { + call: CallId, + label: String, + }, + /// `tools.search` matched these tools; their schemas are offered to the + /// model from the next step of this turn. + ToolsExposed { + call: CallId, + tools: Vec, + }, + ToolFinished { + call: CallId, + outcome: Outcome, + output: Output, + receipt: Option, + }, + /// The turn is parked until an `ApprovalDecided` for this call arrives. + ApprovalRequested { + call: CallId, + approval: ApprovalId, + args_digest: String, + summary: String, + }, + /// The turn is parked until an `Answer` for this call arrives. + Question { + call: CallId, + text: String, + }, + /// The turn is parked until a `ClientToolResult` for this call arrives. + /// The only event that carries tool arguments to a surface; the host + /// delivers it only to the session that declared `tool`, never to every + /// Watch subscriber on the thread. + ClientToolRequested { + call: CallId, + /// The client-declared name from this session's `ClientToolSpec`, + /// not an internal registry name. + tool: ToolName, + args: serde_json::Value, + label: String, + /// The principal whose authority this call runs under. A + /// `ClientToolResult` for this call must come from the same + /// principal. + principal: PrincipalId, + /// Currently the declaring principal's id: this API has no separate + /// session identity yet, so it does not distinguish two concurrent + /// sessions for the same principal. A real session id can replace + /// this value without changing the event's shape. + target_session: String, + /// Unix milliseconds after which this call's client wait counts as + /// lost. Set once, when the call is first requested, and never moved + /// afterward: a restart rehydrates this same value from the log, so + /// the wait's deadline survives the actor that started it. Rows + /// written before this field existed have no deadline of their own; + /// `#[serde(default)]` reads them back as never expiring rather than + /// timing out every call still parked from before this field shipped. + #[serde(default = "never_expires")] + deadline_ms: i64, + }, + /// History up to and including `covers_to_cursor` is replaced by `summary`. + Compaction { + covers_to_cursor: Cursor, + summary: String, + }, + + /// The turn finished; `text` is the last step's text. + Final { + text: String, + }, + Error { + code: ErrorCode, + message: String, + }, + Interrupted, +} + +impl Event { + /// Control events are the ones the engine reads back with + /// `Log::control_since` while it runs. + pub fn is_control(&self) -> bool { + matches!( + self, + Event::Steer { .. } + | Event::Interrupt { .. } + | Event::ApprovalDecided { .. } + | Event::Answer { .. } + | Event::ClientToolResult { .. } + ) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn events_round_trip_through_json() { + let events = vec![ + Event::UserMessage { + turn: TurnId::new("t1"), + message_id: Some(MessageId::new("m1")), + principal: PrincipalId::new("alice"), + text: "hi".into(), + attachments: vec![ArtifactRef::new("a1")], + authorized_tools: Vec::new(), + approval_mode: ApprovalMode::Interactive, + client_tools: vec![ClientToolSpec { + name: ToolName::new("browser.read_tab"), + schema: serde_json::json!({"type": "object"}), + read_only: true, + label: "Reading your tab".into(), + }], + }, + Event::ClientToolRequested { + call: CallId::new("t1-1-0"), + tool: ToolName::new("browser.read_tab"), + args: serde_json::json!({}), + label: "Reading your tab".into(), + principal: PrincipalId::new("alice"), + target_session: "alice".into(), + deadline_ms: 1_700_000_000_000, + }, + Event::ClientToolResult { + call: CallId::new("t1-1-0"), + principal: PrincipalId::new("alice"), + outcome: Outcome::Succeeded, + output: "ok".into(), + }, + Event::Usage(Usage { + input_tokens: 1, + output_tokens: 2, + cost_micros: 3, + }), + Event::ModelStepCompleted { + step: 1, + text: "ok".into(), + calls: vec![ProposedCall::new( + CallId::new("t1-1-0"), + ToolName::new("search"), + serde_json::json!({"q": "x"}), + PrincipalId::new("alice"), + )], + reasoning: None, + }, + Event::ModelStepCompleted { + step: 2, + text: String::new(), + calls: Vec::new(), + reasoning: Some(ProviderReasoning { + format: "google.gemini.v1".into(), + model: "gemini-3.6-flash".into(), + payload: serde_json::json!({"calls": [{"thought_signature": "c2ln"}]}), + }), + }, + Event::ToolFinished { + call: CallId::new("t1-1-0"), + outcome: Outcome::Unknown, + output: Output::Text("outcome unknown".into()), + receipt: None, + }, + Event::Error { + code: ErrorCode::BudgetExhausted, + message: "steps".into(), + }, + Event::Interrupt { + principal: PrincipalId::new("bob"), + }, + Event::Interrupted, + ]; + for event in events { + let json = serde_json::to_string(&event).expect("serialize"); + let back: Event = serde_json::from_str(&json).expect("deserialize"); + assert_eq!(back, event, "{json}"); + } + } + + #[test] + fn a_client_tool_requested_row_written_before_deadline_ms_existed_never_expires() { + let json = serde_json::json!({ + "type": "client_tool_requested", + "call": "t1-1-0", + "tool": "browser.read_tab", + "args": {}, + "label": "Reading your tab", + "principal": "alice", + "target_session": "alice", + }) + .to_string(); + let event: Event = serde_json::from_str(&json).expect("deserialize"); + match event { + Event::ClientToolRequested { deadline_ms, .. } => { + assert_eq!(deadline_ms, i64::MAX); + } + other => panic!("expected ClientToolRequested, got {other:?}"), + } + } + + #[test] + fn approval_mode_defaults_to_interactive_for_old_rows_and_round_trips() { + let old_row = serde_json::json!({ + "type": "user_message", + "turn": "t1", + "principal": "alice", + "text": "hi", + "attachments": [], + }); + match serde_json::from_value::(old_row).expect("old row deserializes") { + Event::UserMessage { approval_mode, .. } => { + assert_eq!(approval_mode, ApprovalMode::Interactive); + } + other => panic!("expected UserMessage, got {other:?}"), + } + assert_eq!( + serde_json::to_value(ApprovalMode::Headless).expect("serialize"), + serde_json::json!("headless") + ); + assert_eq!( + serde_json::from_value::(serde_json::json!("interactive")) + .expect("deserialize"), + ApprovalMode::Interactive + ); + } + + #[test] + fn user_message_without_message_id_round_trips_to_none() { + let event = Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: PrincipalId::new("alice"), + text: "hi".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: ApprovalMode::Interactive, + }; + let json = serde_json::to_string(&event).expect("serialize"); + let back: Event = serde_json::from_str(&json).expect("deserialize"); + assert_eq!(back, event); + + // A log row written before `message_id` existed has no such key at + // all: `#[serde(default)]` must still deserialize it, to `None`, + // rather than fail. + let old_row = serde_json::json!({ + "type": "user_message", + "turn": "t1", + "principal": "alice", + "text": "hi", + "attachments": [], + }); + let back: Event = + serde_json::from_value(old_row).expect("old row without message_id deserializes"); + assert_eq!( + back, + Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: PrincipalId::new("alice"), + text: "hi".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: ApprovalMode::Interactive, + } + ); + } + + /// A step logged before `reasoning` existed has no such key: it still + /// deserializes, to `None`, and a step without reasoning is written + /// without the key, so old readers see the row they always did. + #[test] + fn a_model_step_row_without_reasoning_replays_to_none() { + let old_row = serde_json::json!({ + "type": "model_step_completed", + "step": 1, + "text": "Let me check.", + "calls": [{ + "id": "t1-1-0", + "tool": "search", + "args": {"q": "x"}, + "args_digest": args_digest(&serde_json::json!({"q": "x"})), + "principal": "alice", + }], + }); + let event: Event = + serde_json::from_value(old_row.clone()).expect("old step row deserializes"); + let Event::ModelStepCompleted { reasoning, .. } = &event else { + panic!("expected ModelStepCompleted, got {event:?}"); + }; + assert_eq!(*reasoning, None); + assert_eq!(serde_json::to_value(&event).expect("serialize"), old_row); + } + + /// `Debug` (and so any `{:?}` of an event) never prints the payload. + #[test] + fn reasoning_debug_omits_the_payload() { + let reasoning = ProviderReasoning { + format: "anthropic.messages.v1".into(), + model: "claude-opus-5-5".into(), + payload: serde_json::json!({"content": [{"type": "thinking", "thinking": "tenant secret"}]}), + }; + let printed = format!("{reasoning:?}"); + assert!(!printed.contains("tenant secret"), "{printed}"); + assert!(printed.contains("anthropic.messages.v1"), "{printed}"); + } + + #[test] + fn digest_depends_only_on_args() { + let a = ProposedCall::new( + CallId::new("c"), + ToolName::new("t"), + serde_json::json!({"q": "x"}), + PrincipalId::new("alice"), + ); + let b = ProposedCall::new( + CallId::new("other"), + ToolName::new("t"), + serde_json::json!({"q": "x"}), + PrincipalId::new("bob"), + ); + let c = ProposedCall::new( + CallId::new("c"), + ToolName::new("t"), + serde_json::json!({"q": "y"}), + PrincipalId::new("alice"), + ); + assert_eq!(a.args_digest, b.args_digest); + assert_ne!(a.args_digest, c.args_digest); + assert_eq!(a.args_digest.len(), 64); + } +} diff --git a/vendor/dex-loop/src/lib.rs b/vendor/dex-loop/src/lib.rs new file mode 100644 index 000000000..535a91c8a --- /dev/null +++ b/vendor/dex-loop/src/lib.rs @@ -0,0 +1,46 @@ +//! `dex-loop`: the one Dex agent loop, shared by every surface. +//! +//! The loop appends the user's message, streams the model, commits the +//! proposed tool calls to the log, dispatches them (allowed read-only calls +//! in one parallel wave, mutations one at a time through the effect ledger, +//! all in the model's order), appends the results, and repeats until the +//! model answers without tool calls. Every step is an event in the thread's +//! log; the log is the only state. +//! +//! The crate holds no I/O. The host supplies the ports: +//! [`Log`], [`Model`], [`Tools`], [`Effects`], [`Sanitizer`], and optionally +//! a [`Compactor`]. It never spawns tasks; parallel tool calls are polled +//! concurrently inside [`Engine::run`]. +//! +//! ```text +//! host: append UserMessage +//! ctx = rehydrate(thread, log) // same code path warm or after a crash +//! engine.run(&mut ctx, &cancel) -> Exit // Done | Parked | Asked | Interrupted | Failed +//! host: on approve/answer, append the decision and call run again +//! ``` + +mod budget; +mod compaction; +mod context; +mod engine; +mod event; +mod ports; +mod rehydrate; +mod sanitize; + +pub use budget::{Budget, BudgetAxis}; +pub use compaction::{Compaction, Compactor, NoCompaction, Summarize, Threshold}; +pub use context::{Context, Entry, Message}; +pub use engine::{DEFAULT_TOOL_CALL_DEADLINE, Engine, Exit, TOOLS_SEARCH}; +pub use event::{ + ApprovalId, ApprovalMode, ArtifactRef, CallId, ClientToolSpec, Cursor, ErrorCode, Event, + HEADLESS_AUTO_APPROVER, MessageId, Outcome, Output, OutputRef, PrincipalId, ProposedCall, + ProviderReasoning, ReceiptId, ThreadId, ToolName, ToolResult, TurnId, Usage, args_digest, +}; +pub use ports::{ + Claim, Effects, ExecutorKind, Fenced, GovernanceClass, Log, Model, ModelChunk, ModelError, + ToolSpec, Tools, Verdict, +}; +pub use rehydrate::rehydrate; +pub use sanitize::{DeltaFilter, Lexicon, LexiconFilter, Sanitizer}; +pub use tokio_util::sync::CancellationToken; diff --git a/vendor/dex-loop/src/ports.rs b/vendor/dex-loop/src/ports.rs new file mode 100644 index 000000000..ba5e7f64e --- /dev/null +++ b/vendor/dex-loop/src/ports.rs @@ -0,0 +1,227 @@ +//! The host ports. Each is implemented once, by the service that runs the +//! engine; tests implement them in memory. + +use std::future::Future; + +use futures_util::Stream; +use serde::{Deserialize, Serialize}; +use tokio_util::sync::CancellationToken; + +use crate::context::Context; +use crate::event::{ + ApprovalId, CallId, Cursor, Event, PrincipalId, ProposedCall, ProviderReasoning, ThreadId, + ToolName, ToolResult, Usage, +}; + +/// The log or the effect ledger refused a write. The engine stops at once and +/// appends nothing more. Returned for a lost lease and for any write that +/// cannot be made durable; the next lease holder rehydrates from the log. +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +#[error("write fenced: {reason}")] +pub struct Fenced { + pub reason: String, +} + +impl Fenced { + pub fn new(reason: impl Into) -> Self { + Self { + reason: reason.into(), + } + } +} + +/// The thread's event log, the only source of truth. +pub trait Log: Send + Sync { + /// Appends `events` in order and returns one cursor per event. Everything + /// buffered by `append_text` is written first. + fn append(&self, events: &[Event]) -> impl Future, Fenced>> + Send; + + /// Customer-visible model text. The log coalesces it into `TextDelta` + /// rows of 25-100 ms or a few KB each, so the stored text is what the + /// customer saw; the engine never assumes one row per call. + fn append_text(&self, text: String) -> impl Future> + Send; + + /// Control events (`Event::is_control`) with a cursor greater than + /// `after`, in log order. + fn control_since( + &self, + after: Cursor, + ) -> impl Future, Fenced>> + Send; +} + +/// One chunk of a streaming model response. +#[derive(Clone, Debug, PartialEq)] +pub enum ModelChunk { + Text(String), + /// The engine assigns the `CallId`; provider call ids are not used. + ToolCall { + name: ToolName, + args: serde_json::Value, + }, + Usage(Usage), + /// The step's opaque provider continuation state. At most one per step, + /// sent only after a clean terminal (the same commit rule as `ToolCall` + /// and `Usage`), after the last `ToolCall` and before `Usage`. The engine + /// stores it on `ModelStepCompleted`; history returns it on + /// `Message::Assistant`. + Reasoning(ProviderReasoning), +} + +/// The model call failed after the `Model` port's own retries. +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +#[error("model call failed: {message}")] +pub struct ModelError { + pub message: String, +} + +/// The model, streaming. +pub trait Model: Send + Sync { + /// Streams one response to `ctx.history()` with `tools` offered. The + /// engine drops the stream to cancel it. + fn stream<'a>( + &'a self, + ctx: &'a Context, + tools: &'a [&'a ToolSpec], + ) -> impl Stream> + Send + 'a; +} + +/// Which governance a tool falls under. Read by `Tools::policy`, not by the +/// engine. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum GovernanceClass { + Plain, + Guardrails, + Guardian, + Approval, +} + +/// Where a call runs. The engine only distinguishes `User`: those calls are +/// answered by a person, so the engine asks instead of running them. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ExecutorKind { + InProcess, + ToolExecutor, + Computer, + /// The call's `args.question` string is shown to the user; the answer is + /// the result. + User, + /// Runs on the declaring client session (a browser tab, a desktop + /// client), not on the host. The engine parks the call as + /// `ClientToolRequested` and resumes on that session's + /// `ClientToolResult`, exactly as `User` parks on `Answer`. + Client, +} + +/// One registry entry. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct ToolSpec { + pub name: ToolName, + /// The only tool text a surface may show. + pub label: String, + pub schema: serde_json::Value, + /// Read-only calls that policy allows run together in one parallel wave, + /// and may run again after a restart. Every other call is a mutation and + /// goes through `Effects`. + pub read_only: bool, + /// Offered to the model on every step. Other tools are offered only after + /// `tools.search` exposes them in the turn. + pub core: bool, + pub governance: GovernanceClass, + pub executor: ExecutorKind, +} + +/// The policy decision for one call. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Verdict { + Allow, + /// The model sees `"denied: {reason}"` as the call's result. + Deny(String), + /// Park the turn. `summary` is customer-safe and shown on the approval card. + NeedsApproval { + approval: ApprovalId, + summary: String, + }, +} + +/// The tool registry, policy, and executors. +pub trait Tools: Send + Sync { + /// Every tool this turn may use, core or discoverable. + fn catalog(&self) -> &[ToolSpec]; + + fn spec(&self, name: &ToolName) -> Option<&ToolSpec> { + self.catalog().iter().find(|spec| &spec.name == name) + } + + /// Catalog tools matching `query` that `principal` may use, best first. + fn search( + &self, + principal: &PrincipalId, + query: &str, + ) -> impl Future> + Send; + + /// Current authorization for one call under `call.principal`: governance, + /// grants, guardrails and guardian. Called in the model's call order just + /// before the call would run, and again when a parked call resumes. + fn policy(&self, ctx: &Context, call: &ProposedCall) -> impl Future + Send; + + /// Runs one call. `call.id` is unique only within `thread`: a turn id is + /// caller-chosen, so the same `CallId` string can occur in two different + /// threads. Where the downstream system needs a globally unique + /// idempotency key, build it from `thread` and `call.id` together, not + /// `call.id` alone. For read-only calls `cancel` fires on interrupt and + /// the engine waits for the call to return; for mutations it never + /// fires. + fn run( + &self, + thread: &ThreadId, + call: &ProposedCall, + cancel: &CancellationToken, + ) -> impl Future + Send; + + /// Finishes a call an `ExecutorKind::Client` session already reported an + /// outcome for. The engine never dispatches these through `run` (the + /// client, not this port, already ran the call); it calls this instead, + /// exactly once per call id, to let the host wrap untrusted client + /// content and store a large output before the result reaches history -- + /// the same shaping a live dispatch of any other executor gets. Called + /// identically whether `raw` just arrived or is being adopted on replay + /// after a restart, so the two produce the same wrapped result. The + /// default passes `raw` through unchanged. + fn wrap_client_result( + &self, + thread: &ThreadId, + call: &ProposedCall, + raw: ToolResult, + ) -> impl Future + Send { + let _ = thread; + let _ = call; + std::future::ready(raw) + } +} + +/// The answer to `Effects::claim`. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Claim { + /// First claim for this call: dispatch it, then `record` the outcome. + Granted, + /// Claimed before. The recorded result, reconciled with downstream where + /// possible; `Outcome::Running` or `Outcome::Unknown` if not. The engine + /// never reports `Running` to the model as a call's final outcome — + /// nothing revisits it — so it settles `Running` to `Unknown` first. + Existing(ToolResult), +} + +/// The durable effect ledger, keyed by `CallId`. Every mutation is claimed +/// before dispatch and recorded after, so a mutation is dispatched at most +/// once per call id, across restarts and replicas. +pub trait Effects: Send + Sync { + fn claim(&self, call: &ProposedCall) -> impl Future> + Send; + + fn record( + &self, + call: &CallId, + result: &ToolResult, + ) -> impl Future> + Send; +} diff --git a/vendor/dex-loop/src/rehydrate.rs b/vendor/dex-loop/src/rehydrate.rs new file mode 100644 index 000000000..1fb4e12e4 --- /dev/null +++ b/vendor/dex-loop/src/rehydrate.rs @@ -0,0 +1,41 @@ +//! Rebuilding a thread's context from its log. + +use crate::context::Context; +use crate::event::{Cursor, Event, ThreadId}; + +/// Rebuilds the context a warm actor would hold after these events. +/// +/// `events` is the thread's log in cursor order: all of it, or a suffix that +/// starts at or before both the current turn's `UserMessage` and the latest +/// `Compaction`'s `covers_to_cursor`. +/// +/// What `Engine::run` then does with the result: +/// - a `StepStarted` with no `ModelStepCompleted` gets `ModelAttemptAbandoned` +/// and the model call is issued again as a new step; +/// - a read-only call with `ToolStarted` but no `ToolFinished` runs again; a +/// mutation goes to the `Effects` ledger, which returns its recorded outcome +/// (or `Running`/`Unknown`) instead of dispatching it again; +/// - a parked call waits for its `ApprovalDecided`, then policy runs again, +/// the approval digest is checked, and the calls after it in the same step +/// are dispatched in order, all read from `ModelStepCompleted`. +pub fn rehydrate(thread: ThreadId, events: &[(Cursor, Event)]) -> Context { + let mut ctx = Context::new(thread); + // Kernel contract: no control event with a cursor before this replay's + // starting point is ever re-applied -- not by this replay (it only ever + // observes `events`, which already excludes anything earlier) and not + // afterward, by the engine's own `Log::control_since(ctx.control_cursor())` + // re-fetching what this suffix deliberately left out. A suffix with no + // control-kind event in it at all would otherwise leave + // `control_cursor()` at its default, under-reporting where this replay + // starts and letting an old, already-resolved control event (most + // notably an `Interrupt` for a turn that ended before this one started) + // be fetched again and observed against whatever turn is running now. + // See `Context::advance_control_floor`. + if let Some((first, _)) = events.first() { + ctx.advance_control_floor(*first); + } + for (cursor, event) in events { + ctx.observe(*cursor, event); + } + ctx +} diff --git a/vendor/dex-loop/src/sanitize.rs b/vendor/dex-loop/src/sanitize.rs new file mode 100644 index 000000000..813154725 --- /dev/null +++ b/vendor/dex-loop/src/sanitize.rs @@ -0,0 +1,172 @@ +//! Customer-safe text: every text delta passes through a `Sanitizer` before +//! it is appended, so what streams is what persists. + +use std::sync::Arc; + +/// Makes one `DeltaFilter` per model call. +pub trait Sanitizer: Send + Sync { + type Filter: DeltaFilter; + + fn filter(&self) -> Self::Filter; +} + +/// A streaming filter over one model call's text. +pub trait DeltaFilter: Send { + /// Returns the text that is safe to emit now. Text that could still be the + /// start of a forbidden term is held back. + fn push(&mut self, text: &str) -> String; + + /// Returns whatever is still held back, filtered. Called once at the end. + fn finish(&mut self) -> String; +} + +/// Replaces forbidden terms (internal tool names, product and host names) +/// with customer-safe replacements, ASCII case-insensitively, holding back at +/// most one term's length of text. The host builds it from the registry +/// (tool name to label) and its fixed lexicon. Linear in text length times +/// lexicon size; a host with a large lexicon can supply an automaton instead. +#[derive(Clone, Debug, Default)] +pub struct Lexicon { + /// Longest term first, so the longest match wins. + terms: Arc<[(String, String)]>, +} + +impl Lexicon { + pub fn new(terms: impl IntoIterator) -> Self + where + T: Into, + R: Into, + { + let mut terms: Vec<(String, String)> = terms + .into_iter() + .map(|(term, replacement)| (term.into(), replacement.into())) + .filter(|(term, _)| !term.is_empty()) + .collect(); + terms.sort_by_key(|(term, _)| std::cmp::Reverse(term.len())); + Self { + terms: terms.into(), + } + } +} + +impl Sanitizer for Lexicon { + type Filter = LexiconFilter; + + fn filter(&self) -> LexiconFilter { + LexiconFilter { + terms: Arc::clone(&self.terms), + held: String::new(), + } + } +} + +/// The per-call state of a `Lexicon`. +#[derive(Debug)] +pub struct LexiconFilter { + terms: Arc<[(String, String)]>, + held: String, +} + +impl DeltaFilter for LexiconFilter { + fn push(&mut self, text: &str) -> String { + self.held.push_str(text); + let (out, consumed) = scan(&self.terms, &self.held, false); + self.held.drain(..consumed); + out + } + + fn finish(&mut self) -> String { + let (out, _) = scan(&self.terms, &self.held, true); + self.held.clear(); + out + } +} + +/// Emits `buf` with terms replaced. Unless `eof`, stops at the first position +/// where more text could still complete a term, and returns how many bytes it +/// consumed. +fn scan(terms: &[(String, String)], buf: &str, eof: bool) -> (String, usize) { + let bytes = buf.as_bytes(); + let mut out = String::with_capacity(buf.len()); + let mut at = 0; + while at < bytes.len() { + let rest = &bytes[at..]; + let could_grow = terms.iter().any(|(term, _)| { + term.len() > rest.len() && term.as_bytes()[..rest.len()].eq_ignore_ascii_case(rest) + }); + if could_grow && !eof { + return (out, at); + } + let matched = terms.iter().find(|(term, _)| { + rest.len() >= term.len() && rest[..term.len()].eq_ignore_ascii_case(term.as_bytes()) + }); + if let Some((term, replacement)) = matched { + out.push_str(replacement); + // The matched bytes equal the term up to ASCII case, so they end + // on a char boundary of `buf`. + at += term.len(); + continue; + } + let Some(ch) = buf[at..].chars().next() else { + break; + }; + out.push(ch); + at += ch.len_utf8(); + } + (out, at) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn run(lexicon: &Lexicon, chunks: &[&str]) -> String { + let mut filter = lexicon.filter(); + let mut out: String = chunks.iter().map(|chunk| filter.push(chunk)).collect(); + out.push_str(&filter.finish()); + out + } + + #[test] + fn replaces_terms_split_across_chunks() { + let lexicon = Lexicon::new([ + ("linear.search_issues", "Linear search"), + ("Maestro", "Dex"), + ]); + assert_eq!( + run( + &lexicon, + &["I called line", "ar.search_iss", "ues via maes", "tro."] + ), + "I called Linear search via Dex." + ); + } + + #[test] + fn prefers_the_longest_term() { + let lexicon = Lexicon::new([("maestro", "Dex"), ("maestro-runtime", "the runtime")]); + assert_eq!( + run(&lexicon, &["a maestro", "-runtime b"]), + "a the runtime b" + ); + assert_eq!(run(&lexicon, &["a maestro", " b"]), "a Dex b"); + } + + #[test] + fn holds_back_only_a_possible_prefix() { + let lexicon = Lexicon::new([("sandboxwich", "the computer")]); + let mut filter = lexicon.filter(); + assert_eq!(filter.push("hello sandbox"), "hello "); + assert_eq!(filter.push("wiches!"), "the computeres!"); + assert_eq!(filter.push("héllo sa"), "héllo "); + assert_eq!(filter.finish(), "sa"); + } + + #[test] + fn empty_lexicon_passes_text_through() { + let lexicon = Lexicon::default(); + let mut filter = lexicon.filter(); + assert_eq!(filter.push("héllo"), "héllo"); + assert_eq!(filter.finish(), ""); + } +} diff --git a/vendor/dex-loop/tests/authorized_tools.rs b/vendor/dex-loop/tests/authorized_tools.rs new file mode 100644 index 000000000..c4b88fc72 --- /dev/null +++ b/vendor/dex-loop/tests/authorized_tools.rs @@ -0,0 +1,89 @@ +//! Owner-tool authority belongs to the admitted message, survives replay, +//! and cannot be inherited by a steering actor or a queued turn. +use dex_loop::{Context, Cursor, Event, PrincipalId, ThreadId, ToolName, TurnId}; + +fn message(turn: &str, principal: &str, tools: &[&str]) -> Event { + Event::UserMessage { + turn: TurnId::new(turn), + message_id: None, + principal: PrincipalId::new(principal), + text: "Find the demo request".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: tools.iter().map(|name| ToolName::new(*name)).collect(), + approval_mode: dex_loop::ApprovalMode::Interactive, + } +} + +fn context() -> Context { + Context::new(ThreadId { + org: "org".into(), + workspace: "workspace".into(), + thread: "thread".into(), + }) +} + +#[test] +fn authority_replays_with_the_original_principal_and_does_not_leak_to_the_queue() { + let events = [ + message("first", "reader", &["capture_form.get_submission"]), + Event::Steer { + principal: PrincipalId::new("writer"), + text: "Read it for me".into(), + }, + Event::StepStarted { + step: 1, + control_through: Cursor(2), + }, + message("second", "reader", &[]), + ]; + let mut live = context(); + for (index, event) in events.iter().enumerate() { + live.observe(Cursor(index as i64 + 1), event); + } + assert_eq!(live.acting_principal(), Some(&PrincipalId::new("writer"))); + assert_eq!( + live.authorized_principal(), + Some(&PrincipalId::new("reader")) + ); + assert_eq!( + live.authorized_tools(), + &[ToolName::new("capture_form.get_submission")] + ); + let mut replay = context(); + for (index, event) in events.iter().enumerate() { + let persisted = serde_json::to_vec(event).expect("persist event"); + let decoded = serde_json::from_slice(&persisted).expect("decode event"); + replay.observe(Cursor(index as i64 + 1), &decoded); + } + assert_eq!(replay.authorized_principal(), live.authorized_principal()); + assert_eq!(replay.authorized_tools(), live.authorized_tools()); + replay.observe( + Cursor(5), + &Event::Final { + text: "done".into(), + }, + ); + assert_eq!(replay.turn(), Some(&TurnId::new("second"))); + assert_eq!( + replay.authorized_principal(), + Some(&PrincipalId::new("reader")) + ); + assert!(replay.authorized_tools().is_empty()); +} + +#[test] +fn older_log_rows_have_no_owner_tool_authority() { + let mut persisted = serde_json::to_value(message("legacy", "reader", &[])).expect("event"); + assert!( + persisted + .as_object_mut() + .expect("object") + .remove("authorized_tools") + .is_some() + ); + let event: Event = serde_json::from_value(persisted).expect("legacy event"); + let mut restored = context(); + restored.observe(Cursor(1), &event); + assert!(restored.authorized_tools().is_empty()); +} diff --git a/vendor/dex-loop/tests/client_tool_replay.rs b/vendor/dex-loop/tests/client_tool_replay.rs new file mode 100644 index 000000000..60945f665 --- /dev/null +++ b/vendor/dex-loop/tests/client_tool_replay.rs @@ -0,0 +1,387 @@ +//! Crash/restart behavior for `ExecutorKind::Client` calls. +//! +//! Two gaps in how the engine resumed these calls after a restart: +//! - the wait's deadline lived only in the async task that requested it, so +//! a restart after `ClientToolRequested` rehydrated the call as +//! `AwaitingClient` with nothing left to ever time it out; +//! - a restart after `ClientToolResult` (but before `ToolFinished`) finished +//! the call with the client's raw, unwrapped text, skipping the wrap, +//! output storage and ledger record a live finish gets. +//! +//! Every scenario here builds the log directly (as `Event`s a prior, +//! now-crashed process would have appended) and then rehydrates, exactly the +//! shape a real restart takes. + +// This binary only exercises a slice of `support`'s shared helpers (each +// integration test file is its own compilation unit); the rest are real, +// used by `scenarios.rs`. +#[allow(dead_code)] +mod support; + +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use dex_loop::{ + ApprovalId, Budget, CancellationToken, Cursor, Event, Exit, Outcome, Output, ProposedCall, + ToolName, ToolResult, TurnId, +}; +use serde_json::json; +use support::*; + +fn budget() -> Budget { + Budget { + max_steps: 10, + max_tokens: 1_000_000, + max_cost_micros: 1_000_000, + wall: Duration::from_secs(30), + } +} + +fn now_ms() -> i64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("system clock before 1970") + .as_millis() as i64 +} + +/// `UserMessage`, `StepStarted`, `ModelStepCompleted` proposing `call` alone +/// -- the log a warm engine would have written before parking it. +fn log_up_to_model_step(log: &FakeLog, call: &ProposedCall) { + log.host_append(Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: alice(), + text: "do it".into(), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + log.host_append(Event::StepStarted { + step: 1, + control_through: Cursor::START, + }); + log.host_append(Event::ModelStepCompleted { + step: 1, + text: String::new(), + calls: vec![call.clone()], + reasoning: None, + }); +} + +/// The approval a mutating client tool parks for before it is requested, +/// already decided `approved`, as `dispatch_client_tool` requires before it +/// ever emits `ClientToolRequested`. +fn approved(log: &FakeLog, call: &ProposedCall) { + let approval = ApprovalId::new(format!("client-{}", call.id)); + log.host_append(Event::ApprovalRequested { + call: call.id.clone(), + approval: approval.clone(), + args_digest: call.args_digest.clone(), + summary: "Run it in your browser".into(), + }); + log.host_append(Event::ApprovalDecided { + call: call.id.clone(), + approval, + args_digest: call.args_digest.clone(), + approved: true, + principal: alice(), + }); +} + +fn requested(call: &ProposedCall, deadline_ms: i64) -> Event { + Event::ClientToolRequested { + call: call.id.clone(), + tool: call.tool.clone(), + args: call.args.clone(), + label: format!("Label for {}", call.tool), + principal: call.principal.clone(), + target_session: call.principal.to_string(), + deadline_ms, + } +} + +fn tool_finished(log: &FakeLog, call: &ProposedCall) -> Option<(Outcome, Output)> { + log.events().into_iter().find_map(|event| match event { + Event::ToolFinished { + call: id, + outcome, + output, + .. + } if id == call.id => Some((outcome, output)), + _ => None, + }) +} + +// ---------------------------------------------------------------- Gap 1: +// the deadline must survive a restart. + +// A restart landing right after `ClientToolRequested`, with time still left +// on the clock, must keep waiting -- not time the call out just because the +// process happened to restart. +#[tokio::test] +async fn a_restart_before_the_deadline_still_parks_on_the_client_tool() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.read_tab"), + json!({}), + alice(), + ); + log_up_to_model_step(&log, &call); + log.host_append(requested(&call, now_ms() + 60_000)); + + let model = FakeModel::new(vec![]); + let tools = FakeTools::new(vec![client_executed_tool("browser.read_tab", true)]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::AwaitingClientTool(call.id.clone())), + "a restart before the deadline must not time the call out early" + ); + assert_eq!(log.rehydrate(), ctx); +} + +// A restart landing after the deadline has already passed must finish the +// call as timed out on this very run, not park it forever: nothing else is +// ever going to re-arm a timer for it. +#[tokio::test] +async fn a_restart_after_the_deadline_has_passed_times_the_call_out() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.read_tab"), + json!({}), + alice(), + ); + log_up_to_model_step(&log, &call); + log.host_append(requested(&call, now_ms() - 1)); + + let model = FakeModel::new(vec![vec![text("moving on")]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.read_tab", true)]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + let (outcome, _) = tool_finished(&log, &call).expect("the call must finish, not park forever"); + assert_eq!(outcome, Outcome::Failed); + assert!( + !log.events() + .iter() + .any(|event| matches!(event, Event::ClientToolResult { .. })), + "no client ever answered; the engine's own deadline, not a late result, must have finished this call" + ); + assert_eq!(log.rehydrate(), ctx); +} + +// The same expiry for a mutation: it must go through the effect ledger, and +// a later restart landing on the same expired deadline must adopt the +// recorded outcome rather than manufacture (and ledger) a second one. +#[tokio::test] +async fn a_timed_out_mutation_is_ledgered_once() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.click"), + json!({"selector": "#buy"}), + alice(), + ); + log_up_to_model_step(&log, &call); + approved(&log, &call); + log.host_append(requested(&call, now_ms() - 1)); + + let effects = FakeEffects::default().seed( + call.id.clone(), + Some(ToolResult::unknown("previously recorded timeout")), + ); + let model = FakeModel::new(vec![vec![text("done")]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.click", false)]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + // The ledger already had an entry (as if a still-earlier crash had + // recorded one): this run must adopt it, not overwrite it with a fresh + // timeout message. + let (_, output) = tool_finished(&log, &call).expect("finished"); + assert_eq!(output, Output::Text("previously recorded timeout".into())); + assert_eq!( + effects.recorded(&call.id), + Some(Some(ToolResult::unknown("previously recorded timeout"))), + "the seeded ledger entry must be unchanged, not replaced by a second recording" + ); +} + +// ---------------------------------------------------------------- Gap 2: +// a replayed `ClientToolResult` must go through the same shaping as live. + +// A crash after `ClientToolResult` but before `ToolFinished`: the reported +// mutation must be wrapped, stored and ledgered on replay, not finished with +// the client's raw text. +#[tokio::test] +async fn a_restart_after_client_tool_result_wraps_stores_and_ledgers_the_reported_mutation() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.click"), + json!({"selector": "#buy"}), + alice(), + ); + log_up_to_model_step(&log, &call); + approved(&log, &call); + log.host_append(requested(&call, now_ms() + 60_000)); + // The client answered, but the process crashed before `ToolFinished`. + log.host_append(Event::ClientToolResult { + call: call.id.clone(), + principal: alice(), + outcome: Outcome::Succeeded, + output: "clicked".into(), + }); + + let effects = FakeEffects::default(); + let model = FakeModel::new(vec![vec![text("done")]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.click", false)]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!( + tools.wrapped_calls(), + vec![call.id.clone()], + "the replayed result must be routed through the host's wrap exactly once" + ); + let (outcome, output) = tool_finished(&log, &call).expect("finished"); + assert_eq!(outcome, Outcome::Succeeded); + assert_eq!(output, Output::Text("wrapped:clicked".into())); + assert_eq!( + effects.recorded(&call.id), + Some(Some(ToolResult::text("wrapped:clicked"))), + "the wrapped result, not the client's raw text, must reach the ledger" + ); + assert_eq!(log.rehydrate(), ctx); +} + +// A read-only client tool's replayed result is wrapped too, but never +// touches the effect ledger -- reads carry no ledger entry anywhere else in +// this engine either. +#[tokio::test] +async fn a_restart_after_client_tool_result_wraps_a_reported_read_without_ledgering_it() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.read_tab"), + json!({}), + alice(), + ); + log_up_to_model_step(&log, &call); + log.host_append(requested(&call, now_ms() + 60_000)); + log.host_append(Event::ClientToolResult { + call: call.id.clone(), + principal: alice(), + outcome: Outcome::Succeeded, + output: "Acme pricing".into(), + }); + + let effects = FakeEffects::default(); + let model = FakeModel::new(vec![vec![text("done")]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.read_tab", true)]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!(tools.wrapped_calls(), vec![call.id.clone()]); + let (_, output) = tool_finished(&log, &call).expect("finished"); + assert_eq!(output, Output::Text("wrapped:Acme pricing".into())); + assert_eq!( + effects.recorded(&call.id), + None, + "a read never touches the ledger" + ); + assert_eq!(log.rehydrate(), ctx); +} + +// A still-earlier crash already wrapped, stored and ledgered this exact +// call (it just never got to append `ToolFinished`): this restart must +// adopt that recorded result rather than ask the host to wrap it again. +#[tokio::test] +async fn a_second_restart_adopts_the_ledgered_wrap_instead_of_wrapping_twice() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.click"), + json!({"selector": "#buy"}), + alice(), + ); + log_up_to_model_step(&log, &call); + approved(&log, &call); + log.host_append(requested(&call, now_ms() + 60_000)); + log.host_append(Event::ClientToolResult { + call: call.id.clone(), + principal: alice(), + outcome: Outcome::Succeeded, + output: "clicked".into(), + }); + + let effects = FakeEffects::default().seed( + call.id.clone(), + Some(ToolResult::text("previously recorded wrap")), + ); + let model = FakeModel::new(vec![vec![text("done")]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.click", false)]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert!( + tools.wrapped_calls().is_empty(), + "an existing ledger claim must be adopted, not wrapped a second time" + ); + let (_, output) = tool_finished(&log, &call).expect("finished"); + assert_eq!(output, Output::Text("previously recorded wrap".into())); +} + +// ---------------------------------------------------------------- P2: +// interrupt must still be able to end a client-tool wait. + +// A turn parked on `Exit::AwaitingClientTool` must still end at the next +// `Interrupt`, exactly like every other parked wait. +#[tokio::test] +async fn interrupting_a_turn_parked_on_a_client_tool_ends_it_promptly() { + let log = FakeLog::default(); + let call = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("browser.read_tab"), + json!({}), + alice(), + ); + log_up_to_model_step(&log, &call); + log.host_append(requested(&call, now_ms() + 60_000)); + log.host_append(Event::Interrupt { principal: alice() }); + + let model = FakeModel::new(vec![]); + let tools = FakeTools::new(vec![client_executed_tool("browser.read_tab", true)]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Interrupted) + ); + assert_eq!(log.rehydrate(), ctx); +} diff --git a/vendor/dex-loop/tests/scenarios.rs b/vendor/dex-loop/tests/scenarios.rs new file mode 100644 index 000000000..7ca1e6c01 --- /dev/null +++ b/vendor/dex-loop/tests/scenarios.rs @@ -0,0 +1,1821 @@ +//! End-to-end scenarios for the loop. Each asserts the exact event sequence +//! the engine appends, and most check that `rehydrate` rebuilds the same +//! context the warm engine holds. + +mod support; + +use std::time::{Duration, Instant}; + +use dex_loop::{ + ApprovalId, ApprovalMode, Budget, CancellationToken, Engine, Event, Exit, Lexicon, ModelError, + OutputRef, ProposedCall, Threshold, ToolName, ToolResult, TurnId, Verdict, +}; +use serde_json::json; +use support::*; + +fn budget() -> Budget { + Budget { + max_steps: 10, + max_tokens: 1_000_000, + max_cost_micros: 1_000_000, + wall: Duration::from_secs(30), + } +} + +fn assert_send(_: &T) {} + +// 1. A plain answer: text, then ModelStepCompleted and Final. +#[tokio::test] +async fn plain_answer_streams_text_then_final() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![text("Hel"), text("lo"), usage(3, 2, 7)]]); + let tools = FakeTools::new(vec![]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "hi"); + let cancel = CancellationToken::new(); + + let run = engine.run(&mut ctx, &cancel); + assert_send(&run); + assert_eq!(run.await, Ok(Exit::Done)); + + assert_eq!( + log.shapes(), + strings(&[ + "user:hi", + "step:1", + "delta:Hello", + "usage:5", + "completed:Hello:[]", + "final:Hello", + ]) + ); + // Two model chunks, one coalesced row: the engine does not assume a row + // per delta. + assert_eq!(log.text_writes(), strings(&["Hel", "lo"])); + assert_eq!(ctx.usage().cost_micros, 7); + assert_eq!( + log.rehydrate(), + ctx, + "rehydrate must rebuild the warm context" + ); + // Running a finished turn again appends nothing. + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + assert_eq!(log.len(), 6); +} + +// 2. Three reads run as one wave: they overlap, take about one tool delay, and +// their results enter history in call order although they finish in reverse. +#[tokio::test] +async fn read_only_wave_overlaps_and_keeps_call_order() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![ + call("search", json!({"key": "a"})), + call("search", json!({"key": "b"})), + call("search", json!({"key": "c"})), + ], + vec![text("done")], + ]); + let tools = FakeTools::new(vec![read_tool("search")]) + .barrier(&["a", "b", "c"]) + .delay("a", Duration::from_millis(300)) + .delay("b", Duration::from_millis(200)) + .delay("c", Duration::from_millis(100)); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "look"); + + let started = Instant::now(); + let exit = engine.run(&mut ctx, &CancellationToken::new()).await; + let elapsed = started.elapsed(); + + assert_eq!(exit, Ok(Exit::Done)); + assert!( + elapsed < Duration::from_millis(500), + "wave took {elapsed:?}; serial dispatch takes 600ms" + ); + assert!(tools.runs().iter().all(|run| !run.cancelled)); + assert_eq!( + log.shapes(), + strings(&[ + "user:look", + "step:1", + "completed::[t1-1-0,t1-1-1,t1-1-2]", + "started:t1-1-0", + "started:t1-1-1", + "started:t1-1-2", + "finished:t1-1-2:ok", + "finished:t1-1-1:ok", + "finished:t1-1-0:ok", + "step:2", + "delta:done", + "completed:done:[]", + "final:done", + ]) + ); + assert_eq!( + view(&model.seen()[1]), + strings(&[ + "user:look", + "assistant::[t1-1-0,t1-1-1,t1-1-2]", + "tool:t1-1-0:ok:out/t1-1-0", + "tool:t1-1-1:ok:out/t1-1-1", + "tool:t1-1-2:ok:out/t1-1-2", + ]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 3. A mutation waits for the wave before it; a read after it starts a new +// wave. The mutation is claimed in the ledger and its outcome recorded. +#[tokio::test] +async fn mutating_call_runs_serially_after_the_wave() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![ + call("search", json!({"key": "a"})), + call("search", json!({"key": "b"})), + call("update", json!({"key": "w"})), + call("search", json!({"key": "c"})), + ], + vec![text("ok")], + ]); + let tools = FakeTools::new(vec![read_tool("search"), write_tool("update")]) + .barrier(&["a", "b"]) + .delay("a", Duration::from_millis(50)) + .delay("b", Duration::from_millis(80)); + let effects = FakeEffects::default(); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.start_turn("t1", "change it"); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!( + log.shapes(), + strings(&[ + "user:change it", + "step:1", + "completed::[t1-1-0,t1-1-1,t1-1-2,t1-1-3]", + "started:t1-1-0", + "started:t1-1-1", + "finished:t1-1-0:ok", + "finished:t1-1-1:ok", + "started:t1-1-2", + "finished:t1-1-2:ok", + "started:t1-1-3", + "finished:t1-1-3:ok", + "step:2", + "delta:ok", + "completed:ok:[]", + "final:ok", + ]) + ); + let write = tools.run_of(&call_id("t1", 1, 2)); + for read in [0, 1] { + let read = tools.run_of(&call_id("t1", 1, read)); + assert!( + write.started >= read.finished, + "the write overlapped a read" + ); + } + assert!(tools.run_of(&call_id("t1", 1, 3)).started >= write.finished); + assert_eq!( + effects.recorded(&call_id("t1", 1, 2)), + Some(Some(output_for(&call_id("t1", 1, 2)))) + ); + assert_eq!( + effects.recorded(&call_id("t1", 1, 0)), + None, + "reads skip the ledger" + ); +} + +// 4a. Park B with C pending, crash, rehydrate: after approval, policy runs +// again, then B and C run with their original arguments, read from the log. +#[tokio::test] +async fn park_crash_rehydrate_runs_the_rest_of_the_step_with_original_args() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![ + call("search", json!({"key": "a"})), + call("send_email", json!({"key": "w", "to": "ops@example.com"})), + call("search", json!({"key": "b"})), + ], + vec![text("sent")], + ]); + let tools = FakeTools::new(vec![read_tool("search"), write_tool("send_email")]) + .verdict("send_email", approval("ap-1")); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "email them"); + let cancel = CancellationToken::new(); + + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(ApprovalId::new("ap-1"))) + ); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0,t1-1-1,t1-1-2]", + "started:t1-1-0", + "finished:t1-1-0:ok", + "approval:t1-1-1", + ]) + ); + assert_eq!(tools.run_ids(), strings(&["t1-1-0"])); + // Parked with no decision: running again changes nothing. + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(ApprovalId::new("ap-1"))) + ); + let parked_len = log.len(); + + // A decision for another approval id authorizes nothing. + log.host_append(Event::ApprovalDecided { + call: call_id("t1", 1, 1), + approval: ApprovalId::new("ap-other"), + args_digest: log.requested_digest(&call_id("t1", 1, 1)), + approved: true, + principal: alice(), + }); + // -- crash: a fresh engine and context from the log -- + let engine = support::engine(&log, &model, &tools, budget()); + let mut ctx = log.rehydrate(); + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(ApprovalId::new("ap-1"))) + ); + assert_eq!(log.len(), parked_len + 1); + + log.decide(&call_id("t1", 1, 1), "ap-1", true); + let mut ctx = log.rehydrate(); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + + assert_eq!( + log.shapes_after(parked_len + 2), + strings(&[ + "started:t1-1-1", + "finished:t1-1-1:ok", + "started:t1-1-2", + "finished:t1-1-2:ok", + "step:2", + "delta:sent", + "completed:sent:[]", + "final:sent", + ]) + ); + assert_eq!(tools.run_ids(), strings(&["t1-1-0", "t1-1-1", "t1-1-2"])); + assert_eq!( + tools.run_of(&call_id("t1", 1, 1)).args, + json!({"key": "w", "to": "ops@example.com"}) + ); + assert_eq!(tools.run_of(&call_id("t1", 1, 2)).args, json!({"key": "b"})); + // Policy ran for B at proposal and again on resume, and once for C. + let checks: Vec = tools + .policy_checks() + .into_iter() + .map(|(call, _)| call) + .collect(); + assert_eq!(checks, strings(&["t1-1-0", "t1-1-1", "t1-1-1", "t1-1-2"])); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&[ + "tool:t1-1-0:ok:out/t1-1-0", + "tool:t1-1-1:ok:out/t1-1-1", + "tool:t1-1-2:ok:out/t1-1-2", + ]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 4b. A declined approval becomes a result the model sees, and the rest of the +// step still runs. This variant resumes on the warm context. +#[tokio::test] +async fn declined_approval_is_a_visible_result_and_the_step_continues() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![ + call("send_email", json!({"key": "w"})), + call("search", json!({"key": "b"})), + ], + vec![text("not sent")], + ]); + let tools = FakeTools::new(vec![read_tool("search"), write_tool("send_email")]) + .verdict("send_email", approval("ap-1")); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "email them"); + let cancel = CancellationToken::new(); + + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(ApprovalId::new("ap-1"))) + ); + let parked_len = log.len(); + log.decide(&call_id("t1", 1, 0), "ap-1", false); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + + assert_eq!( + log.shapes_after(parked_len + 1), + strings(&[ + "finished:t1-1-0:err", + "started:t1-1-1", + "finished:t1-1-1:ok", + "step:2", + "delta:not sent", + "completed:not sent:[]", + "final:not sent", + ]) + ); + assert_eq!(tools.run_ids(), strings(&["t1-1-1"])); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&[ + "tool:t1-1-0:err:denied: the approver declined this call", + "tool:t1-1-1:ok:out/t1-1-1", + ]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 4c. Approval is necessary, not sufficient: a grant revoked while the call +// was parked denies the approved call on resume. +#[tokio::test] +async fn revoked_grant_denies_an_approved_call_on_resume() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("send_email", json!({"key": "w"}))], + vec![text("could not send")], + ]); + let tools = + FakeTools::new(vec![write_tool("send_email")]).verdict("send_email", approval("ap-1")); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "email them"); + let cancel = CancellationToken::new(); + assert!(matches!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(_)) + )); + + log.decide(&call_id("t1", 1, 0), "ap-1", true); + tools.set_verdict( + "send_email", + Verdict::Deny("the mail grant was revoked".into()), + ); + let mut ctx = log.rehydrate(); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + + assert!( + tools.runs().is_empty(), + "a revoked grant still ran the call" + ); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&["tool:t1-1-0:err:denied: the mail grant was revoked"]) + ); +} + +// 4d. An approval must match the proposed call's argument digest. +#[tokio::test] +async fn approval_decision_with_a_different_digest_is_denied() { + for approved in [true, false] { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("send_email", json!({"key": "w"}))], + vec![text("no")], + ]); + let tools = + FakeTools::new(vec![write_tool("send_email")]).verdict("send_email", approval("ap-1")); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "email them"); + let cancel = CancellationToken::new(); + assert!(matches!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(_)) + )); + + log.host_append(Event::ApprovalDecided { + call: call_id("t1", 1, 0), + approval: ApprovalId::new("ap-1"), + args_digest: dex_loop::args_digest(&json!({"key": "other"})), + approved, + principal: alice(), + }); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + assert!(tools.runs().is_empty()); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&["tool:t1-1-0:err:denied: the approval does not match this call's arguments"]) + ); + } +} + +// 5. Bob steers in Alice's turn: the steer becomes Bob's user message before +// the next model call, and the calls that step proposes are checked under Bob. +#[tokio::test] +async fn steer_from_another_principal_is_checked_under_that_principal() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("search", json!({"key": "a"}))], + vec![call("update", json!({"key": "w"}))], + vec![text("Bob cannot update")], + ]); + let steer_log = log.clone(); + let tools = FakeTools::new(vec![read_tool("search"), write_tool("update")]) + .verdict_for("update", bob(), Verdict::Deny("bob lacks access".into())) + .on_run(move |call| { + if call.tool.as_str() == "search" { + steer_log.host_append(Event::Steer { + principal: bob(), + text: "also update staging".into(), + }); + } + }); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "check prod"); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!( + log.shapes(), + strings(&[ + "user:check prod", + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "steer:also update staging", + "finished:t1-1-0:ok", + "step:2", + "completed::[t1-2-0]", + "finished:t1-2-0:err", + "step:3", + "delta:Bob cannot update", + "completed:Bob cannot update:[]", + "final:Bob cannot update", + ]) + ); + assert_eq!( + view(&model.seen()[1]), + strings(&[ + "user:check prod", + "assistant::[t1-1-0]", + "tool:t1-1-0:ok:out/t1-1-0", + "user:also update staging", + ]) + ); + assert_eq!( + tools.policy_checks(), + vec![ + ("t1-1-0".to_owned(), "alice".to_owned()), + ("t1-2-0".to_owned(), "bob".to_owned()), + ] + ); + assert!(tools.run_ids().iter().all(|id| id != "t1-2-0")); + assert!(log.events().iter().any(|event| matches!( + event, + Event::ToolStarted { principal, .. } if principal == &alice() + ))); + assert_eq!(log.rehydrate(), ctx); +} + +// 5b. A steer that lands while the model is answering continues the turn +// instead of being dropped. +#[tokio::test] +async fn steer_during_a_final_answer_continues_the_turn() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![text("fi"), text("rst")], vec![text("second")]]) + .with_chunk_delay(Duration::from_millis(100)); + let tools = FakeTools::new(vec![]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "q"); + let cancel = CancellationToken::new(); + + let host = async { + tokio::time::sleep(Duration::from_millis(30)).await; + log.host_append(Event::Steer { + principal: alice(), + text: "shorter please".into(), + }); + }; + let (exit, ()) = tokio::join!(engine.run(&mut ctx, &cancel), host); + + assert_eq!(exit, Ok(Exit::Done)); + assert_eq!( + log.shapes(), + strings(&[ + "user:q", + "step:1", + "steer:shorter please", + "delta:first", + "completed:first:[]", + "step:2", + "delta:second", + "completed:second:[]", + "final:second", + ]) + ); + assert_eq!( + view(&model.seen()[1]), + strings(&["user:q", "assistant:first:[]", "user:shorter please"]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 6. Interrupt during a wave cancels the running reads, closes the calls that +// never started, and ends the turn. +#[tokio::test] +async fn interrupt_mid_wave_cancels_reads_and_emits_interrupted() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![ + call("search", json!({"key": "a"})), + call("search", json!({"key": "b"})), + call("update", json!({"key": "w"})), + ]]); + let tools = FakeTools::new(vec![read_tool("search"), write_tool("update")]) + .delay("a", Duration::from_secs(10)) + .delay("b", Duration::from_secs(10)); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "go"); + let cancel = CancellationToken::new(); + + let host = async { + tokio::time::sleep(Duration::from_millis(100)).await; + log.host_append(Event::Interrupt { principal: alice() }); + cancel.cancel(); + }; + let started = Instant::now(); + let (exit, ()) = tokio::join!(engine.run(&mut ctx, &cancel), host); + + assert_eq!(exit, Ok(Exit::Interrupted)); + assert!(started.elapsed() < Duration::from_secs(2)); + assert!(tools.runs().iter().all(|run| run.cancelled)); + assert_eq!(tools.run_ids().len(), 2, "the write never started"); + let shapes = log.shapes(); + assert_eq!( + shapes[..6], + strings(&[ + "user:go", + "step:1", + "completed::[t1-1-0,t1-1-1,t1-1-2]", + "started:t1-1-0", + "started:t1-1-1", + "interrupt", + ]) + ); + // The two reads finish in either order once cancelled. + let mut reads = shapes[6..8].to_vec(); + reads.sort(); + assert_eq!( + reads, + strings(&["finished:t1-1-0:err", "finished:t1-1-1:err"]) + ); + assert_eq!( + shapes[8..], + strings(&["finished:t1-1-2:err", "interrupted"]) + ); + assert_eq!(history(&log.rehydrate()), history(&ctx)); + // An interrupted turn stays interrupted. + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Interrupted) + ); +} + +// 6b. Interrupt during a mutation lets that mutation complete, then stops +// before the next effect. +#[tokio::test] +async fn interrupt_during_a_mutation_completes_it_then_stops() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![ + call("update", json!({"key": "w"})), + call("update", json!({"key": "x"})), + ]]); + let tools = FakeTools::new(vec![write_tool("update")]).delay("w", Duration::from_millis(300)); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "go"); + let cancel = CancellationToken::new(); + + let host = async { + tokio::time::sleep(Duration::from_millis(50)).await; + log.host_append(Event::Interrupt { principal: alice() }); + cancel.cancel(); + }; + let (exit, ()) = tokio::join!(engine.run(&mut ctx, &cancel), host); + + assert_eq!(exit, Ok(Exit::Interrupted)); + assert_eq!(tools.run_ids(), strings(&["t1-1-0"])); + assert!( + !tools.runs()[0].cancelled, + "a started mutation was cancelled" + ); + assert_eq!( + log.shapes(), + strings(&[ + "user:go", + "step:1", + "completed::[t1-1-0,t1-1-1]", + "started:t1-1-0", + "interrupt", + "finished:t1-1-0:ok", + "finished:t1-1-1:err", + "interrupted", + ]) + ); +} + +// 6c. An Interrupt already in the log ends a parked turn without a token. +#[tokio::test] +async fn interrupt_event_ends_a_parked_turn() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![call("send_email", json!({}))]]); + let tools = + FakeTools::new(vec![write_tool("send_email")]).verdict("send_email", approval("ap-1")); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "go"); + let cancel = CancellationToken::new(); + assert!(matches!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(_)) + )); + log.host_append(Event::Interrupt { principal: bob() }); + let mut ctx = log.rehydrate(); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Interrupted)); + assert_eq!( + log.shapes_after(3), + strings(&[ + "approval:t1-1-0", + "interrupt", + "finished:t1-1-0:err", + "interrupted", + ]) + ); + assert!(tools.runs().is_empty()); +} + +// 7. Each budget axis stops the turn with budget_exhausted. +async fn run_to_budget(budget: Budget, model: FakeModel) -> (FakeLog, FakeModel, Exit) { + let log = FakeLog::default(); + let tools = FakeTools::new(vec![read_tool("search")]); + let engine = engine(&log, &model, &tools, budget); + let mut ctx = log.start_turn("t1", "go"); + let exit = engine + .run(&mut ctx, &CancellationToken::new()) + .await + .expect("not fenced"); + assert_eq!(log.rehydrate(), ctx); + (log, model, exit) +} + +#[tokio::test] +async fn step_cap_gets_an_answer_only_step_that_ends_in_final() { + let (log, model, exit) = run_to_budget( + Budget { + max_steps: 1, + ..budget() + }, + FakeModel::new(vec![ + vec![call("search", json!({}))], + vec![text("here is what I found")], + ]), + ) + .await; + assert_eq!(exit, Exit::Done); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "finished:t1-1-0:ok", + "step:2", + "delta:here is what I found", + "completed:here is what I found:[]", + "final:here is what I found", + ]) + ); + assert_eq!(model.calls(), 2); + let offered = model.offered(); + assert!(!offered[0].is_empty(), "step 1 offers tools"); + assert!(offered[1].is_empty(), "the answer-only step offers none"); +} + +#[tokio::test] +async fn tool_call_on_the_answer_only_step_ends_budget_exhausted() { + let (log, model, exit) = run_to_budget( + Budget { + max_steps: 1, + ..budget() + }, + FakeModel::new(vec![ + vec![call("search", json!({}))], + vec![call("search", json!({}))], + ]), + ) + .await; + assert_eq!(exit, Exit::Failed); + assert_eq!( + log.shapes().last().cloned(), + Some( + "error:budget_exhausted:step budget exhausted: 1 steps and the answer-only step after them" + .to_owned() + ) + ); + assert_eq!(model.calls(), 2); + assert_eq!( + log.shapes() + .iter() + .filter(|s| s.starts_with("started:")) + .count(), + 1, + "the refused call never ran" + ); +} + +#[tokio::test] +async fn token_budget_exhaustion() { + let (log, model, exit) = run_to_budget( + Budget { + max_tokens: 100, + ..budget() + }, + FakeModel::new(vec![vec![usage(90, 20, 0), call("search", json!({}))]]), + ) + .await; + assert_eq!(exit, Exit::Failed); + assert_eq!( + log.shapes().last().cloned(), + Some("error:budget_exhausted:token budget exhausted: 110 of 100 tokens".to_owned()) + ); + assert_eq!(model.calls(), 1); +} + +#[tokio::test] +async fn cost_budget_exhaustion() { + let (log, model, exit) = run_to_budget( + Budget { + max_cost_micros: 500, + ..budget() + }, + FakeModel::new(vec![vec![usage(1, 1, 500), call("search", json!({}))]]), + ) + .await; + assert_eq!(exit, Exit::Failed); + assert_eq!( + log.shapes().last().cloned(), + Some("error:budget_exhausted:cost budget exhausted: 500 of 500 micros".to_owned()) + ); + assert_eq!(model.calls(), 1); +} + +#[tokio::test] +async fn wall_budget_exhaustion() { + let (log, model, exit) = run_to_budget( + Budget { + wall: Duration::from_millis(100), + ..budget() + }, + FakeModel::new(vec![vec![text("thinking"), call("search", json!({}))]]) + .with_chunk_delay(Duration::from_millis(60)), + ) + .await; + assert_eq!(exit, Exit::Failed); + assert_eq!( + log.shapes().last().cloned(), + Some("error:budget_exhausted:wall budget exhausted: 100ms".to_owned()) + ); + assert_eq!(model.calls(), 1); +} + +// 8a. A mutation crashed after its effect and the ledger cannot reconcile it: +// the outcome is Unknown, the model sees it, and the call is not dispatched +// again. +#[tokio::test] +async fn mutation_crashed_after_effect_is_unknown_and_not_redispatched() { + let log = FakeLog::default(); + let write = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("update"), + json!({"key": "w"}), + alice(), + ); + crashed_after_start(&log, &write); + let effects = FakeEffects::default().seed( + write.id.clone(), + Some(ToolResult::unknown( + "outcome unknown: the executor lost the call", + )), + ); + let model = FakeModel::new(vec![vec![text("not sure it applied")]]); + let tools = FakeTools::new(vec![write_tool("update")]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert!( + tools.runs().is_empty(), + "an unknown mutation was dispatched again" + ); + assert!(tools.policy_checks().is_empty()); + assert_eq!( + log.shapes_after(4), + strings(&[ + "finished:t1-1-0:unknown", + "step:2", + "delta:not sure it applied", + "completed:not sure it applied:[]", + "final:not sure it applied", + ]) + ); + assert_eq!( + view(&model.seen()[0])[2..], + strings(&["tool:t1-1-0:unknown:outcome unknown: the executor lost the call"]) + ); +} + +// 8b. A recorded outcome is adopted; a claim with no recorded outcome +// settles to unknown -- never `Running`, which nothing would ever update -- +// and the ledger is updated to match, so a later resume gets the same +// answer. Neither dispatches again. +#[tokio::test] +async fn crashed_mutation_adopts_the_ledger_outcome() { + for (recorded, expected_shape, expected_ledger_outcome) in [ + ( + Some(ToolResult::stored(OutputRef::new("ledger/w"), None)), + "finished:t1-1-0:ok", + dex_loop::Outcome::Succeeded, + ), + (None, "finished:t1-1-0:unknown", dex_loop::Outcome::Unknown), + ] { + let log = FakeLog::default(); + let write = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("update"), + json!({}), + alice(), + ); + crashed_after_start(&log, &write); + let effects = FakeEffects::default().seed(write.id.clone(), recorded); + let model = FakeModel::new(vec![vec![text("done")]]); + let tools = FakeTools::new(vec![write_tool("update")]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert!(tools.runs().is_empty()); + assert_eq!(log.shapes()[4], expected_shape); + assert_eq!( + effects + .recorded(&write.id) + .flatten() + .expect("the ledger must hold a settled outcome") + .outcome, + expected_ledger_outcome + ); + } +} + +// 8c. A read that started before a crash simply runs again. +#[tokio::test] +async fn crashed_read_runs_again() { + let log = FakeLog::default(); + let read = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("search"), + json!({"key": "a"}), + alice(), + ); + crashed_after_start(&log, &read); + let model = FakeModel::new(vec![vec![text("found")]]); + let tools = FakeTools::new(vec![read_tool("search")]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.rehydrate(); + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!(tools.run_ids(), strings(&["t1-1-0"])); + assert_eq!( + log.shapes()[4..6], + strings(&["started:t1-1-0", "finished:t1-1-0:ok"]) + ); +} + +// 8d. Crash mid-stream: the attempt is marked abandoned, its text leaves the +// model's context, and the model call is issued again. +#[tokio::test] +async fn crash_mid_stream_abandons_the_attempt_and_reissues_it() { + let log = FakeLog::default(); + log.host_append(Event::UserMessage { + turn: dex_loop::TurnId::new("t1"), + message_id: None, + principal: alice(), + text: "hi".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + log.host_append(Event::StepStarted { + step: 1, + control_through: dex_loop::Cursor::START, + }); + log.host_append(Event::TextDelta { + text: "partial ans".into(), + }); + let model = FakeModel::new(vec![vec![text("fresh")]]); + let tools = FakeTools::new(vec![]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!(view(&model.seen()[0]), strings(&["user:hi"])); + assert_eq!( + log.shapes_after(3), + strings(&[ + "abandoned:1", + "step:2", + "delta:fresh", + "completed:fresh:[]", + "final:fresh", + ]) + ); + assert_eq!( + history(&log.rehydrate()), + strings(&["user:hi", "assistant:fresh:[]"]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 8d2. Usage is batched with whichever terminal event ends the attempt, not +// appended on its own as it streams in. A model attempt that reports usage +// and then fails must still land that usage in the log, alongside +// `ModelAttemptAbandoned`, so budgets never under-count a failed attempt. +#[tokio::test] +async fn usage_reported_before_a_failed_attempt_still_lands_with_the_abandon() { + let (log, model, exit) = run_to_budget( + budget(), + FakeModel::new(vec![vec![ + usage(10, 5, 100), + Err(ModelError { + message: "boom".into(), + }), + ]]), + ) + .await; + assert_eq!(exit, Exit::Failed); + assert_eq!( + log.shapes_after(2), + strings(&["usage:15", "abandoned:1", "error:model_failed:boom",]) + ); + assert_eq!(model.calls(), 1); +} + +// 8e. A `Started` call whose tool vanished from the offered catalog (deploy, +// grant revoke) resolves through the ledger, never as "unknown tool": the +// model must not be told to retry a call that may have already run under +// a different tool identity. +#[tokio::test] +async fn started_call_with_vanished_tool_settles_unknown_and_is_not_redispatched() { + let log = FakeLog::default(); + let write = ProposedCall::new( + call_id("t1", 1, 0), + ToolName::new("deploy"), + json!({}), + alice(), + ); + crashed_after_start(&log, &write); + let effects = FakeEffects::default(); + let model = FakeModel::new(vec![vec![text("ok")]]); + // The tool is no longer offered: catalog is empty, as after a grant + // revoke or a deploy that dropped it. + let tools = FakeTools::new(vec![]); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = log.rehydrate(); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert!(tools.runs().is_empty(), "a vanished tool was dispatched"); + assert_eq!(log.shapes()[4], "finished:t1-1-0:unknown"); + let recorded = effects + .recorded(&write.id) + .expect("a vanished-tool call must claim itself in the ledger") + .expect("the ledger must record the settled outcome"); + assert_eq!(recorded.outcome, dex_loop::Outcome::Unknown); + assert_eq!( + view(&model.seen()[0])[2], + "tool:t1-1-0:unknown:outcome unknown: this call already started; \ + do not retry it without first checking whether it took effect" + ); +} + +// 8f. `CallId` is unique only within its thread (a turn id is +// caller-chosen); two threads that pick the same turn id get the same raw +// call id string. `Tools::run` still receives the call's thread separately, +// so a downstream key built from both stays globally unique -- the contract +// `ports.rs` documents and the dex-tools idempotency-key fix relies on. +#[tokio::test] +async fn same_turn_id_in_two_threads_dispatches_under_distinct_threads() { + async fn run_one(thread: dex_loop::ThreadId) -> FakeTools { + let log = FakeLog::default(); + log.host_append(Event::UserMessage { + turn: dex_loop::TurnId::new("t1"), + message_id: None, + principal: alice(), + text: "go".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + let model = FakeModel::new(vec![vec![call("update", json!({}))], vec![text("done")]]); + let tools = FakeTools::new(vec![write_tool("update")]); + let effects = FakeEffects::default(); + let engine = engine_with(&log, &model, &tools, &effects, budget()); + let mut ctx = dex_loop::rehydrate(thread, &log.entries()); + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + tools + } + + let thread_a = dex_loop::ThreadId { + org: "org-1".into(), + workspace: "ws-1".into(), + thread: "thread-a".into(), + }; + let thread_b = dex_loop::ThreadId { + org: "org-2".into(), + workspace: "ws-1".into(), + thread: "thread-b".into(), + }; + let tools_a = run_one(thread_a.clone()).await; + let tools_b = run_one(thread_b.clone()).await; + + // The raw CallId string collides: same turn, step and index in both + // threads. + assert_eq!(tools_a.run_ids(), strings(&["t1-1-0"])); + assert_eq!(tools_b.run_ids(), strings(&["t1-1-0"])); + let run_a = tools_a.run_of(&call_id("t1", 1, 0)); + let run_b = tools_b.run_of(&call_id("t1", 1, 0)); + assert_eq!(run_a.thread, thread_a); + assert_eq!(run_b.thread, thread_b); + assert_ne!( + run_a.thread, run_b.thread, + "the same CallId was dispatched under different threads" + ); +} + +// 9. A fenced write stops the engine: no further writes and no further model +// calls. +#[tokio::test] +async fn fenced_write_stops_immediately() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![text("a"), text("b"), text("c"), call("search", json!({}))], + vec![text("never")], + ]); + let tools = FakeTools::new(vec![read_tool("search")]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "go"); + // StepStarted and the first text land; the second text is refused. + log.fence_after(2); + + let result = engine.run(&mut ctx, &CancellationToken::new()).await; + + assert!(matches!(result, Err(dex_loop::Fenced { .. }))); + assert_eq!(log.refused(), 1, "the engine kept writing after the fence"); + assert_eq!(log.shapes(), strings(&["user:go", "step:1", "delta:a"])); + assert_eq!(model.calls(), 1); + assert!(tools.runs().is_empty()); +} + +#[tokio::test] +async fn fenced_during_a_wave_drops_the_remaining_results() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![ + call("search", json!({"key": "a"})), + call("search", json!({"key": "b"})), + ]]); + let tools = FakeTools::new(vec![read_tool("search")]).delay("b", Duration::from_millis(50)); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "go"); + // StepStarted, ModelStepCompleted and the ToolStarted batch land; the + // first ToolFinished is refused. + log.fence_after(3); + + let result = engine.run(&mut ctx, &CancellationToken::new()).await; + + assert!(result.is_err()); + assert_eq!(log.refused(), 1); + assert_eq!( + log.shapes(), + strings(&[ + "user:go", + "step:1", + "completed::[t1-1-0,t1-1-1]", + "started:t1-1-0", + "started:t1-1-1", + ]) + ); +} + +// 10. Unknown, unexposed and denied calls produce results the model sees; +// nothing runs for them. +#[tokio::test] +async fn unknown_and_denied_calls_are_visible_to_the_model() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![ + call("no_such_tool", json!({})), + call("delete_all", json!({})), + call("crm.lookup", json!({})), + call("search", json!({"key": "a"})), + ], + vec![text("sorry")], + ]); + let tools = FakeTools::new(vec![ + read_tool("search"), + write_tool("delete_all"), + hidden_read_tool("crm.lookup"), + ]) + .verdict( + "delete_all", + Verdict::Deny("destructive calls are off".into()), + ); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "clean up"); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!(tools.run_ids(), strings(&["t1-1-3"])); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0,t1-1-1,t1-1-2,t1-1-3]", + "finished:t1-1-0:err", + "finished:t1-1-1:err", + "finished:t1-1-2:err", + "started:t1-1-3", + "finished:t1-1-3:ok", + "step:2", + "delta:sorry", + "completed:sorry:[]", + "final:sorry", + ]) + ); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&[ + "tool:t1-1-0:err:unknown tool: no_such_tool; use tools.search to find tools", + "tool:t1-1-1:err:denied: destructive calls are off", + "tool:t1-1-2:err:unknown tool: crm.lookup; use tools.search to find tools", + "tool:t1-1-3:ok:out/t1-1-3", + ]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 10b. Headless turns: no human can answer, so a `NeedsApproval` verdict is +// approved by policy. The log keeps the request and the decision, attributed +// to the policy principal; a `Deny` verdict stays denied; an interactive turn +// still parks. +fn headless_tools() -> FakeTools { + FakeTools::new(vec![write_tool("send_email"), write_tool("delete_all")]) + .verdict("send_email", approval("ap-1")) + .verdict( + "delete_all", + Verdict::Deny("destructive calls are off".into()), + ) +} + +#[tokio::test] +async fn headless_turn_auto_approves_an_ask_gated_tool_and_records_the_audit_pair() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("send_email", json!({"key": "w"}))], + vec![text("sent")], + ]); + let tools = headless_tools(); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn_with_approval_mode("t1", "email them", ApprovalMode::Headless); + assert_eq!(ctx.approval_mode(), ApprovalMode::Headless); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!(tools.run_ids(), strings(&["t1-1-0"])); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "approval:t1-1-0", + "decided:t1-1-0:true", + "started:t1-1-0", + "finished:t1-1-0:ok", + "step:2", + "delta:sent", + "completed:sent:[]", + "final:sent", + ]) + ); + let call = call_id("t1", 1, 0); + let events = log.events(); + let requested = events + .iter() + .find_map(|event| match event { + Event::ApprovalRequested { + approval, + args_digest, + .. + } => Some((approval.clone(), args_digest.clone())), + _ => None, + }) + .expect("approval_requested is recorded"); + assert!(events.iter().any(|event| matches!( + event, + Event::ApprovalDecided { + call: decided_call, + approval, + args_digest, + approved: true, + principal, + } if *decided_call == call + && *approval == requested.0 + && *args_digest == requested.1 + && principal.as_str() == dex_loop::HEADLESS_AUTO_APPROVER + ))); + // Resumes keep the mode: it is on the logged UserMessage. + let rehydrated = log.rehydrate(); + assert_eq!(rehydrated.approval_mode(), ApprovalMode::Headless); + assert_eq!(rehydrated, ctx); +} + +#[tokio::test] +async fn headless_turn_keeps_a_hard_deny_denied() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("delete_all", json!({}))], + vec![text("refused")], + ]); + let tools = headless_tools(); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn_with_approval_mode("t1", "clean up", ApprovalMode::Headless); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert!(tools.run_ids().is_empty()); + assert!( + !log.events().iter().any(|event| matches!( + event, + Event::ApprovalRequested { .. } | Event::ApprovalDecided { .. } + )), + "a denied call is never offered for approval" + ); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&["tool:t1-1-0:err:denied: destructive calls are off"]) + ); +} + +#[tokio::test] +async fn interactive_turn_still_parks_on_the_same_ask_gated_tool() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![call("send_email", json!({"key": "w"}))]]); + let tools = headless_tools(); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn_with_approval_mode("t1", "email them", ApprovalMode::Interactive); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Parked(ApprovalId::new("ap-1"))) + ); + assert!(tools.run_ids().is_empty()); + assert_eq!( + log.shapes_after(1), + strings(&["step:1", "completed::[t1-1-0]", "approval:t1-1-0"]) + ); +} + +#[tokio::test] +async fn headless_turn_auto_approves_a_mutating_client_tool() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![call( + "browser.click", + json!({"selector": "#buy"}), + )]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.click", false)]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn_with_approval_mode("t1", "click buy", ApprovalMode::Headless); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::AwaitingClientTool(call_id("t1", 1, 0))) + ); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "approval:t1-1-0", + "decided:t1-1-0:true", + "client_tool:t1-1-0:browser.click", + ]) + ); +} + +// 11. Compaction is an event, applied before the model call, and rehydration +// applies it the same way. +#[tokio::test] +async fn compaction_is_emitted_and_applied() { + let log = FakeLog::default(); + let first = FakeModel::new(vec![vec![text(&"long answer ".repeat(10))]]); + let tools = FakeTools::new(vec![]); + let mut ctx = log.start_turn("t0", "first question"); + assert_eq!( + engine(&log, &first, &tools, budget()) + .run(&mut ctx, &CancellationToken::new()) + .await, + Ok(Exit::Done) + ); + let before = log.len(); + + let model = FakeModel::new(vec![vec![text("short")]]); + let engine = Engine::new( + log.clone(), + model.clone(), + tools.clone(), + FakeEffects::default(), + Lexicon::default(), + budget(), + ) + .with_compactor(Threshold::new(100, 1, FakeSummarizer)); + let mut ctx = log.start_turn("t1", "second question"); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!( + log.shapes_after(before), + strings(&[ + "user:second question", + "compaction:summary of 2 entries", + "step:1", + "delta:short", + "completed:short:[]", + "final:short", + ]) + ); + assert_eq!( + view(&model.seen()[0]), + strings(&["summary:summary of 2 entries", "user:second question"]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// 12. The sanitizer replaces internal identifiers in deltas, including ones +// split across chunks; the stored text equals the streamed text. +#[tokio::test] +async fn sanitizer_replaces_internal_tool_identifiers() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![ + text("Using line"), + text("ar.search_issues from Maes"), + text("tro now"), + ]]); + let tools = FakeTools::new(vec![]); + let engine = Engine::new( + log.clone(), + model.clone(), + tools.clone(), + FakeEffects::default(), + Lexicon::new([ + ("linear.search_issues", "Linear search"), + ("maestro", "Dex"), + ]), + budget(), + ); + let mut ctx = log.start_turn("t1", "status?"); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!( + log.text_writes(), + strings(&["Using ", "Linear search from ", "Dex now"]) + ); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "delta:Using Linear search from Dex now", + "completed:Using Linear search from Dex now:[]", + "final:Using Linear search from Dex now", + ]) + ); + let serialized = serde_json::to_string(&log.events()).expect("serialize events"); + assert!(!serialized.contains("search_issues")); + assert!(!serialized.to_lowercase().contains("maestro")); +} + +// 13. tools.search exposes a non-core tool; its schema is offered from the +// next step, and the exposure is an event. +#[tokio::test] +async fn tools_search_exposes_schemas_for_the_next_step() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("tools.search", json!({"query": "crm"}))], + vec![call("crm.lookup", json!({"key": "acme"}))], + vec![text("found acme")], + ]); + let tools = FakeTools::new(vec![read_tool("search"), hidden_read_tool("crm.lookup")]) + .search_result("crm", &["crm.lookup"]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "find acme in the crm"); + + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done) + ); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "exposed:[crm.lookup]", + "finished:t1-1-0:ok", + "step:2", + "completed::[t1-2-0]", + "started:t1-2-0", + "finished:t1-2-0:ok", + "step:3", + "delta:found acme", + "completed:found acme:[]", + "final:found acme", + ]) + ); + let offered = model.offered(); + assert_eq!(offered[0], strings(&["tools.search", "search"])); + assert_eq!( + offered[1], + strings(&["tools.search", "search", "crm.lookup"]) + ); + assert_eq!(tools.run_ids(), strings(&["t1-2-0"])); + assert_eq!( + view(&model.seen()[1])[2..], + strings(&["tool:t1-1-0:ok:crm.lookup: Label for crm.lookup"]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// A question parks the turn; the answer is the call's result. +#[tokio::test] +async fn question_parks_and_answer_resumes() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("ask_user", json!({"question": "Which region?"}))], + vec![text("using eu")], + ]); + let tools = FakeTools::new(vec![ask_tool("ask_user")]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "deploy"); + let cancel = CancellationToken::new(); + + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Asked(call_id("t1", 1, 0))) + ); + log.host_append(Event::Answer { + call: call_id("t1", 1, 0), + principal: alice(), + text: "eu".into(), + }); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "question:t1-1-0:Which region?", + "answer:t1-1-0:eu", + "finished:t1-1-0:ok", + "step:2", + "delta:using eu", + "completed:using eu:[]", + "final:using eu", + ]) + ); + assert_eq!(view(&model.seen()[1])[2..], strings(&["tool:t1-1-0:ok:eu"])); + assert!(tools.runs().is_empty()); + assert_eq!(log.rehydrate(), ctx); +} + +// `Context` stores a turn's client-declared tools exactly as logged, for a +// host to compose into `Tools::catalog()` on rehydrate. +#[tokio::test] +async fn context_carries_the_declared_client_tools_unmodified() { + let log = FakeLog::default(); + let declared = vec![client_tool("browser.read_tab", true)]; + let ctx = log.start_turn_with_client_tools("t1", "hi", declared.clone()); + assert_eq!(ctx.client_tools(), declared.as_slice()); + assert_eq!(log.rehydrate().client_tools(), declared.as_slice()); +} + +// A read-only client tool parks the turn; SubmitToolResult resumes it. +#[tokio::test] +async fn client_tool_is_requested_and_result_resumes() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("browser.read_tab", json!({}))], + vec![text("Acme's pricing page")], + ]); + let tools = FakeTools::new(vec![client_executed_tool("browser.read_tab", true)]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "what's on my tab?"); + let cancel = CancellationToken::new(); + + let call = call_id("t1", 1, 0); + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::AwaitingClientTool(call.clone())) + ); + log.submit_tool_result(&call, dex_loop::Outcome::Succeeded, "Acme pricing"); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "client_tool:t1-1-0:browser.read_tab", + "client_tool_result:t1-1-0:ok", + "finished:t1-1-0:ok", + "step:2", + "delta:Acme's pricing page", + "completed:Acme's pricing page:[]", + "final:Acme's pricing page", + ]) + ); + // The client's raw text is routed through `Tools::wrap_client_result` + // before the model sees it -- the same shaping a live dispatch of any + // other executor gets -- not appended to history unwrapped. + assert_eq!( + view(&model.seen()[1])[2..], + strings(&["tool:t1-1-0:ok:wrapped:Acme pricing"]) + ); + assert_eq!(tools.wrapped_calls(), vec![call.clone()]); + assert!( + tools.runs().is_empty(), + "the client, not Tools::run, executes it" + ); + assert_eq!(log.rehydrate(), ctx); +} + +// A mutating client tool parks for approval before it is requested; denying +// it never reaches the client. +#[tokio::test] +async fn a_mutating_client_tool_needs_approval_first() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![vec![call( + "browser.click", + json!({"selector": "#buy"}), + )]]); + let tools = FakeTools::new(vec![client_executed_tool("browser.click", false)]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "click buy"); + let cancel = CancellationToken::new(); + + let call = call_id("t1", 1, 0); + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(ApprovalId::new(format!("client-{call}")))) + ); + log.decide(&call, &format!("client-{call}"), false); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Failed)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "approval:t1-1-0", + "decided:t1-1-0:false", + "finished:t1-1-0:err", + "step:2", + "abandoned:2", + "error:model_failed:no script left", + ]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// A model failure abandons the attempt and ends the turn with model_failed. +#[tokio::test] +async fn model_failure_abandons_the_attempt_and_fails() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![]); + let tools = FakeTools::new(vec![]); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "hi"); + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Failed) + ); + assert_eq!( + log.shapes_after(1), + strings(&["step:1", "abandoned:1", "error:model_failed:no script left",]) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// Kernel contract: rehydrating a log suffix must never let a control event +// from before the suffix's start be (re-)applied against whatever turn is +// current once the replay catches up -- neither directly, nor indirectly +// via the engine's own `Log::control_since(ctx.control_cursor())` call +// re-fetching it. + +// 12a. A suffix that excludes an old, already-resolved `Interrupt` (but +// includes its `Interrupted` and everything after) must leave +// `control_cursor()` high enough that `control_since` never re-fetches that +// interrupt -- otherwise it is observed again once the new turn is +// `Running` and kills it before it ever gets to run. +#[tokio::test] +async fn a_stale_interrupt_excluded_from_the_rehydrated_suffix_must_not_kill_the_new_turn() { + let log = FakeLog::default(); + log.host_append(Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: alice(), + text: "long task".into(), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); // cursor 1 + log.host_append(Event::Interrupt { principal: alice() }); // cursor 2 -- excluded from the suffix below + log.host_append(Event::Interrupted); // cursor 3 + log.host_append(Event::UserMessage { + turn: TurnId::new("t2"), + message_id: None, + principal: alice(), + text: "fresh task".into(), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); // cursor 4 + + // Exactly what `read_for_rehydrate` returns when a compaction boundary + // lands between an old turn's `Interrupt` and its `Interrupted`: the + // suffix contains zero control-kind events even though a real one (the + // `Interrupt` at cursor 2) exists earlier in the full log. + let suffix: Vec<_> = log + .entries() + .into_iter() + .filter(|(cursor, _)| cursor.0 >= 3) + .collect(); + assert!( + suffix.iter().all(|(_, event)| !event.is_control()), + "this suffix must contain no control-kind event for the scenario to be meaningful" + ); + let mut ctx = dex_loop::rehydrate(thread(), &suffix); + assert_eq!(ctx.turn(), Some(&TurnId::new("t2"))); + + let model = FakeModel::new(vec![vec![text("hi again")]]); + let tools = FakeTools::new(vec![]); + let engine = engine(&log, &model, &tools, budget()); + assert_eq!( + engine.run(&mut ctx, &CancellationToken::new()).await, + Ok(Exit::Done), + "an Interrupt excluded from the rehydrated suffix must not be re-fetched and \ + applied against the new turn" + ); +} + +// 12b. A `Steer` queued for a turn whose own `UserMessage` is before the +// rehydrate point (excluded from the suffix) must not survive to be flushed +// into a later, unrelated turn's history. +#[tokio::test] +async fn a_steer_from_before_the_rehydrate_point_is_not_carried_into_a_later_turn() { + let log = FakeLog::default(); + log.host_append(Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: alice(), + text: "long task".into(), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); // cursor 1 -- excluded from the suffix below + log.host_append(Event::Steer { + principal: bob(), + text: "focus on the renewal date".into(), + }); // cursor 2 -- never flushed within t1 + log.host_append(Event::Final { + text: "done".into(), + }); // cursor 3 + log.host_append(Event::UserMessage { + turn: TurnId::new("t2"), + message_id: None, + principal: alice(), + text: "a completely different question".into(), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); // cursor 4 + + // Excludes t1's own `UserMessage` (cursor 1) but includes its orphaned + // `Steer` (cursor 2) and everything after -- exactly what a compaction + // boundary landing mid-turn, before the steer is flushed, produces. + let suffix: Vec<_> = log + .entries() + .into_iter() + .filter(|(cursor, _)| cursor.0 >= 2) + .collect(); + let ctx = dex_loop::rehydrate(thread(), &suffix); + assert_eq!( + history(&ctx), + strings(&["user:a completely different question"]), + "bob's steer for a turn this replay never saw start must not appear in t2's history" + ); +} + +// Provider reasoning is committed with its step and survives a park and a +// crash: after a fresh engine rehydrates the log and the approval resumes +// the step, the next model call sees it on the assistant message. +#[tokio::test] +async fn reasoning_is_committed_with_its_step_and_returned_after_park_and_crash() { + let reasoning = dex_loop::ProviderReasoning { + format: "google.gemini.v1".into(), + model: "gemini-3.6-flash".into(), + payload: json!({"calls": [{"extra_content": {"google": {"thought_signature": "c2ln"}}}]}), + }; + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![ + text("On it."), + call("send_email", json!({"key": "w"})), + Ok(dex_loop::ModelChunk::Reasoning(reasoning.clone())), + usage(1, 1, 0), + ], + vec![text("sent")], + ]); + let tools = + FakeTools::new(vec![write_tool("send_email")]).verdict("send_email", approval("ap-1")); + let engine = engine(&log, &model, &tools, budget()); + let mut ctx = log.start_turn("t1", "email them"); + let cancel = CancellationToken::new(); + assert_eq!( + engine.run(&mut ctx, &cancel).await, + Ok(Exit::Parked(ApprovalId::new("ap-1"))) + ); + let committed: Vec<_> = log + .events() + .into_iter() + .filter_map(|event| match event { + Event::ModelStepCompleted { reasoning, .. } => Some(reasoning), + _ => None, + }) + .collect(); + assert_eq!(committed, vec![Some(reasoning.clone())]); + + // -- crash: a fresh engine and context from the log, then the approval -- + let engine = support::engine(&log, &model, &tools, budget()); + log.decide(&call_id("t1", 1, 0), "ap-1", true); + let mut ctx = log.rehydrate(); + assert_eq!(engine.run(&mut ctx, &cancel).await, Ok(Exit::Done)); + + let seen = model.seen(); + let returned: Vec<_> = seen[1] + .iter() + .filter_map(|message| match message { + dex_loop::Message::Assistant { reasoning, .. } => Some(reasoning.clone()), + _ => None, + }) + .collect(); + assert_eq!(returned, vec![Some(reasoning)]); + assert_eq!(log.rehydrate(), ctx); +} diff --git a/vendor/dex-loop/tests/sim.rs b/vendor/dex-loop/tests/sim.rs new file mode 100644 index 000000000..9459cc33f --- /dev/null +++ b/vendor/dex-loop/tests/sim.rs @@ -0,0 +1,448 @@ +//! A deterministic simulator for the dex-loop kernel: seeded, adversarial, +//! in-memory ports (`SimLog`, `SimEffects`, `SimModel`, `SimTools`) plus a +//! crash-injection primitive and a lease-generation fence, driven by a +//! sequence of host `Action`s and checked against the invariants in +//! `invariants.rs`. +//! +//! Everything adversarial is a pure function of `(seed, stable key)` -- +//! never a shared mutable RNG consumed in call order -- so the outcome does +//! not depend on how the tokio scheduler happens to interleave concurrent +//! work: the same seed always produces the same script, however the two +//! halves of a race are scheduled. +//! +//! Scope: this proves the `dex-loop` kernel contract (`Engine`, `Context`, +//! `rehydrate`, the five ports) under crashes, a lease-generation race, and +//! adversarial model/tool/control input. It does not run dex-runtime's own +//! Postgres-backed `Log`/`Effects`/lease implementation (`log.rs`, +//! `lease.rs`, `actor.rs`): `SimLog`/`SimEffects` model the same *contract* +//! those implement (durable writes, `Fenced` on a stale generation, control +//! events readable by cursor), not their SQL. `invariants::would_release` +//! re-implements `lease::finish`'s three SQL predicates in pure Rust against +//! a hand-built event log, so `stale_control_kinds_would_release_a_lease_with_unprocessed_control_event` +//! below can check dex-runtime's real `CONTROL_KINDS` list (copied into +//! `invariants.rs`, with a static-parity test tying the copy to +//! `dex_loop::Event::is_control`) without a database. +//! +//! Client-executor tools (`ExecutorKind::Client`) are included in the +//! action space because the brief asks for them, but per-service wiring for +//! client tools is mid-flight in another PR at the time this was written +//! (dex-tools/platform-api); invariant violations whose only offending call +//! is a `Client`-executor call are recorded as `Violation::known_pending` +//! and reported, not failed, so this suite does not block on a gap already +//! being repaired elsewhere. Every other violation fails the test. + +// `tests/sim.rs` is this test binary's crate root, so a plain `mod fakes;` +// would look for `tests/fakes.rs` (a sibling of `sim.rs`, not a `sim/` +// subdirectory: that convention is for modules declared *inside* a +// non-root file). `#[path]` keeps the actual files grouped under `tests/sim/` +// without them being picked up as their own top-level test binaries (which +// `tests/fakes.rs` directly under `tests/` would be, per Cargo's autotests). +#[path = "sim/fakes.rs"] +mod fakes; +#[path = "sim/invariants.rs"] +mod invariants; +#[path = "sim/scenario.rs"] +mod scenario; + +use std::time::Duration; + +use dex_loop::{ + Budget, CancellationToken, Context, Effects, Engine, Event, ExecutorKind, Exit, + GovernanceClass, Lexicon, Log, PrincipalId, ThreadId, ToolName, ToolSpec, TurnId, rehydrate, +}; +use fakes::{CrashBudget, SimEffects, SimLog, SimModel, SimTools}; +use proptest::prelude::*; +use scenario::{action_sequence, run_actions, run_seed}; + +/// How many seeds the default (CI) run covers. Overridden by `DEX_SIM_SEEDS` +/// for a long soak (`DEX_SIM_SEEDS=20000` before pushing, `=100000` for an +/// overnight run); the default keeps the whole suite comfortably under a +/// minute. +fn seed_count() -> u64 { + std::env::var("DEX_SIM_SEEDS") + .ok() + .and_then(|value| value.parse().ok()) + .unwrap_or(200) +} + +fn base_seed() -> u64 { + // A fixed offset, not 0: seed 0 is a legitimate and already-covered case + // (the empty-ish action sequence), and starting soaks at a non-zero + // offset means a `DEX_SIM_SEEDS=20000` soak and the default 200-seed run + // don't just repeat the same first 200 seeds every time this env var + // grows -- each extra seed explores new ground. + std::env::var("DEX_SIM_SEED_START") + .ok() + .and_then(|value| value.parse().ok()) + .unwrap_or(0) +} + +/// The bounded, default-CI sweep: every seed's action sequence and +/// adversarial choices are fully determined by the seed (see +/// `scenario::actions_for_seed` and `fakes::SimModel`/`SimTools`), so a +/// failure here is reproducible by re-running `run_seed(seed)` alone. +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn dst_bounded_seeds() { + let start = base_seed(); + let count = seed_count(); + let mut failures = Vec::new(); + for seed in start..start + count { + let violations = run_seed(seed).await; + let hard: Vec<_> = violations.iter().filter(|v| !v.known_pending).collect(); + if !hard.is_empty() { + failures.push(( + seed, + hard.iter() + .map(|v| v.description.clone()) + .collect::>(), + )); + } + for pending in violations.iter().filter(|v| v.known_pending) { + eprintln!( + "seed {seed}: known-pending (client tools): {}", + pending.description + ); + } + } + assert!( + failures.is_empty(), + "dex-loop DST found {} failing seed(s) out of {count} (start {start}):\n{}", + failures.len(), + failures + .iter() + .map(|(seed, messages)| format!(" seed {seed}: {}", messages.join("; "))) + .collect::>() + .join("\n") + ); +} + +proptest! { + #![proptest_config(ProptestConfig { + // The default CI run stays bounded; a long soak sets + // `PROPTEST_CASES` (proptest's own env var) directly. + cases: 64, + // This is an integration test binary (`tests/sim.rs`), not + // `src/lib.rs`: proptest's default `.proptest-regressions` file + // lookup assumes a crate root under `src/` and warns on every run + // that it cannot find one. The failure message itself already + // prints the shrunk `Vec` and the `seed` needed to + // reproduce, so this only gives up automatic persistence across + // runs, not reproducibility of any single failure. + failure_persistence: None, + .. ProptestConfig::default() + })] + + /// The same invariants as `dst_bounded_seeds`, but over `proptest`-shrunk + /// `Action` sequences: a failure here prints a minimized trace (the + /// shrunk `Vec`) and the seed that drove the model/tool + /// adversarial choices, per `proptest`'s own failure output plus the + /// `panic!` message below. + #[test] + fn dst_action_sequences_hold_invariants(seed in any::(), actions in action_sequence()) { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .start_paused(true) + .build() + .expect("build a current-thread runtime"); + let violations = runtime.block_on(run_actions(seed, &actions)); + let hard: Vec<_> = violations.iter().filter(|v| !v.known_pending).collect(); + prop_assert!( + hard.is_empty(), + "seed {seed} actions {actions:?}:\n{}", + hard.iter().map(|v| v.description.clone()).collect::>().join("\n") + ); + } +} + +/// Static parity between the kernel's `Event::is_control` classification and +/// the ingress-kind lists dex-runtime's actor keys wake/lease-finish +/// decisions on (`CONTROL_KINDS` in `log.rs`, `WAKE_KINDS` in `actor.rs`). +/// As of this writing both lists include `client_tool_result` (fixed by +/// #11109); this test pins that so a future edit to either list, or to +/// `Event::is_control`, has to update `invariants::DEX_RUNTIME_CONTROL_KINDS` +/// / `DEX_RUNTIME_WAKE_KINDS` deliberately instead of silently drifting. +#[test] +fn control_kind_parity_matches_kernel() { + invariants::control_kind_parity() + .expect("dex-runtime's kind lists must match Event::is_control"); +} + +/// Pure re-implementation of `lease::finish`'s three predicates, isolating +/// its *second* one (an unseen control event) from its first (any unseen +/// write at all): `seen` is set past the `ClientToolResult`'s own cursor (as +/// if the actor's last write landed after it without ever having read it), +/// so only the control-kind-specific predicate stands between "release" and +/// "run again". A control-kind list that omits `client_tool_result` (the +/// pre-#11109 state) fails to catch this and would release; the current +/// list catches it. In the actually-reachable pre-#11109 failure, nothing +/// this narrow was needed: `WAKE_KINDS` (also missing `client_tool_result` +/// before #11109) meant no actor was ever woken to reach `lease::finish` at +/// all, which `control_kind_parity_matches_kernel` above now pins for both +/// lists. This test additionally pins `lease::finish`'s own predicate as a +/// second line of defense, in case an actor ever *is* running when a +/// control event outside `CONTROL_KINDS` arrives. +#[test] +fn stale_control_kinds_would_release_a_lease_with_unprocessed_control_event() { + let events = vec![ + ( + dex_loop::Cursor(1), + Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: PrincipalId::new("alice"), + text: "hi".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }, + ), + ( + dex_loop::Cursor(2), + Event::ClientToolRequested { + call: dex_loop::CallId::new("t1-1-0"), + tool: ToolName::new("client.read"), + args: serde_json::json!({}), + label: "Client read".into(), + principal: PrincipalId::new("alice"), + target_session: "alice".into(), + deadline_ms: i64::MAX, + }, + ), + ( + // Never read via `control_since` in this scenario (that is + // exactly the bug being isolated): `control_seen` below stays at + // its pre-request value. + dex_loop::Cursor(3), + Event::ClientToolResult { + call: dex_loop::CallId::new("t1-1-0"), + principal: PrincipalId::new("alice"), + outcome: dex_loop::Outcome::Succeeded, + output: "ok".into(), + }, + ), + ( + // A later, unrelated write this actor made *without* ever having + // observed cursor 3: pushes `seen` past the missed event so + // predicate 1 (any unseen write) does not also catch it, + // isolating predicate 2 (an unseen *control* event). + dex_loop::Cursor(4), + Event::ToolProgress { + call: dex_loop::CallId::new("t1-1-1"), + label: "still working".into(), + }, + ), + ]; + let seen = dex_loop::Cursor(4); + let control_seen = dex_loop::Cursor(2); + let stale_control_kinds = ["steer", "interrupt", "approval_decided", "answer"]; + let current_control_kinds = invariants::DEX_RUNTIME_CONTROL_KINDS; + + assert!( + invariants::would_release(&events, &stale_control_kinds, seen, control_seen, "t1"), + "with the pre-#11109 CONTROL_KINDS list, lease::finish's own predicates would \ + (wrongly) release the lease despite an unprocessed client_tool_result" + ); + assert!( + !invariants::would_release(&events, ¤t_control_kinds, seen, control_seen, "t1"), + "with the current CONTROL_KINDS list, lease::finish correctly refuses to release \ + and the actor loops again to pick up the client tool result" + ); +} + +fn thread_for(name: &str) -> ThreadId { + ThreadId { + org: "org-fence".into(), + workspace: "ws-fence".into(), + thread: name.into(), + } +} + +fn mutation_catalog() -> Vec { + vec![ToolSpec { + name: ToolName::new("mutator"), + label: "Mutator".into(), + schema: serde_json::json!({"type": "object"}), + read_only: false, + core: true, + governance: GovernanceClass::Plain, + executor: ExecutorKind::ToolExecutor, + }] +} + +fn mutation_budget() -> Budget { + Budget { + max_steps: 10, + max_tokens: u64::MAX, + max_cost_micros: u64::MAX, + wall: Duration::from_secs(5), + } +} + +/// Invariant (1) under a genuine lease race: two replicas share the same +/// durable `SimLog`/`SimEffects` but start at the SAME generation (a +/// stronger, adversarial version of "the lease worked" -- this is the +/// defense-in-depth case where it didn't, and two actors briefly believe +/// they both hold the thread). Even then, `Effects::claim`'s shared ledger +/// mutex lets only one of the two concurrent `run()` calls dispatch the +/// mutation; the other adopts the recorded result instead of running it +/// again. +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn dst_two_replicas_racing_the_same_generation_dispatch_once() { + let thread = thread_for("race"); + let crash = CrashBudget::none(); + let log = SimLog::new(crash.clone()); + let effects = SimEffects::new(crash.clone()); + let tools_seed = 7; + let tools = SimTools::new(mutation_catalog(), tools_seed, crash.clone()); + let model = SimModel::fixed(tools_seed, fakes::StepScript::OneMutation); + + log.host_append(Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: PrincipalId::new("alice"), + text: "mutate".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + + let build = || { + Engine::new( + log.for_replica(0, crash.clone()), + model.clone(), + tools.with_crash(crash.clone()), + effects.for_replica(0, crash.clone()), + Lexicon::default(), + mutation_budget(), + ) + }; + let (engine_a, engine_b) = (build(), build()); + let mut ctx_a: Context = rehydrate(thread.clone(), &log.entries()); + let mut ctx_b: Context = rehydrate(thread.clone(), &log.entries()); + let (cancel_a, cancel_b) = (CancellationToken::new(), CancellationToken::new()); + + let (result_a, result_b) = tokio::join!( + engine_a.run(&mut ctx_a, &cancel_a), + engine_b.run(&mut ctx_b, &cancel_b), + ); + // Both replicas may see a successful run (each just replays whichever of + // the two outcomes the shared ledger ended up recording): what matters + // is that the mutation itself only ran once. + assert!( + result_a.is_ok() && result_b.is_ok(), + "{result_a:?} {result_b:?}" + ); + let dispatches = tools.dispatches(); + assert_eq!( + dispatches.len(), + 1, + "the mutation must be dispatched exactly once across both racing replicas, got {dispatches:?}" + ); +} + +/// Invariant (1) under lease loss: replica A runs partway, then a later +/// replica steals the lease (the fence advances); every write A attempts +/// after that must fail with `Fenced`, and replica B -- rehydrating from +/// whatever A left durable -- finishes the mutation exactly once. +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn dst_two_replica_lease_fencing() { + let thread = thread_for("fence"); + let crash_a = CrashBudget::at(0); // A crashes at its very first port op. + let log = SimLog::new(CrashBudget::none()); + let effects = SimEffects::new(CrashBudget::none()); + let tools_seed = 11; + let tools = SimTools::new(mutation_catalog(), tools_seed, CrashBudget::none()); + let model = SimModel::fixed(tools_seed, fakes::StepScript::OneMutation); + + log.host_append(Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: PrincipalId::new("alice"), + text: "mutate".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + + // Replica A: generation 0, crashes immediately. + let log_a = log.for_replica(0, crash_a.clone()); + let effects_a = effects.for_replica(0, crash_a.clone()); + let engine_a = Engine::new( + log_a.clone(), + model.clone(), + tools.with_crash(crash_a), + effects_a.clone(), + Lexicon::default(), + mutation_budget(), + ); + let mut ctx_a = rehydrate(thread.clone(), &log.entries()); + let cancel_a = CancellationToken::new(); + let outcome_a = + tokio::time::timeout(Duration::from_secs(5), engine_a.run(&mut ctx_a, &cancel_a)).await; + assert!( + outcome_a.is_err(), + "replica A's crashed attempt must not resolve" + ); + + // A later replica steals the lease: the fence advances to generation 1. + // A's handles (still generation 0) must now be refused on every write. + let new_generation = log.fence().steal(); + assert_eq!( + effects.fence().steal(), + new_generation, + "log and effect fences move together in this test" + ); + let stray_append = log_a + .append(&[Event::Interrupt { + principal: PrincipalId::new("alice"), + }]) + .await; + assert!( + stray_append.is_err(), + "replica A must be fenced once the lease generation moves, not just slow" + ); + let stray_claim = effects_a + .claim(&dex_loop::ProposedCall::new( + dex_loop::CallId::new("t1-1-0"), + ToolName::new("mutator"), + serde_json::json!({}), + PrincipalId::new("alice"), + )) + .await; + assert!( + stray_claim.is_err(), + "replica A's effects handle must be fenced too" + ); + + // Replica B: the new generation, rehydrating from whatever A left + // durable (in this case, nothing -- A crashed before its first op). + let log_b = log.for_replica(new_generation, CrashBudget::none()); + let effects_b = effects.for_replica(new_generation, CrashBudget::none()); + let engine_b = Engine::new( + log_b, + model.clone(), + tools.with_crash(CrashBudget::none()), + effects_b, + Lexicon::default(), + mutation_budget(), + ); + let mut ctx_b = rehydrate(thread.clone(), &log.entries()); + let cancel_b = CancellationToken::new(); + let outcome_b = engine_b.run(&mut ctx_b, &cancel_b).await; + assert_eq!(outcome_b, Ok(Exit::Done)); + + let dispatches = tools.dispatches(); + assert_eq!( + dispatches.len(), + 1, + "the mutation must be dispatched exactly once across the fenced replica and its successor, got {dispatches:?}" + ); + let full = rehydrate(thread.clone(), &log.entries()); + assert_eq!( + full, ctx_b, + "rehydrate(full log) must equal replica B's live context" + ); +} diff --git a/vendor/dex-loop/tests/sim/fakes.rs b/vendor/dex-loop/tests/sim/fakes.rs new file mode 100644 index 000000000..65821e83a --- /dev/null +++ b/vendor/dex-loop/tests/sim/fakes.rs @@ -0,0 +1,615 @@ +//! Seeded, adversarial in-memory ports. + +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; + +use dex_loop::{ + CallId, CancellationToken, Claim, Context, Cursor, Effects, Event, Fenced, Log, Model, + ModelChunk, ModelError, Outcome, Output, OutputRef, PrincipalId, ProposedCall, ThreadId, + ToolName, ToolResult, ToolSpec, Tools, Verdict, +}; +use futures_util::{Stream, stream}; + +pub(crate) fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex.lock().unwrap_or_else(PoisonError::into_inner) +} + +fn hash_of(parts: &[&str]) -> u64 { + let mut hasher = std::collections::hash_map::DefaultHasher::new(); + for part in parts { + part.hash(&mut hasher); + } + hasher.finish() +} + +// --------------------------------------------------------------- Fence + +/// A lease-like generation counter shared by every simulated replica of one +/// thread. Advancing it fences every older handle at once: the same CAS +/// dex-runtime's real lease does with `lease_generation` in Postgres +/// (`lease::acquire`/`lease::finish`), modeled here without a database. +#[derive(Clone, Default)] +pub struct Fence(Arc); + +impl Fence { + pub fn generation(&self) -> u64 { + self.0.load(Ordering::SeqCst) + } + + /// A new replica takes the lease: bumps the generation and returns it. + pub fn steal(&self) -> u64 { + self.0.fetch_add(1, Ordering::SeqCst) + 1 + } +} + +// --------------------------------------------------------------- CrashBudget + +/// Lets a seed pick one operation, across a whole simulated `Engine::run` +/// attempt, to hang forever: every port operation calls `checkpoint()` once +/// it has made whatever mutation it makes durable, so hanging there models +/// the process dying with that mutation already landed and nothing that +/// would have happened afterward (including the engine's own continuation) +/// ever happening. The caller races the whole `Engine::run` against a wall +/// budget (paused virtual time; free in real time either way) and treats a +/// timeout exactly like a crash: drop the future, rehydrate, continue with a +/// fresh `Engine`. +#[derive(Clone)] +pub struct CrashBudget { + counter: Arc, + crash_at: Option, +} + +impl CrashBudget { + pub fn none() -> Self { + Self { + counter: Arc::new(AtomicU64::new(0)), + crash_at: None, + } + } + + pub fn at(n: u64) -> Self { + Self { + counter: Arc::new(AtomicU64::new(0)), + crash_at: Some(n), + } + } + + pub async fn checkpoint(&self) { + let seen = self.counter.fetch_add(1, Ordering::SeqCst); + if self.crash_at == Some(seen) { + std::future::pending::<()>().await; + } + } +} + +// --------------------------------------------------------------- SimLog + +#[derive(Default)] +struct LogState { + events: Vec<(Cursor, Event)>, +} + +impl LogState { + fn push(&mut self, event: Event) -> Cursor { + let cursor = Cursor(self.events.len() as i64 + 1); + self.events.push((cursor, event)); + cursor + } +} + +/// The thread's event log: durable, cursor-ordered, and fenced once a later +/// replica has stolen the lease (see `Fence`). +#[derive(Clone)] +pub struct SimLog { + state: Arc>, + fence: Fence, + generation: u64, + crash: CrashBudget, +} + +impl SimLog { + pub fn new(crash: CrashBudget) -> Self { + Self { + state: Arc::default(), + fence: Fence::default(), + generation: 0, + crash, + } + } + + /// A handle for a replica running at `generation`: shares the same + /// durable state as every other handle from this log, but every write + /// through it is refused the instant the fence moves past `generation`. + pub fn for_replica(&self, generation: u64, crash: CrashBudget) -> Self { + Self { + state: Arc::clone(&self.state), + fence: self.fence.clone(), + generation, + crash, + } + } + + pub fn fence(&self) -> &Fence { + &self.fence + } + + pub fn generation(&self) -> u64 { + self.generation + } + + fn fenced(&self) -> bool { + self.fence.generation() != self.generation + } + + /// A host-side ingress write (`send`/`steer`/`interrupt`/`approve`/ + /// `answer`): never fenced, since ingress in dex-runtime writes through + /// a thread-row lock, not the actor's lease. + pub fn host_append(&self, event: Event) -> Cursor { + lock(&self.state).push(event) + } + + pub fn entries(&self) -> Vec<(Cursor, Event)> { + lock(&self.state).events.clone() + } + + /// Every event with a cursor greater than `after`, in log order: the + /// same contract a `Watch` resumption offers a subscriber. + pub fn since(&self, after: Cursor) -> Vec<(Cursor, Event)> { + lock(&self.state) + .events + .iter() + .filter(|(cursor, _)| *cursor > after) + .cloned() + .collect() + } +} + +impl Log for SimLog { + async fn append(&self, events: &[Event]) -> Result, Fenced> { + if self.fenced() { + return Err(Fenced::new("lease generation moved")); + } + let cursors: Vec = { + let mut state = lock(&self.state); + events + .iter() + .map(|event| state.push(event.clone())) + .collect() + }; + self.crash.checkpoint().await; + Ok(cursors) + } + + async fn append_text(&self, text: String) -> Result<(), Fenced> { + if self.fenced() { + return Err(Fenced::new("lease generation moved")); + } + { + let mut state = lock(&self.state); + if let Some((_, Event::TextDelta { text: row })) = state.events.last_mut() { + row.push_str(&text); + } else { + state.push(Event::TextDelta { text }); + } + } + self.crash.checkpoint().await; + Ok(()) + } + + async fn control_since(&self, after: Cursor) -> Result, Fenced> { + // Reads are never fenced: a stale replica may still read (it just + // cannot act on what it reads without a write landing), matching + // `Log`'s doc ("the log or the effect ledger refused *a write*"). + Ok(lock(&self.state) + .events + .iter() + .filter(|(cursor, event)| *cursor > after && event.is_control()) + .cloned() + .collect()) + } +} + +// --------------------------------------------------------------- SimEffects + +/// The durable effect ledger: `CallId` to its recorded outcome, or `None` +/// while claimed but not yet recorded (the "crashed mid-dispatch" state that +/// settles to `Outcome::Unknown`). +#[derive(Clone)] +pub struct SimEffects { + ledger: Arc>>>, + fence: Fence, + generation: u64, + crash: CrashBudget, +} + +impl SimEffects { + pub fn new(crash: CrashBudget) -> Self { + Self { + ledger: Arc::default(), + fence: Fence::default(), + generation: 0, + crash, + } + } + + pub fn for_replica(&self, generation: u64, crash: CrashBudget) -> Self { + Self { + ledger: Arc::clone(&self.ledger), + fence: self.fence.clone(), + generation, + crash, + } + } + + pub fn fence(&self) -> &Fence { + &self.fence + } + + fn fenced(&self) -> bool { + self.fence.generation() != self.generation + } +} + +impl Effects for SimEffects { + async fn claim(&self, call: &ProposedCall) -> Result { + if self.fenced() { + return Err(Fenced::new("lease generation moved")); + } + let claim = { + let mut ledger = lock(&self.ledger); + match ledger.get(&call.id) { + Some(Some(result)) => Claim::Existing(result.clone()), + Some(None) => Claim::Existing(ToolResult { + outcome: Outcome::Running, + output: Output::Text("dispatched; no outcome recorded yet".into()), + receipt: None, + }), + None => { + ledger.insert(call.id.clone(), None); + Claim::Granted + } + } + }; + self.crash.checkpoint().await; + Ok(claim) + } + + async fn record(&self, call: &CallId, result: &ToolResult) -> Result<(), Fenced> { + if self.fenced() { + return Err(Fenced::new("lease generation moved")); + } + lock(&self.ledger).insert(call.clone(), Some(result.clone())); + self.crash.checkpoint().await; + Ok(()) + } +} + +// --------------------------------------------------------------- SimModel + +/// What the model does on one step, chosen deterministically from +/// `(seed, turn, step)`. A pure function, never a shared mutable RNG: the +/// same seed produces the same script regardless of how concurrent work is +/// scheduled around it. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum StepScript { + /// A plain answer with no tool calls; ends the turn. + PlainAnswer, + /// One call to the first read-only tool offered (falls back to a plain + /// answer if none is offered). + OneRead, + /// One call to the first mutation tool offered. + OneMutation, + /// One call to the first `ExecutorKind::Client` tool offered. + OneClientTool, + /// The same call (same tool, same args) proposed twice in one step. + DuplicateCalls, + /// A call to a tool name that is not offered. + UnknownTool, + /// One chunk of text, then the stream ends with no `Err` and no more + /// chunks -- truncated mid-answer, not abandoned. + Truncated, + /// One chunk of text, then the stream errors. + Abandoned, + /// The stream never yields anything at all (nor ends, nor errors). + Stalled, + /// A plain answer carrying an adversarial payload: NUL bytes, a huge + /// string, or non-ASCII text with colon-bearing content. + AdversarialPayload, +} + +const STEP_SCRIPTS: [StepScript; 10] = [ + StepScript::PlainAnswer, + StepScript::OneRead, + StepScript::OneMutation, + StepScript::OneClientTool, + StepScript::DuplicateCalls, + StepScript::UnknownTool, + StepScript::Truncated, + StepScript::Abandoned, + StepScript::Stalled, + StepScript::AdversarialPayload, +]; + +pub fn adversarial_payload(seed: u64) -> String { + match seed % 4 { + 0 => "plain reply".to_owned(), + 1 => format!("has a NUL\u{0}byte and a colon: {seed}"), + 2 => "🜁 unicode 漢字 café \u{200b} zero-width".to_owned(), + _ => "x".repeat(200_000), + } +} + +/// One `Model` whose per-step behavior is a pure function of `(seed, turn, +/// step)`: a fresh `SimModel` re-synthesizes the same sequence of steps a +/// crashed one already committed, so replays after a crash line up with what +/// the log actually recorded, but a *new* step (one that was never +/// committed) can still land differently across attempts, as a real +/// non-deterministic model would. +#[derive(Clone)] +pub struct SimModel { + seed: u64, + only: Option, +} + +impl SimModel { + pub fn new(seed: u64) -> Self { + Self { seed, only: None } + } + + /// Always the same script, ignoring the seed's choice of variant (its + /// payload/tool selection still varies by seed). Used by properties that + /// isolate one dimension (e.g. the lease-fencing test, which wants a + /// plain mutation every turn, not an adversarial stream on top of it). + pub fn fixed(seed: u64, script: StepScript) -> Self { + Self { + seed, + only: Some(script), + } + } + + /// `fixed()`'s override applies only until history already shows a + /// committed step that proposed a call, so a fixed "one mutation" model + /// still lets the turn conclude afterward instead of proposing the same + /// call forever and running the budget out. Gated on committed history + /// rather than the step *number*: a crash can abandon an attempt (no + /// `ModelStepCompleted`, so nothing enters history) and force a retry at + /// a higher step number for what is still, from the model's point of + /// view, its first real turn at bat. + fn script_for(&self, ctx: &Context, key: u64) -> StepScript { + let already_proposed = ctx.history().iter().any(|entry| { + matches!(&entry.message, dex_loop::Message::Assistant { calls, .. } if !calls.is_empty()) + }); + match self.only { + Some(script) if !already_proposed => script, + Some(_) => StepScript::PlainAnswer, + None => STEP_SCRIPTS[(key % STEP_SCRIPTS.len() as u64) as usize], + } + } +} + +impl Model for SimModel { + fn stream<'a>( + &'a self, + ctx: &'a Context, + tools: &'a [&'a ToolSpec], + ) -> impl Stream> + Send + 'a { + let turn = ctx.turn().map(ToString::to_string).unwrap_or_default(); + // `ctx.step()` already reflects the step this call is for: the + // engine appends and observes `StepStarted { step, .. }` before + // calling `Model::stream` (see `Engine::model_step`), so this must + // not add 1 again -- doing so put every real first step at "step 2" + // here, so `fixed()`'s "only on step 1" override never fired. + let step = ctx.step(); + let seed_s = self.seed.to_string(); + let step_s = step.to_string(); + let key = hash_of(&[&seed_s, &turn, &step_s]); + let script = self.script_for(ctx, key); + let read = tools + .iter() + .find(|spec| spec.read_only && spec.name.as_str() != "tools.search"); + let mutation = tools + .iter() + .find(|spec| !spec.read_only && spec.executor != dex_loop::ExecutorKind::Client); + let client = tools + .iter() + .find(|spec| spec.executor == dex_loop::ExecutorKind::Client); + let text_reply = |text: String| vec![Ok(ModelChunk::Text(text))]; + let call = |spec: &&ToolSpec| { + vec![Ok(ModelChunk::ToolCall { + name: spec.name.clone(), + args: serde_json::json!({"key": spec.name.as_str(), "seed": key}), + })] + }; + let script_chunks: Vec> = match script { + StepScript::PlainAnswer => text_reply(format!("done ({key})")), + StepScript::OneRead => read + .map(call) + .unwrap_or_else(|| text_reply("no read tool".into())), + StepScript::OneMutation => mutation + .map(call) + .unwrap_or_else(|| text_reply("no mutation tool".into())), + StepScript::OneClientTool => client + .map(call) + .unwrap_or_else(|| text_reply("no client tool".into())), + StepScript::DuplicateCalls => match read.or(mutation) { + Some(spec) => { + let mut chunks = call(spec); + chunks.extend(call(spec)); + chunks + } + None => text_reply("no tool to duplicate".into()), + }, + StepScript::UnknownTool => vec![Ok(ModelChunk::ToolCall { + name: dex_loop::ToolName::new(format!("nonexistent.tool.{key}")), + args: serde_json::json!({}), + })], + StepScript::Truncated => vec![Ok(ModelChunk::Text("truncated mid".into()))], + StepScript::Abandoned => vec![ + Ok(ModelChunk::Text("about to fail".into())), + Err(ModelError { + message: format!("upstream reset ({key})"), + }), + ], + StepScript::Stalled => Vec::new(), + StepScript::AdversarialPayload => text_reply(adversarial_payload(key)), + }; + // `Stalled` never yields and never ends -- a genuinely different + // shape from "a short `script_chunks`", which properly ends. Boxing + // is the simplest way to return either shape from one opaque + // `impl Stream` return type. + let stalled = matches!(script, StepScript::Stalled); + let boxed: std::pin::Pin< + Box> + Send + 'a>, + > = if stalled { + Box::pin(stream::pending()) + } else { + Box::pin(stream::iter(script_chunks)) + }; + boxed + } +} + +// --------------------------------------------------------------- SimTools + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum OutcomeHint { + Succeed, + Fail, + Unknown, +} + +/// The tool registry: a fixed catalog (mirroring real hosts, whose registry +/// does not change tool-by-tool at runtime) plus a `vanished` set that hides +/// specific tools from `spec()` -- and therefore from dispatch -- without +/// touching `catalog()`, modeling a tool that a deploy or a grant revoke +/// removed mid-turn (the model was already offered it and proposed a call; +/// the offer is gone by the time the engine goes to run it). +#[derive(Clone)] +pub struct SimTools { + catalog: Arc<[ToolSpec]>, + vanished: Arc>>, + /// Tools policy denies regardless of governance, set by an action that + /// simulates a grant revoke or a policy change while a call sits parked: + /// `dispatch` always re-runs `Tools::policy` on resume, so a call already + /// approved must still be denied once this is set. + denied: Arc>>, + seed: u64, + crash: CrashBudget, + dispatches: Arc>>, +} + +impl SimTools { + pub fn new(catalog: Vec, seed: u64, crash: CrashBudget) -> Self { + Self { + catalog: catalog.into(), + vanished: Arc::default(), + denied: Arc::default(), + seed, + crash, + dispatches: Arc::default(), + } + } + + /// The same catalog, ledger and dispatch history, but a fresh crash + /// budget: used to inject a crash into exactly one `Engine::run` attempt + /// without disturbing anything else about the simulated tool registry. + pub fn with_crash(&self, crash: CrashBudget) -> Self { + Self { + crash, + ..self.clone() + } + } + + pub fn vanish(&self, name: &str) { + lock(&self.vanished).insert(name.to_owned()); + } + + pub fn restore(&self, name: &str) { + lock(&self.vanished).remove(name); + } + + pub fn force_deny(&self, name: &str) { + lock(&self.denied).insert(name.to_owned()); + } + + pub fn clear_deny(&self, name: &str) { + lock(&self.denied).remove(name); + } + + /// Every `CallId` this instance actually dispatched (`Tools::run` + /// returned for it), in dispatch order. Used to check that a mutation is + /// never dispatched twice. + pub fn dispatches(&self) -> Vec { + lock(&self.dispatches).clone() + } + + fn outcome_hint(&self, call: &CallId) -> OutcomeHint { + let seed_s = self.seed.to_string(); + match hash_of(&[&seed_s, call.as_str()]) % 10 { + 0..=6 => OutcomeHint::Succeed, + 7..=8 => OutcomeHint::Fail, + _ => OutcomeHint::Unknown, + } + } +} + +impl Tools for SimTools { + fn catalog(&self) -> &[ToolSpec] { + &self.catalog + } + + fn spec(&self, name: &ToolName) -> Option<&ToolSpec> { + if lock(&self.vanished).contains(name.as_str()) { + return None; + } + self.catalog.iter().find(|spec| &spec.name == name) + } + + async fn search(&self, _principal: &PrincipalId, query: &str) -> Vec { + self.catalog + .iter() + .filter(|spec| spec.name.as_str().contains(query) && self.spec(&spec.name).is_some()) + .map(|spec| spec.name.clone()) + .collect() + } + + async fn policy(&self, _ctx: &Context, call: &ProposedCall) -> Verdict { + if lock(&self.denied).contains(call.tool.as_str()) { + return Verdict::Deny("policy changed while this call was pending".into()); + } + match self.spec(&call.tool) { + Some(spec) if spec.governance == dex_loop::GovernanceClass::Approval => { + Verdict::NeedsApproval { + approval: dex_loop::ApprovalId::new(format!("approval-{}", call.id)), + summary: format!("Approve {}", spec.label), + } + } + _ => Verdict::Allow, + } + } + + async fn run( + &self, + _thread: &ThreadId, + call: &ProposedCall, + _cancel: &CancellationToken, + ) -> ToolResult { + // The dispatch itself is the irreversible side effect: record it + // before the crash checkpoint, so a crash that lands *inside* this + // call still shows up as "dispatched", exactly like a real mutation + // whose network call went out before the process died. + lock(&self.dispatches).push(call.id.clone()); + self.crash.checkpoint().await; + match self.outcome_hint(&call.id) { + OutcomeHint::Succeed => { + ToolResult::stored(OutputRef::new(format!("out/{}", call.id)), None) + } + OutcomeHint::Fail => ToolResult::error(format!("tool {} failed", call.id)), + OutcomeHint::Unknown => { + ToolResult::unknown(format!("tool {} timed out; effect unknown", call.id)) + } + } + } +} diff --git a/vendor/dex-loop/tests/sim/invariants.rs b/vendor/dex-loop/tests/sim/invariants.rs new file mode 100644 index 000000000..627cdf7d0 --- /dev/null +++ b/vendor/dex-loop/tests/sim/invariants.rs @@ -0,0 +1,444 @@ +//! Invariant checks against a finished (or quiescent) simulated log. + +use std::collections::{HashMap, HashSet}; + +use dex_loop::{CallId, Cursor, Event, Outcome, PrincipalId, ThreadId, ToolName, rehydrate}; + +use super::fakes::SimLog; + +/// A single invariant violation, or a known-pending one: client-executor +/// tool wiring is mid-flight in another PR at the time this simulator was +/// written (see `tests/sim.rs`'s module doc), so a violation whose only +/// offending call used `ExecutorKind::Client` is reported, not failed. +#[derive(Debug, Clone)] +pub struct Violation { + pub description: String, + pub known_pending: bool, +} + +impl Violation { + pub fn new(description: impl Into) -> Self { + Self { + description: description.into(), + known_pending: false, + } + } +} + +/// Exact copies of `Engine`'s private message constants, used only to tell a +/// "not run: interrupted" from an "unknown: interrupted" `ToolFinished` when +/// scanning the log from outside the crate. If `engine.rs` ever rewords +/// these, update this copy: the check that uses it fails loudly (as a +/// harmless false positive) rather than silently, since it only compares +/// against the outcome kind, never a message substring, for anything +/// load-bearing. +const NOT_RUN_INTERRUPTED: &str = "not run: the turn was interrupted"; + +/// dex-runtime's `actor.rs::WAKE_KINDS` / `log.rs::CONTROL_KINDS`, copied so +/// the static-parity check does not need dex-runtime as a dependency of +/// dex-loop's test target. Kept honest by `control_kind_parity_matches_kernel` +/// in `tests/sim.rs`, which fails loudly the day these fall out of sync +/// instead of silently testing a stale copy. +pub const DEX_RUNTIME_CONTROL_KINDS: [&str; 5] = [ + "steer", + "interrupt", + "approval_decided", + "answer", + "client_tool_result", +]; +pub const DEX_RUNTIME_WAKE_KINDS: [&str; 6] = [ + "user_message", + "steer", + "interrupt", + "approval_decided", + "answer", + "client_tool_result", +]; + +fn kind_str(event: &Event) -> &'static str { + match event { + Event::UserMessage { .. } => "user_message", + Event::Steer { .. } => "steer", + Event::Interrupt { .. } => "interrupt", + Event::ApprovalDecided { .. } => "approval_decided", + Event::Answer { .. } => "answer", + Event::ClientToolResult { .. } => "client_tool_result", + Event::StepStarted { .. } => "step_started", + Event::TextDelta { .. } => "text_delta", + Event::Usage(_) => "usage", + Event::ModelStepCompleted { .. } => "model_step_completed", + Event::ModelAttemptAbandoned { .. } => "model_attempt_abandoned", + Event::ToolStarted { .. } => "tool_started", + Event::ToolProgress { .. } => "tool_progress", + Event::ToolsExposed { .. } => "tools_exposed", + Event::ToolFinished { .. } => "tool_finished", + Event::ApprovalRequested { .. } => "approval_requested", + Event::Question { .. } => "question", + Event::ClientToolRequested { .. } => "client_tool_requested", + Event::Compaction { .. } => "compaction", + Event::Final { .. } => "final", + Event::Error { .. } => "error", + Event::Interrupted => "interrupted", + } +} + +/// Static parity between the kernel's own `Event::is_control` classification +/// and the ingress-kind string lists dex-runtime's actor keys its wake and +/// lease-finish decisions on. A kind present in `is_control` but missing +/// from `CONTROL_KINDS` means `lease::finish` cannot see that a control +/// event landed after the actor's last read, and will hand the thread back +/// as though nothing more needs doing. A kind present in `is_control` (other +/// than the ones ingress never turns into a wake, if any) but missing from +/// `WAKE_KINDS` means an actor is never woken for it. +pub fn control_kind_parity() -> Result<(), String> { + let is_control_kinds: HashSet<&'static str> = [ + "steer", + "interrupt", + "approval_decided", + "answer", + "client_tool_result", + ] + .into_iter() + .collect(); + let control: HashSet<&'static str> = DEX_RUNTIME_CONTROL_KINDS.into_iter().collect(); + let wake: HashSet<&'static str> = DEX_RUNTIME_WAKE_KINDS.into_iter().collect(); + if is_control_kinds != control { + return Err(format!( + "Event::is_control kinds {is_control_kinds:?} != dex-runtime CONTROL_KINDS {control:?}" + )); + } + // WAKE_KINDS additionally wakes on `user_message` (starting a thread), + // which is not itself a control event. + let mut expected_wake = control.clone(); + expected_wake.insert("user_message"); + if expected_wake != wake { + return Err(format!( + "expected WAKE_KINDS {expected_wake:?} (CONTROL_KINDS plus user_message), got {wake:?}" + )); + } + Ok(()) +} + +/// Pure re-implementation of `lease::finish`'s three `NOT EXISTS` predicates, +/// against an in-memory event log instead of Postgres, so actor-liveness +/// properties can be checked without a database. `control_kinds` is the +/// caller's copy of dex-runtime's `CONTROL_KINDS` (pass +/// `DEX_RUNTIME_CONTROL_KINDS` to check against the real list, or `is_control` +/// kinds directly to check what *should* happen). +pub fn would_release( + events: &[(Cursor, Event)], + control_kinds: &[&str], + seen: Cursor, + control_seen: Cursor, + turn: &str, +) -> bool { + let has_new_write = events.iter().any(|(cursor, _)| *cursor > seen); + let has_new_control = events + .iter() + .any(|(cursor, event)| *cursor > control_seen && control_kinds.contains(&kind_str(event))); + let latest_user_turn = events.iter().rev().find_map(|(_, event)| match event { + Event::UserMessage { turn, .. } => Some(turn.as_str().to_owned()), + _ => None, + }); + let turn_mismatch = latest_user_turn.is_some_and(|latest| latest != turn); + !(has_new_write || has_new_control || turn_mismatch) +} + +/// Per-call bookkeeping used by several checks below. +struct CallHistory { + tool: ToolName, + started_at: Option, + finished: Vec<(Cursor, Outcome, String)>, + approval_requested_digest: Option, + approvals_decided: Vec<(Cursor, bool, String, PrincipalId)>, +} + +fn by_call(events: &[(Cursor, Event)]) -> HashMap { + let mut calls: HashMap = HashMap::new(); + let mut names: HashMap = HashMap::new(); + for (_, event) in events { + if let Event::ModelStepCompleted { + calls: proposed, .. + } = event + { + for call in proposed { + names.insert(call.id.clone(), call.tool.clone()); + } + } + } + for (cursor, event) in events { + match event { + Event::ToolStarted { call, .. } => { + calls + .entry(call.clone()) + .or_insert_with(|| CallHistory { + tool: names + .get(call) + .cloned() + .unwrap_or_else(|| ToolName::new("?")), + started_at: None, + finished: Vec::new(), + approval_requested_digest: None, + approvals_decided: Vec::new(), + }) + .started_at + .get_or_insert(*cursor); + } + Event::ApprovalRequested { + call, args_digest, .. + } => { + calls + .entry(call.clone()) + .or_insert_with(|| CallHistory { + tool: names + .get(call) + .cloned() + .unwrap_or_else(|| ToolName::new("?")), + started_at: None, + finished: Vec::new(), + approval_requested_digest: None, + approvals_decided: Vec::new(), + }) + .approval_requested_digest = Some(args_digest.clone()); + } + Event::ApprovalDecided { + call, + args_digest, + approved, + principal, + .. + } => { + calls + .entry(call.clone()) + .or_insert_with(|| CallHistory { + tool: names + .get(call) + .cloned() + .unwrap_or_else(|| ToolName::new("?")), + started_at: None, + finished: Vec::new(), + approval_requested_digest: None, + approvals_decided: Vec::new(), + }) + .approvals_decided + .push((*cursor, *approved, args_digest.clone(), principal.clone())); + } + Event::ToolFinished { + call, + outcome, + output, + .. + } => { + let text = match output { + dex_loop::Output::Text(text) => text.clone(), + dex_loop::Output::Ref(reference) => reference.to_string(), + }; + calls + .entry(call.clone()) + .or_insert_with(|| CallHistory { + tool: names + .get(call) + .cloned() + .unwrap_or_else(|| ToolName::new("?")), + started_at: None, + finished: Vec::new(), + approval_requested_digest: None, + approvals_decided: Vec::new(), + }) + .finished + .push((*cursor, *outcome, text)); + } + _ => {} + } + } + calls +} + +/// Checks (1) at-most-once mutation dispatch, (2) `Unknown` never +/// re-dispatched, (3) approval binding (digest match; denial blocks the +/// call), and (6) interrupt never turns a started call into "not run". +/// `dispatched` is every `CallId` `SimTools::run` actually ran (in order, +/// duplicates included); `mutation_names`/`client_names` classify tool names +/// from the catalog actually used in the scenario. +pub fn check_log( + events: &[(Cursor, Event)], + dispatched: &[CallId], + mutation_names: &HashSet, + client_names: &HashSet, +) -> Vec { + let mut violations = Vec::new(); + let calls = by_call(events); + + // (1) + (2): count dispatches of mutation-class calls. + let mut counts: HashMap<&CallId, u32> = HashMap::new(); + for call in dispatched { + *counts.entry(call).or_default() += 1; + } + for (call, history) in &calls { + if !mutation_names.contains(history.tool.as_str()) { + continue; // reads may legitimately run more than once. + } + let count = counts.get(call).copied().unwrap_or(0); + if count > 1 { + let violation = Violation::new(format!( + "mutation {call} ({}) dispatched {count} times, want at most once", + history.tool + )); + violations.push(tag_known_pending(violation, &history.tool, client_names)); + } + } + + // (3) Approval binding. + for (call, history) in &calls { + let Some(requested_digest) = &history.approval_requested_digest else { + continue; + }; + let Some((_, approved, decided_digest, _)) = history.approvals_decided.first() else { + continue; // never decided in this trace: nothing to check yet. + }; + let succeeded = history + .finished + .iter() + .any(|(_, outcome, _)| *outcome == Outcome::Succeeded); + if !approved && succeeded { + let violation = Violation::new(format!( + "call {call} ({}) ran despite a denied approval", + history.tool + )); + violations.push(tag_known_pending(violation, &history.tool, client_names)); + } + if decided_digest != requested_digest && succeeded { + let violation = Violation::new(format!( + "call {call} ({}) ran despite an approval digest that does not match the requested one", + history.tool + )); + violations.push(tag_known_pending(violation, &history.tool, client_names)); + } + } + + // (6) Interrupt never turns a started call into "not run". + for (call, history) in &calls { + if history.started_at.is_none() { + continue; + } + for (_, outcome, message) in &history.finished { + if *outcome == Outcome::Failed && message == NOT_RUN_INTERRUPTED { + let violation = Violation::new(format!( + "call {call} ({}) started, but interrupt marked it \"not run\" instead of \"unknown\"", + history.tool + )); + violations.push(tag_known_pending(violation, &history.tool, client_names)); + } + } + } + + violations +} + +fn tag_known_pending( + violation: Violation, + tool: &ToolName, + client_names: &HashSet, +) -> Violation { + if client_names.contains(tool.as_str()) { + Violation { + known_pending: true, + ..violation + } + } else { + violation + } +} + +/// Invariant (4): at quiescence (no turn running, nothing parked), every +/// `UserMessage` has exactly one terminal (`Final`, `Error` or +/// `Interrupted`) -- no lost turns, no double finals. +pub fn check_quiescent_turn_count(events: &[(Cursor, Event)]) -> Result<(), String> { + let sent = events + .iter() + .filter(|(_, event)| matches!(event, Event::UserMessage { .. })) + .count(); + let terminals = events + .iter() + .filter(|(_, event)| { + matches!( + event, + Event::Final { .. } | Event::Error { .. } | Event::Interrupted + ) + }) + .count(); + if sent != terminals { + return Err(format!( + "{sent} UserMessage events but {terminals} terminal events at quiescence" + )); + } + Ok(()) +} + +/// Invariant (5): the principal on every `ModelStepCompleted`'s proposed +/// calls matches `rehydrate`'s own acting principal at that point -- the +/// kernel is the oracle here, so this also doubles as a cross-check that +/// `Context::observe`'s acting-principal bookkeeping (steers, pending turns) +/// agrees with itself when replayed incrementally versus in one shot. +pub fn check_principal_attribution( + thread: &ThreadId, + events: &[(Cursor, Event)], +) -> Vec { + let mut violations = Vec::new(); + for (index, (cursor, event)) in events.iter().enumerate() { + let Event::ModelStepCompleted { calls, .. } = event else { + continue; + }; + if calls.is_empty() { + continue; + } + let prefix = &events[..index]; + let ctx = rehydrate(thread.clone(), prefix); + let Some(expected) = ctx.acting_principal() else { + violations.push(Violation::new(format!( + "cursor {cursor:?}: calls proposed with no acting principal in context" + ))); + continue; + }; + for call in calls { + if &call.principal != expected { + violations.push(Violation::new(format!( + "cursor {cursor:?}: call {} principal {} != acting principal {} at that point", + call.id, call.principal, expected + ))); + } + } + } + violations +} + +/// Invariant (7), the "exactly once" half: a `Watch` resuming from any +/// cursor it already reached sees every later event exactly once, with no +/// gap and no duplicate, regardless of how it chooses to poll. Simulated by +/// repeatedly calling `SimLog::since` at every cursor the log itself +/// produced and stitching the results back together; cursor monotonicity +/// and gaplessness (the other half of invariant 7) hold by construction in +/// `SimLog` (each event is assigned `len() + 1` under the same lock as the +/// push), so this check is the one half worth stating as a property rather +/// than taking on faith. +pub fn check_watch_replay(log: &SimLog) -> Result<(), String> { + let all = log.entries(); + let mut replayed = Vec::with_capacity(all.len()); + let mut after = Cursor(0); + loop { + let batch = log.since(after); + if batch.is_empty() { + break; + } + after = batch.last().map(|(cursor, _)| *cursor).unwrap_or(after); + replayed.extend(batch); + } + if replayed != all { + return Err(format!( + "watch replay via repeated `since` calls produced {} events, the log itself holds {}", + replayed.len(), + all.len() + )); + } + Ok(()) +} diff --git a/vendor/dex-loop/tests/sim/scenario.rs b/vendor/dex-loop/tests/sim/scenario.rs new file mode 100644 index 000000000..94587ab61 --- /dev/null +++ b/vendor/dex-loop/tests/sim/scenario.rs @@ -0,0 +1,623 @@ +//! The host-action driver: a sequence of `Action`s interpreted against one +//! (or two, for the lease-fencing property) simulated replicas, checked with +//! `invariants::check_log` after every `Tick` and once more at quiescence. + +use std::collections::HashSet; +use std::time::Duration; + +use dex_loop::{ + ApprovalId, Budget, CallId, CancellationToken, Engine, Event, ExecutorKind, Exit, + GovernanceClass, Lexicon, Outcome, PrincipalId, ThreadId, ToolName, ToolSpec, TurnId, + rehydrate, +}; +use proptest::prelude::*; + +use super::fakes::{CrashBudget, SimEffects, SimLog, SimModel, SimTools, adversarial_payload}; +use super::invariants::{self, Violation}; + +type SimEngine = Engine; + +fn thread() -> ThreadId { + ThreadId { + org: "org-sim".into(), + workspace: "ws-sim".into(), + thread: "thread-sim".into(), + } +} + +fn budget() -> Budget { + Budget { + max_steps: 20, + max_tokens: u64::MAX, + max_cost_micros: u64::MAX, + wall: Duration::from_secs(5), + } +} + +/// alice and bob may decide approvals in these scenarios; mallory may not. +/// Kernel-level approvals carry no authorization check of their own (only +/// `call` + `approval` + `args_digest` are verified, against the durable +/// `ApprovalRequested` row -- see `dex_runtime::ingress::Ingress::approve`): +/// authorizing the *principal* is a host responsibility performed before an +/// `ApprovalDecided` is ever appended. This driver plays that host role, so +/// `ApprovalChoice::FromUnauthorized` exercises "the host correctly refuses +/// to forward it", not "the kernel independently checks and refuses it". +fn principal(index: u8) -> PrincipalId { + match index % 3 { + 0 => PrincipalId::new("alice"), + 1 => PrincipalId::new("bob"), + _ => PrincipalId::new("mallory"), + } +} + +fn authorized(principal: &PrincipalId) -> bool { + principal.as_str() != "mallory" +} + +fn catalog() -> Vec { + let spec = + |name: &str, read_only: bool, governance: GovernanceClass, executor: ExecutorKind| { + ToolSpec { + name: ToolName::new(name), + label: format!("Label for {name}"), + schema: serde_json::json!({"type": "object"}), + read_only, + core: true, + governance, + executor, + } + }; + vec![ + spec( + "reader", + true, + GovernanceClass::Plain, + ExecutorKind::InProcess, + ), + spec( + "mutator", + false, + GovernanceClass::Approval, + ExecutorKind::ToolExecutor, + ), + spec( + "client.read", + true, + GovernanceClass::Plain, + ExecutorKind::Client, + ), + spec( + "client.write", + false, + GovernanceClass::Approval, + ExecutorKind::Client, + ), + ] +} + +fn mutation_names() -> HashSet { + ["mutator", "client.write"] + .into_iter() + .map(String::from) + .collect() +} + +fn client_names() -> HashSet { + ["client.read", "client.write"] + .into_iter() + .map(String::from) + .collect() +} + +// ---------------------------------------------------------------- Action + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Payload { + Plain, + Nul, + Unicode, + Huge, +} + +impl Payload { + fn text(self, seed: u64) -> String { + let bucket = match self { + Payload::Plain => 0, + Payload::Nul => 1, + Payload::Unicode => 2, + Payload::Huge => 3, + }; + adversarial_payload(seed.wrapping_add(bucket)) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ApprovalChoice { + Correct, + Denied, + WrongDigest, + WrongApprovalId, + FromUnauthorized, + /// Decides correctly, then submits the identical decision again -- + /// duplicate delivery of the same `ApprovalDecided`. + Duplicate, +} + +#[derive(Clone, Debug)] +pub enum Action { + Send { + principal: u8, + payload: Payload, + /// Reuse the most recent turn id instead of minting a new one: + /// duplicate `Send` delivery, which a real host's ingress makes an + /// idempotent no-op (see `run_actions`'s `used_turns`). + reuse_turn: bool, + }, + Steer { + principal: u8, + payload: Payload, + }, + Interrupt { + principal: u8, + }, + Approve(ApprovalChoice), + Answer { + principal: u8, + payload: Payload, + }, + ClientResult { + principal: u8, + succeed: bool, + }, + VanishMutator, + RestoreMutator, + DenyMutator, + AllowMutator, + /// Arms a crash at the `n`th port operation of the *next* `Tick` only + /// (`n` taken mod a small cap so shrinking stays meaningful). + CrashAt(u8), + Tick, +} + +fn payload_strategy() -> impl Strategy { + prop_oneof![ + Just(Payload::Plain), + Just(Payload::Nul), + Just(Payload::Unicode), + Just(Payload::Huge), + ] +} + +fn approval_choice_strategy() -> impl Strategy { + prop_oneof![ + Just(ApprovalChoice::Correct), + Just(ApprovalChoice::Denied), + Just(ApprovalChoice::WrongDigest), + Just(ApprovalChoice::WrongApprovalId), + Just(ApprovalChoice::FromUnauthorized), + Just(ApprovalChoice::Duplicate), + ] +} + +/// One `Action`. `proptest`'s shrinker works on this directly: on failure it +/// drops and simplifies elements of the generated `Vec` until no +/// smaller trace still reproduces the violation. +pub fn action() -> impl Strategy { + prop_oneof![ + 3 => (0u8..3, payload_strategy(), any::()) + .prop_map(|(principal, payload, reuse_turn)| Action::Send { principal, payload, reuse_turn }), + 1 => (0u8..3, payload_strategy()).prop_map(|(principal, payload)| Action::Steer { principal, payload }), + 1 => (0u8..3).prop_map(|principal| Action::Interrupt { principal }), + 2 => approval_choice_strategy().prop_map(Action::Approve), + 1 => (0u8..3, payload_strategy()).prop_map(|(principal, payload)| Action::Answer { principal, payload }), + 1 => (0u8..3, any::()).prop_map(|(principal, succeed)| Action::ClientResult { principal, succeed }), + 1 => Just(Action::VanishMutator), + 1 => Just(Action::RestoreMutator), + 1 => Just(Action::DenyMutator), + 1 => Just(Action::AllowMutator), + 2 => (0u8..12).prop_map(Action::CrashAt), + 6 => Just(Action::Tick), + ] +} + +pub fn action_sequence() -> impl Strategy> { + proptest::collection::vec(action(), 1..30) +} + +// ---------------------------------------------------------------- Interpreter + +fn find_approval_request(log: &SimLog, approval: &ApprovalId) -> Option<(CallId, String)> { + log.entries() + .into_iter() + .rev() + .find_map(|(_, event)| match event { + Event::ApprovalRequested { + call, + approval: a, + args_digest, + .. + } if &a == approval => Some((call, args_digest)), + _ => None, + }) +} + +/// `true` once every `UserMessage` the log holds has its own terminal +/// (`Final`, `Error` or `Interrupted`): the ground truth this driver uses for +/// "nothing more will ever happen on its own", computed from the log alone +/// (not from `Context`, whose `status()`/`open_step()` are crate-private) so +/// it also catches the case where finishing turn A silently promotes an +/// already-queued turn B (`Context::begin_next_pending_turn`) in the same +/// `emit()` that ended A: B's `UserMessage` was already in the log, so the +/// message/terminal counts stay unequal until B gets its own terminal too, +/// even though the `Engine::run` call that ended A already returned +/// `Exit::Done`. +fn log_settled(log: &SimLog) -> bool { + let events = log.entries(); + let sent = events + .iter() + .filter(|(_, event)| matches!(event, Event::UserMessage { .. })) + .count(); + let terminals = events + .iter() + .filter(|(_, event)| { + matches!( + event, + Event::Final { .. } | Event::Error { .. } | Event::Interrupted + ) + }) + .count(); + sent == terminals +} + +/// One drive iteration: rehydrate fresh from the current log (never carry a +/// `Context` over between ticks -- `dex_runtime::actor::Runtime::drive` does +/// not either, since `Engine::run` only ever reads control-kind events +/// mid-flight, never `UserMessage`; a `Context` held across two separate +/// `run()` calls would silently miss a `Send` that landed in between), run +/// one attempt, and check invariant 8 (`rehydrate(log) == live state`) +/// against the resulting log. +async fn tick( + thread: &ThreadId, + log: &SimLog, + model: &SimModel, + tools: &SimTools, + effects: &SimEffects, + crash: CrashBudget, +) -> Option> { + let mut c = rehydrate(thread.clone(), &log.entries()); + let engine: SimEngine = Engine::new( + log.for_replica(log.generation(), crash.clone()), + model.clone(), + tools.with_crash(crash.clone()), + effects.for_replica(effects.fence().generation(), crash), + Lexicon::default(), + budget(), + ); + let cancel = CancellationToken::new(); + match tokio::time::timeout(Duration::from_secs(60), engine.run(&mut c, &cancel)).await { + Ok(result) => { + if let Ok(exit) = &result { + let full = rehydrate(thread.clone(), &log.entries()); + assert_eq!( + full, c, + "rehydrate(full log) must equal the live context right after a completed `Engine::run` \ + (exit {exit:?})" + ); + } + Some(result) + } + Err(_timeout) => { + // Simulated crash: whatever already landed in the log stays + // durable; `c`'s in-memory state (possibly behind the log by + // whatever this attempt never got to `observe`) is discarded. + None + } + } +} + +/// Interprets `actions` against one replica, then drains to quiescence +/// (bounded) and checks every invariant. `seed` drives the model's and +/// tools' adversarial choices (see `fakes.rs`); it is independent of the +/// `Action` sequence itself, so proptest shrinks the sequence without also +/// needing to shrink the seed. +pub async fn run_actions(seed: u64, actions: &[Action]) -> Vec { + let thread = thread(); + let crash_free = CrashBudget::none(); + let log = SimLog::new(crash_free.clone()); + let effects = SimEffects::new(crash_free.clone()); + let tools = SimTools::new(catalog(), seed, crash_free.clone()); + let model = SimModel::new(seed); + + let mut pending_exit: Option = None; + let mut used_turns: HashSet = HashSet::new(); + let mut turn_counter: u32 = 0; + let mut next_crash: Option = None; + let mut gave_up = false; + + for action in actions { + match action.clone() { + Action::Send { + principal: p, + payload, + reuse_turn, + } => { + let turn = if reuse_turn && turn_counter > 0 { + format!("t{turn_counter}") + } else { + turn_counter += 1; + format!("t{turn_counter}") + }; + if used_turns.insert(turn.clone()) { + log.host_append(Event::UserMessage { + turn: TurnId::new(turn), + message_id: None, + principal: principal(p), + text: payload.text(seed), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + } + // else: a real ingress makes a repeated turn id a no-op. + } + Action::Steer { + principal: p, + payload, + } => { + log.host_append(Event::Steer { + principal: principal(p), + text: payload.text(seed), + }); + } + Action::Interrupt { principal: p } => { + log.host_append(Event::Interrupt { + principal: principal(p), + }); + } + Action::Approve(choice) => { + if let Some(Exit::Parked(approval)) = &pending_exit + && let Some((call, requested_digest)) = find_approval_request(&log, approval) + { + let event = + |digest: String, approval: ApprovalId, who: PrincipalId, approved: bool| { + Event::ApprovalDecided { + call: call.clone(), + approval, + args_digest: digest, + approved, + principal: who, + } + }; + match choice { + ApprovalChoice::Correct => { + log.host_append(event( + requested_digest, + approval.clone(), + principal(0), + true, + )); + } + ApprovalChoice::Denied => { + log.host_append(event( + requested_digest, + approval.clone(), + principal(0), + false, + )); + } + ApprovalChoice::WrongDigest => { + log.host_append(event( + format!("{requested_digest}00"), + approval.clone(), + principal(0), + true, + )); + } + ApprovalChoice::WrongApprovalId => { + log.host_append(event( + requested_digest, + ApprovalId::new(format!("{approval}-bogus")), + principal(0), + true, + )); + } + ApprovalChoice::FromUnauthorized => { + // The host refuses to forward this: nothing + // appended (see the module doc on `principal`). + assert!( + !authorized(&principal(2)), + "mallory must not be authorized in this scenario" + ); + } + ApprovalChoice::Duplicate => { + log.host_append(event( + requested_digest.clone(), + approval.clone(), + principal(0), + true, + )); + log.host_append(event( + requested_digest, + approval.clone(), + principal(0), + true, + )); + } + } + } + } + Action::Answer { + principal: p, + payload, + } => { + if let Some(Exit::Asked(call)) = &pending_exit { + log.host_append(Event::Answer { + call: call.clone(), + principal: principal(p), + text: payload.text(seed), + }); + } + } + Action::ClientResult { + principal: p, + succeed, + } => { + if let Some(Exit::AwaitingClientTool(call)) = &pending_exit { + let outcome = if succeed { + Outcome::Succeeded + } else { + Outcome::Failed + }; + log.host_append(Event::ClientToolResult { + call: call.clone(), + principal: principal(p), + outcome, + output: "client result".into(), + }); + } + } + Action::VanishMutator => tools.vanish("mutator"), + Action::RestoreMutator => tools.restore("mutator"), + Action::DenyMutator => tools.force_deny("mutator"), + Action::AllowMutator => tools.clear_deny("mutator"), + Action::CrashAt(n) => next_crash = Some(n), + Action::Tick => { + let crash = next_crash + .take() + .map(|n| CrashBudget::at(n as u64)) + .unwrap_or_else(CrashBudget::none); + match tick(&thread, &log, &model, &tools, &effects, crash).await { + Some(Ok(exit)) => pending_exit = Some(exit), + Some(Err(_fenced)) => { + // Single replica in this property: the generation + // never moves, so a genuine `Fenced` would itself be + // a bug worth surfacing rather than swallowing. + pending_exit = None; + } + None => pending_exit = None, // crashed; rehydrate next time. + } + } + } + } + + // Drain to quiescence with no further host actions and no crash, + // bounded so a real livelock fails the test instead of hanging it. + let mut final_parked = false; + for _ in 0..64 { + if log_settled(&log) { + break; + } + match tick(&thread, &log, &model, &tools, &effects, CrashBudget::none()).await { + Some(Ok(exit)) => { + let parked = matches!( + exit, + Exit::Parked(_) | Exit::Asked(_) | Exit::AwaitingClientTool(_) + ); + pending_exit = Some(exit); + if parked { + final_parked = true; + break; + } + } + _ => { + gave_up = true; + break; + } + } + } + if !log_settled(&log) && !final_parked { + gave_up = true; + } + + let mut violations = invariants::check_log( + &log.entries(), + &tools.dispatches(), + &mutation_names(), + &client_names(), + ); + violations.extend(invariants::check_principal_attribution( + &thread, + &log.entries(), + )); + if let Err(message) = invariants::check_watch_replay(&log) { + violations.push(Violation::new(message)); + } + if gave_up { + violations.push(Violation::new(format!( + "drain loop did not reach quiescence within 64 ticks (last exit: {pending_exit:?})" + ))); + } else if !final_parked + && let Err(message) = invariants::check_quiescent_turn_count(&log.entries()) + { + violations.push(Violation::new(message)); + } + violations +} + +// ---------------------------------------------------------------- Bounded seeds + +/// Generates a small, seed-derived action sequence for the plain +/// seed-sweep entry point (`dst_bounded_seeds` / `DEX_SIM_SEEDS` soak): every +/// seed is reproducible on its own, without going through `proptest`. +pub fn actions_for_seed(seed: u64) -> Vec { + use rand::Rng; + use rand::SeedableRng; + let mut rng = rand::rngs::StdRng::seed_from_u64(seed); + let len = rng.gen_range(1..30); + let mut actions = Vec::with_capacity(len); + for _ in 0..len { + actions.push(match rng.gen_range(0..12) { + 0..=2 => Action::Send { + principal: rng.gen_range(0..3), + payload: [ + Payload::Plain, + Payload::Nul, + Payload::Unicode, + Payload::Huge, + ][rng.gen_range(0..4)], + reuse_turn: rng.gen_bool(0.2), + }, + 3 => Action::Steer { + principal: rng.gen_range(0..3), + payload: Payload::Plain, + }, + 4 => Action::Interrupt { + principal: rng.gen_range(0..3), + }, + 5 | 6 => Action::Approve( + [ + ApprovalChoice::Correct, + ApprovalChoice::Denied, + ApprovalChoice::WrongDigest, + ApprovalChoice::WrongApprovalId, + ApprovalChoice::FromUnauthorized, + ApprovalChoice::Duplicate, + ][rng.gen_range(0..6)], + ), + 7 => Action::Answer { + principal: rng.gen_range(0..3), + payload: Payload::Plain, + }, + 8 => Action::ClientResult { + principal: rng.gen_range(0..3), + succeed: rng.gen_bool(0.5), + }, + 9 => { + if rng.gen_bool(0.5) { + Action::VanishMutator + } else { + Action::RestoreMutator + } + } + 10 => Action::CrashAt(rng.gen_range(0..12)), + _ => Action::Tick, + }); + } + actions.push(Action::Tick); + actions +} + +pub async fn run_seed(seed: u64) -> Vec { + run_actions(seed, &actions_for_seed(seed)).await +} diff --git a/vendor/dex-loop/tests/support/mod.rs b/vendor/dex-loop/tests/support/mod.rs new file mode 100644 index 000000000..461179d92 --- /dev/null +++ b/vendor/dex-loop/tests/support/mod.rs @@ -0,0 +1,850 @@ +//! In-memory ports for scenario tests. + +use std::collections::{HashMap, VecDeque}; +use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::time::{Duration, Instant}; + +use dex_loop::{ + ApprovalId, Budget, CallId, CancellationToken, Claim, ClientToolSpec, Context, Cursor, Effects, + Engine, Entry, Event, ExecutorKind, Fenced, GovernanceClass, Lexicon, Log, Message, Model, + ModelChunk, ModelError, NoCompaction, Outcome, Output, OutputRef, PrincipalId, ProposedCall, + Summarize, ThreadId, ToolName, ToolResult, ToolSpec, Tools, TurnId, Usage, Verdict, rehydrate, +}; +use futures_util::{Stream, StreamExt, stream}; +use tokio::sync::Barrier; + +fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex.lock().unwrap_or_else(PoisonError::into_inner) +} + +pub fn thread() -> ThreadId { + ThreadId { + org: "org-1".into(), + workspace: "ws-1".into(), + thread: "thread-1".into(), + } +} + +pub fn alice() -> PrincipalId { + PrincipalId::new("alice") +} + +pub fn bob() -> PrincipalId { + PrincipalId::new("bob") +} + +pub fn call_id(turn: &str, step: u32, index: usize) -> CallId { + CallId::new(format!("{turn}-{step}-{index}")) +} + +pub type TestEngine = + Engine; + +pub fn engine(log: &FakeLog, model: &FakeModel, tools: &FakeTools, budget: Budget) -> TestEngine { + engine_with(log, model, tools, &FakeEffects::default(), budget) +} + +pub fn engine_with( + log: &FakeLog, + model: &FakeModel, + tools: &FakeTools, + effects: &FakeEffects, + budget: Budget, +) -> TestEngine { + Engine::new( + log.clone(), + model.clone(), + tools.clone(), + effects.clone(), + Lexicon::default(), + budget, + ) +} + +// ---------------------------------------------------------------- Log + +#[derive(Default)] +struct LogState { + events: Vec<(Cursor, Event)>, + text_writes: Vec, + /// Writes still allowed before every write is refused. + allowed_writes: Option, + refused: usize, +} + +impl LogState { + fn admit(&mut self) -> Result<(), Fenced> { + if let Some(allowed) = &mut self.allowed_writes { + if *allowed == 0 { + self.refused += 1; + return Err(Fenced::new("lease generation moved")); + } + *allowed -= 1; + } + Ok(()) + } + + fn push(&mut self, event: Event) -> Cursor { + let cursor = Cursor(self.events.len() as i64 + 1); + self.events.push((cursor, event)); + cursor + } +} + +/// Coalesces consecutive text into one `TextDelta` row, as a real log does. +#[derive(Clone, Default)] +pub struct FakeLog { + state: Arc>, +} + +impl FakeLog { + /// A host-side write (ingress). Never fenced. + pub fn host_append(&self, event: Event) -> Cursor { + lock(&self.state).push(event) + } + + /// Appends Alice's `UserMessage` and returns the rehydrated context. + pub fn start_turn(&self, turn: &str, text: &str) -> Context { + self.start_turn_with_client_tools(turn, text, Vec::new()) + } + + /// Appends Alice's `UserMessage`, declaring `client_tools`, and returns + /// the rehydrated context. + pub fn start_turn_with_client_tools( + &self, + turn: &str, + text: &str, + client_tools: Vec, + ) -> Context { + self.host_append(Event::UserMessage { + turn: TurnId::new(turn), + message_id: None, + principal: alice(), + text: text.into(), + attachments: Vec::new(), + client_tools, + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }); + self.rehydrate() + } + + /// Appends Alice's `UserMessage` under `approval_mode` and returns the + /// rehydrated context. + pub fn start_turn_with_approval_mode( + &self, + turn: &str, + text: &str, + approval_mode: dex_loop::ApprovalMode, + ) -> Context { + self.host_append(Event::UserMessage { + turn: TurnId::new(turn), + message_id: None, + principal: alice(), + text: text.into(), + attachments: Vec::new(), + client_tools: Vec::new(), + authorized_tools: Vec::new(), + approval_mode, + }); + self.rehydrate() + } + + /// The client's report for a `ClientToolRequested` call. + pub fn submit_tool_result(&self, call: &CallId, outcome: Outcome, output: &str) { + self.host_append(Event::ClientToolResult { + call: call.clone(), + principal: alice(), + outcome, + output: output.into(), + }); + } + + pub fn rehydrate(&self) -> Context { + rehydrate(thread(), &self.entries()) + } + + pub fn entries(&self) -> Vec<(Cursor, Event)> { + lock(&self.state).events.clone() + } + + pub fn events(&self) -> Vec { + self.entries().into_iter().map(|(_, event)| event).collect() + } + + pub fn shapes(&self) -> Vec { + self.events().iter().map(shape).collect() + } + + /// Shapes of the events after the first `skip`. + pub fn shapes_after(&self, skip: usize) -> Vec { + self.shapes().into_iter().skip(skip).collect() + } + + pub fn len(&self) -> usize { + lock(&self.state).events.len() + } + + /// Every `append_text` call, before coalescing. + pub fn text_writes(&self) -> Vec { + lock(&self.state).text_writes.clone() + } + + /// The engine may make `writes` more writes; later writes fail. + pub fn fence_after(&self, writes: usize) { + lock(&self.state).allowed_writes = Some(writes); + } + + pub fn refused(&self) -> usize { + lock(&self.state).refused + } + + /// The digest the engine put on the approval request for `call`. + pub fn requested_digest(&self, call: &CallId) -> String { + self.events() + .into_iter() + .find_map(|event| match event { + Event::ApprovalRequested { + call: requested, + args_digest, + .. + } if &requested == call => Some(args_digest), + _ => None, + }) + .unwrap_or_else(|| panic!("no approval requested for {call}")) + } + + pub fn decide(&self, call: &CallId, approval: &str, approved: bool) { + let args_digest = self.requested_digest(call); + self.host_append(Event::ApprovalDecided { + call: call.clone(), + approval: ApprovalId::new(approval), + args_digest, + approved, + principal: alice(), + }); + } +} + +impl Log for FakeLog { + async fn append(&self, events: &[Event]) -> Result, Fenced> { + let mut state = lock(&self.state); + state.admit()?; + Ok(events + .iter() + .map(|event| state.push(event.clone())) + .collect()) + } + + async fn append_text(&self, text: String) -> Result<(), Fenced> { + let mut state = lock(&self.state); + state.admit()?; + state.text_writes.push(text.clone()); + if let Some((_, Event::TextDelta { text: row })) = state.events.last_mut() { + row.push_str(&text); + } else { + state.push(Event::TextDelta { text }); + } + Ok(()) + } + + async fn control_since(&self, after: Cursor) -> Result, Fenced> { + Ok(lock(&self.state) + .events + .iter() + .filter(|(cursor, event)| *cursor > after && event.is_control()) + .cloned() + .collect()) + } +} + +// ---------------------------------------------------------------- Model + +pub fn text(text: &str) -> Result { + Ok(ModelChunk::Text(text.into())) +} + +pub fn call(name: &str, args: serde_json::Value) -> Result { + Ok(ModelChunk::ToolCall { + name: ToolName::new(name), + args, + }) +} + +pub fn usage( + input_tokens: u64, + output_tokens: u64, + cost_micros: u64, +) -> Result { + Ok(ModelChunk::Usage(Usage { + input_tokens, + output_tokens, + cost_micros, + })) +} + +#[derive(Default)] +struct ModelState { + scripts: VecDeque>>, + chunk_delay: Duration, + seen: Vec>, + offered: Vec>, +} + +/// Replays one script per model call and records what each call was sent. +#[derive(Clone, Default)] +pub struct FakeModel { + state: Arc>, +} + +impl FakeModel { + pub fn new(scripts: Vec>>) -> Self { + let model = Self::default(); + lock(&model.state).scripts = scripts.into(); + model + } + + pub fn with_chunk_delay(self, delay: Duration) -> Self { + lock(&self.state).chunk_delay = delay; + self + } + + /// The history of every model call, in order. + pub fn seen(&self) -> Vec> { + lock(&self.state).seen.clone() + } + + /// The tool names offered to every model call, in order. + pub fn offered(&self) -> Vec> { + lock(&self.state).offered.clone() + } + + pub fn calls(&self) -> usize { + lock(&self.state).seen.len() + } +} + +impl Model for FakeModel { + fn stream<'a>( + &'a self, + ctx: &'a Context, + tools: &'a [&'a ToolSpec], + ) -> impl Stream> + Send + 'a { + let mut state = lock(&self.state); + state.seen.push( + ctx.history() + .iter() + .map(|entry| entry.message.clone()) + .collect(), + ); + state + .offered + .push(tools.iter().map(|spec| spec.name.to_string()).collect()); + let script = state.scripts.pop_front().unwrap_or_else(|| { + vec![Err(ModelError { + message: "no script left".into(), + })] + }); + let delay = state.chunk_delay; + stream::iter(script).then(move |chunk| async move { + if !delay.is_zero() { + tokio::time::sleep(delay).await; + } + chunk + }) + } +} + +// ---------------------------------------------------------------- Tools + +pub fn read_tool(name: &str) -> ToolSpec { + spec(name, true, true, ExecutorKind::InProcess) +} + +pub fn write_tool(name: &str) -> ToolSpec { + spec(name, false, true, ExecutorKind::ToolExecutor) +} + +pub fn ask_tool(name: &str) -> ToolSpec { + spec(name, true, true, ExecutorKind::User) +} + +/// A tool a test's client session declares on `Send`, exactly as +/// `Context::client_tools` stores it (unresolved: the client's own claim). +/// Composing that into an actual `Tools::catalog()` entry is a host concern +/// (`dex_tools::client::declare` in dex-runtime); a dex-loop-level test that +/// wants a dispatchable `Client`-executor tool uses `client_executed_tool` +/// on `FakeTools` directly instead. +pub fn client_tool(name: &str, read_only: bool) -> ClientToolSpec { + ClientToolSpec { + name: ToolName::new(name), + schema: serde_json::json!({"type": "object"}), + read_only, + label: format!("Label for {name}"), + } +} + +/// A catalog entry with `ExecutorKind::Client`, as a host's `Tools::catalog` +/// would offer one already resolved from a client's declaration. +pub fn client_executed_tool(name: &str, read_only: bool) -> ToolSpec { + ToolSpec { + name: ToolName::new(name), + label: format!("Label for {name}"), + schema: serde_json::json!({"type": "object"}), + read_only, + core: true, + governance: if read_only { + GovernanceClass::Plain + } else { + GovernanceClass::Approval + }, + executor: ExecutorKind::Client, + } +} + +/// Not core: offered only after `tools.search` exposes it. +pub fn hidden_read_tool(name: &str) -> ToolSpec { + spec(name, true, false, ExecutorKind::ToolExecutor) +} + +fn spec(name: &str, read_only: bool, core: bool, executor: ExecutorKind) -> ToolSpec { + ToolSpec { + name: ToolName::new(name), + label: format!("Label for {name}"), + schema: serde_json::json!({"type": "object"}), + read_only, + core, + governance: GovernanceClass::Plain, + executor, + } +} + +#[derive(Clone, Debug)] +pub struct RunRecord { + pub call: CallId, + /// The thread `Tools::run` was called with. `CallId` alone is unique + /// only within a thread, so tests that check downstream uniqueness + /// assert on the pair. + pub thread: ThreadId, + pub args: serde_json::Value, + pub started: Instant, + pub finished: Instant, + pub cancelled: bool, +} + +type Hook = Arc; + +#[derive(Default)] +struct ToolState { + verdicts: HashMap, + principal_verdicts: HashMap<(String, PrincipalId), Verdict>, + searches: HashMap>, + delays: HashMap, + barrier: Option<(Arc, Vec)>, + on_run: Option, + runs: Vec, + policy_checks: Vec<(CallId, PrincipalId)>, + /// Every call `Tools::wrap_client_result` was asked to finish, in order. + wrapped: Vec, +} + +#[derive(Clone)] +pub struct FakeTools { + catalog: Arc<[ToolSpec]>, + state: Arc>, +} + +impl FakeTools { + pub fn new(catalog: Vec) -> Self { + Self { + catalog: catalog.into(), + state: Arc::default(), + } + } + + pub fn verdict(self, tool: &str, verdict: Verdict) -> Self { + self.set_verdict(tool, verdict); + self + } + + /// Changes policy for later checks (a grant revoked while parked). + pub fn set_verdict(&self, tool: &str, verdict: Verdict) { + lock(&self.state).verdicts.insert(tool.into(), verdict); + } + + /// Policy for `tool` when the call acts under `principal`. + pub fn verdict_for(self, tool: &str, principal: PrincipalId, verdict: Verdict) -> Self { + lock(&self.state) + .principal_verdicts + .insert((tool.into(), principal), verdict); + self + } + + pub fn search_result(self, query: &str, tools: &[&str]) -> Self { + lock(&self.state).searches.insert( + query.into(), + tools.iter().map(|t| ToolName::new(*t)).collect(), + ); + self + } + + /// Calls whose `args.key` equals `key` sleep for `delay`. + pub fn delay(self, key: &str, delay: Duration) -> Self { + lock(&self.state).delays.insert(key.into(), delay); + self + } + + /// Calls with these `args.key` values wait for each other before + /// sleeping. Serial dispatch would deadlock; the timeout turns that into + /// an error result. + pub fn barrier(self, keys: &[&str]) -> Self { + lock(&self.state).barrier = Some(( + Arc::new(Barrier::new(keys.len())), + keys.iter().map(|key| (*key).to_owned()).collect(), + )); + self + } + + pub fn on_run(self, hook: impl Fn(&ProposedCall) + Send + Sync + 'static) -> Self { + lock(&self.state).on_run = Some(Arc::new(hook)); + self + } + + pub fn runs(&self) -> Vec { + lock(&self.state).runs.clone() + } + + pub fn run_ids(&self) -> Vec { + self.runs().iter().map(|run| run.call.to_string()).collect() + } + + pub fn run_of(&self, call: &CallId) -> RunRecord { + self.runs() + .into_iter() + .find(|run| &run.call == call) + .unwrap_or_else(|| panic!("{call} never ran")) + } + + pub fn policy_checks(&self) -> Vec<(String, String)> { + lock(&self.state) + .policy_checks + .iter() + .map(|(call, principal)| (call.to_string(), principal.to_string())) + .collect() + } + + /// Every call id `wrap_client_result` was asked to finish, in order -- + /// the engine's own record of when it routed a client-reported result + /// through the host, whether live or on replay. + pub fn wrapped_calls(&self) -> Vec { + lock(&self.state).wrapped.clone() + } +} + +fn key(call: &ProposedCall) -> String { + call.args + .get("key") + .and_then(serde_json::Value::as_str) + .unwrap_or_default() + .to_owned() +} + +pub fn output_for(call: &CallId) -> ToolResult { + ToolResult::stored(OutputRef::new(format!("out/{call}")), None) +} + +impl Tools for FakeTools { + fn catalog(&self) -> &[ToolSpec] { + &self.catalog + } + + async fn search(&self, _principal: &PrincipalId, query: &str) -> Vec { + lock(&self.state) + .searches + .get(query) + .cloned() + .unwrap_or_default() + } + + async fn policy(&self, _ctx: &Context, call: &ProposedCall) -> Verdict { + let mut state = lock(&self.state); + state + .policy_checks + .push((call.id.clone(), call.principal.clone())); + let by_principal = (call.tool.to_string(), call.principal.clone()); + state + .principal_verdicts + .get(&by_principal) + .or_else(|| state.verdicts.get(call.tool.as_str())) + .cloned() + .unwrap_or(Verdict::Allow) + } + + async fn run( + &self, + thread: &ThreadId, + call: &ProposedCall, + cancel: &CancellationToken, + ) -> ToolResult { + let started = Instant::now(); + let key = key(call); + let (delay, barrier, hook) = { + let state = lock(&self.state); + let barrier = state + .barrier + .as_ref() + .filter(|(_, keys)| keys.contains(&key)) + .map(|(barrier, _)| Arc::clone(barrier)); + ( + state.delays.get(&key).copied().unwrap_or_default(), + barrier, + state.on_run.clone(), + ) + }; + if let Some(hook) = hook { + hook(call); + } + let mut result = output_for(&call.id); + let mut cancelled = false; + if let Some(barrier) = barrier + && tokio::time::timeout(Duration::from_secs(5), barrier.wait()) + .await + .is_err() + { + result = ToolResult::error("barrier timed out: calls did not overlap"); + } + tokio::select! { + () = tokio::time::sleep(delay) => {} + () = cancel.cancelled() => { + cancelled = true; + result = ToolResult::error("cancelled"); + } + } + lock(&self.state).runs.push(RunRecord { + call: call.id.clone(), + thread: thread.clone(), + args: call.args.clone(), + started, + finished: Instant::now(), + cancelled, + }); + result + } + + /// Marks that this call's result was wrapped, and rewrites a text output + /// so a test can tell a wrapped result from the client's raw one. + async fn wrap_client_result( + &self, + _thread: &ThreadId, + call: &ProposedCall, + raw: ToolResult, + ) -> ToolResult { + lock(&self.state).wrapped.push(call.id.clone()); + match raw.output { + Output::Text(text) => ToolResult { + output: Output::Text(format!("wrapped:{text}")), + ..raw + }, + output => ToolResult { output, ..raw }, + } + } +} + +// ---------------------------------------------------------------- Effects + +/// A ledger: claimed calls map to their recorded result, or `None` while the +/// outcome is not recorded. +#[derive(Clone, Default)] +pub struct FakeEffects { + ledger: Arc>>>, +} + +impl FakeEffects { + /// A claim from before a crash, with or without a recorded outcome. + pub fn seed(self, call: CallId, recorded: Option) -> Self { + lock(&self.ledger).insert(call, recorded); + self + } + + pub fn recorded(&self, call: &CallId) -> Option> { + lock(&self.ledger).get(call).cloned() + } +} + +impl Effects for FakeEffects { + async fn claim(&self, call: &ProposedCall) -> Result { + let mut ledger = lock(&self.ledger); + match ledger.get(&call.id) { + Some(Some(result)) => Ok(Claim::Existing(result.clone())), + Some(None) => Ok(Claim::Existing(ToolResult { + outcome: Outcome::Running, + output: Output::Text("dispatched; no outcome recorded yet".into()), + receipt: None, + })), + None => { + ledger.insert(call.id.clone(), None); + Ok(Claim::Granted) + } + } + } + + async fn record(&self, call: &CallId, result: &ToolResult) -> Result<(), Fenced> { + lock(&self.ledger).insert(call.clone(), Some(result.clone())); + Ok(()) + } +} + +// ---------------------------------------------------------------- Summarizer + +#[derive(Clone, Default)] +pub struct FakeSummarizer; + +impl Summarize for FakeSummarizer { + async fn summarize(&self, entries: &[Entry]) -> Option { + Some(format!("summary of {} entries", entries.len())) + } +} + +// ---------------------------------------------------------------- Assertions + +fn outcome(outcome: Outcome) -> &'static str { + match outcome { + Outcome::Succeeded => "ok", + Outcome::Failed => "err", + Outcome::Running => "running", + Outcome::Unknown => "unknown", + } +} + +fn ids(calls: &[ProposedCall]) -> String { + calls + .iter() + .map(|call| call.id.to_string()) + .collect::>() + .join(",") +} + +/// A compact, readable form of an event for sequence assertions. +pub fn shape(event: &Event) -> String { + match event { + Event::UserMessage { text, .. } => format!("user:{text}"), + Event::Steer { text, .. } => format!("steer:{text}"), + Event::Interrupt { .. } => "interrupt".into(), + Event::ApprovalDecided { call, approved, .. } => format!("decided:{call}:{approved}"), + Event::Answer { call, text, .. } => format!("answer:{call}:{text}"), + Event::StepStarted { step, .. } => format!("step:{step}"), + Event::TextDelta { text } => format!("delta:{text}"), + Event::Usage(usage) => format!("usage:{}", usage.tokens()), + Event::ModelStepCompleted { text, calls, .. } => { + format!("completed:{text}:[{}]", ids(calls)) + } + Event::ModelAttemptAbandoned { step } => format!("abandoned:{step}"), + Event::ToolStarted { call, .. } => format!("started:{call}"), + Event::ToolProgress { call, label } => format!("progress:{call}:{label}"), + Event::ToolsExposed { tools, .. } => format!( + "exposed:[{}]", + tools + .iter() + .map(ToString::to_string) + .collect::>() + .join(",") + ), + Event::ToolFinished { + call, + outcome: result, + .. + } => format!("finished:{call}:{}", outcome(*result)), + Event::ApprovalRequested { call, .. } => format!("approval:{call}"), + Event::Question { call, text } => format!("question:{call}:{text}"), + Event::ClientToolRequested { call, tool, .. } => format!("client_tool:{call}:{tool}"), + Event::ClientToolResult { + call, + outcome: result, + .. + } => format!("client_tool_result:{call}:{}", outcome(*result)), + Event::Compaction { summary, .. } => format!("compaction:{summary}"), + Event::Final { text } => format!("final:{text}"), + Event::Error { code, message } => format!("error:{}:{message}", code.as_str()), + Event::Interrupted => "interrupted".into(), + } +} + +/// A compact form of the model's view. +pub fn view(messages: &[Message]) -> Vec { + messages + .iter() + .map(|message| match message { + Message::User { text, .. } => format!("user:{text}"), + Message::Assistant { text, calls, .. } => { + format!("assistant:{text}:[{}]", ids(calls)) + } + Message::Tool { + call, + outcome: result, + output, + .. + } => { + let output = match output { + Output::Ref(reference) => reference.to_string(), + Output::Text(text) => text.clone(), + }; + format!("tool:{call}:{}:{output}", outcome(*result)) + } + Message::Summary { text } => format!("summary:{text}"), + }) + .collect() +} + +pub fn history(ctx: &Context) -> Vec { + view( + &ctx.history() + .iter() + .map(|entry| entry.message.clone()) + .collect::>(), + ) +} + +pub fn strings(items: &[&str]) -> Vec { + items.iter().map(|item| (*item).to_owned()).collect() +} + +pub fn approval(id: &str) -> Verdict { + Verdict::NeedsApproval { + approval: ApprovalId::new(id), + summary: format!("Approve {id}"), + } +} + +/// Log rows for a turn that crashed after `ToolStarted` for `call`. +pub fn crashed_after_start(log: &FakeLog, call: &ProposedCall) { + for event in [ + Event::UserMessage { + turn: TurnId::new("t1"), + message_id: None, + principal: alice(), + text: "do it".into(), + attachments: vec![], + client_tools: vec![], + authorized_tools: Vec::new(), + approval_mode: dex_loop::ApprovalMode::Interactive, + }, + Event::StepStarted { + step: 1, + control_through: Cursor::START, + }, + Event::ModelStepCompleted { + step: 1, + text: String::new(), + calls: vec![call.clone()], + reasoning: None, + }, + Event::ToolStarted { + call: call.id.clone(), + tool: call.tool.clone(), + label: format!("Label for {}", call.tool), + principal: call.principal.clone(), + }, + ] { + log.host_append(event); + } +} diff --git a/vendor/dex-loop/tests/tool_deadline.rs b/vendor/dex-loop/tests/tool_deadline.rs new file mode 100644 index 000000000..dd0dbebca --- /dev/null +++ b/vendor/dex-loop/tests/tool_deadline.rs @@ -0,0 +1,211 @@ +//! Regression: one tool call must not hold `Engine::run` open past the +//! engine's per-call deadline, and never past `budget.wall`. +//! +//! On 2026-09-29 every Dex computer turn in production stalled inside one +//! `dex.compute` call: tool-execution never dispatched the approved command, +//! and the engine sat in `Tools::run` with nothing bounding it (`budget.wall` +//! is only checked between steps; the mutation path runs with a cancel token +//! that never fires). The turn only ended when dex-tools' own 10-minute +//! remote wait gave up. These tests pin the engine's own bound: a call that +//! overruns is finished as `Failed` (a read: safe to retry) or `Unknown` (a +//! mutation: recorded in the ledger), the model gets that result on its next +//! step, and a call never outlives the wall budget. Paused virtual time keeps +//! them instant. + +// Each `tests/*.rs` file is its own crate; this one uses a subset of the +// shared helpers. +#[allow(dead_code)] +mod support; + +use std::time::Duration; + +use dex_loop::{Budget, CancellationToken, Exit, Outcome, Output}; +use serde_json::json; +use support::*; + +const NEVER: Duration = Duration::from_secs(60 * 60); + +fn budget_with_wall(wall: Duration) -> Budget { + Budget { + max_steps: 10, + max_tokens: u64::MAX, + max_cost_micros: u64::MAX, + wall, + } +} + +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn a_read_that_overruns_the_deadline_is_finished_failed_and_the_turn_continues() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("search", json!({"key": "slow"}))], + vec![text("done")], + ]); + let tools = FakeTools::new(vec![read_tool("search")]).delay("slow", NEVER); + let engine = engine(&log, &model, &tools, budget_with_wall(NEVER)) + .with_tool_call_deadline(Duration::from_millis(50)); + let mut ctx = log.start_turn("t1", "look"); + + let exit = tokio::time::timeout( + Duration::from_secs(5), + engine.run(&mut ctx, &CancellationToken::new()), + ) + .await + .expect("Engine::run must return once the call's deadline elapses"); + + assert_eq!(exit, Ok(Exit::Done)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "finished:t1-1-0:err", + "step:2", + "delta:done", + "completed:done:[]", + "final:done", + ]) + ); + // The model sees the timeout as this call's result, not a dangling call. + assert_eq!( + view(&model.seen()[1]), + strings(&[ + "user:look", + "assistant::[t1-1-0]", + "tool:t1-1-0:err:not completed: the call did not finish within Dex's time limit; it is safe to try again", + ]) + ); + assert!( + tools.runs().is_empty(), + "the overrunning read was dropped, not awaited to completion" + ); + assert_eq!(log.rehydrate(), ctx); +} + +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn a_mutation_that_overruns_the_deadline_is_recorded_unknown() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("update", json!({"key": "slow"}))], + vec![text("checked")], + ]); + let tools = FakeTools::new(vec![write_tool("update")]).delay("slow", NEVER); + let effects = FakeEffects::default(); + let engine = engine_with(&log, &model, &tools, &effects, budget_with_wall(NEVER)) + .with_tool_call_deadline(Duration::from_millis(50)); + let mut ctx = log.start_turn("t1", "change it"); + + let exit = tokio::time::timeout( + Duration::from_secs(5), + engine.run(&mut ctx, &CancellationToken::new()), + ) + .await + .expect("Engine::run must return once the call's deadline elapses"); + + assert_eq!(exit, Ok(Exit::Done)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "finished:t1-1-0:unknown", + "step:2", + "delta:checked", + "completed:checked:[]", + "final:checked", + ]) + ); + // The ledger holds the same answer, so a resume adopts it instead of + // dispatching the mutation again. + let recorded = effects + .recorded(&call_id("t1", 1, 0)) + .flatten() + .expect("the timed-out mutation is recorded under its claim"); + assert_eq!(recorded.outcome, Outcome::Unknown); + assert_eq!( + recorded.output, + Output::Text( + "outcome unknown: the call did not finish within Dex's time limit; check whether it took effect before trying again" + .into() + ) + ); + assert_eq!(log.rehydrate(), ctx); +} + +// Real time, not paused: the loop's between-steps wall check reads a +// `std::time::Instant`, which tokio's paused clock does not advance. 100ms +// of real time is the whole cost; the stuck call is dropped, never awaited. +#[tokio::test(flavor = "current_thread")] +async fn a_call_never_outlives_the_wall_budget() { + // The per-call deadline is generous; the wall budget is not. The call is + // cut at the wall, its result appended, and the turn ends on the wall + // budget -- with the call's result in history, so the thread stays well + // formed for the next turn. + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("search", json!({"key": "slow"}))], + vec![text("never reached")], + ]); + let tools = FakeTools::new(vec![read_tool("search")]).delay("slow", NEVER); + let engine = engine( + &log, + &model, + &tools, + budget_with_wall(Duration::from_millis(100)), + ) + .with_tool_call_deadline(NEVER); + let mut ctx = log.start_turn("t1", "look"); + + let exit = tokio::time::timeout( + Duration::from_secs(5), + engine.run(&mut ctx, &CancellationToken::new()), + ) + .await + .expect("Engine::run must return once budget.wall elapses, even mid-call"); + + assert_eq!(exit, Ok(Exit::Failed)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "finished:t1-1-0:err", + "error:budget_exhausted:wall budget exhausted: 100ms", + ]) + ); + assert_eq!(model.seen().len(), 1, "no model step after the wall budget"); +} + +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn a_call_that_finishes_in_time_is_unaffected() { + let log = FakeLog::default(); + let model = FakeModel::new(vec![ + vec![call("search", json!({"key": "quick"}))], + vec![text("done")], + ]); + let tools = FakeTools::new(vec![read_tool("search")]).delay("quick", Duration::from_millis(10)); + let engine = engine(&log, &model, &tools, budget_with_wall(NEVER)) + .with_tool_call_deadline(Duration::from_millis(50)); + let mut ctx = log.start_turn("t1", "look"); + + let exit = engine.run(&mut ctx, &CancellationToken::new()).await; + + assert_eq!(exit, Ok(Exit::Done)); + assert_eq!(tools.run_ids(), strings(&["t1-1-0"])); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "completed::[t1-1-0]", + "started:t1-1-0", + "finished:t1-1-0:ok", + "step:2", + "delta:done", + "completed:done:[]", + "final:done", + ]) + ); +} diff --git a/vendor/dex-loop/tests/wall_budget.rs b/vendor/dex-loop/tests/wall_budget.rs new file mode 100644 index 000000000..ccf4a604d --- /dev/null +++ b/vendor/dex-loop/tests/wall_budget.rs @@ -0,0 +1,125 @@ +//! Regression: `budget.wall` must bound the whole `Engine::run` call, +//! including time spent waiting on the model stream, not only the time +//! between steps. +//! +//! Before the fix this accompanies, a model stream that yields chunks and +//! then never resolves again (no error, no `None`, and never cancelled) hung +//! `Engine::run` forever: the budget check in the outer loop only runs +//! *between* `model_step` calls, and the inner chunk-reading loop raced the +//! stream only against `cancel`, never against the wall deadline. This test +//! fails (times out) on the pre-fix engine and passes on the fixed one, +//! using paused virtual time so it costs no real wall-clock time either way. + +// This test only needs a handful of `support` helpers; the rest are dead +// code from this test binary's own point of view (each `tests/*.rs` file is +// its own crate), even though `scenarios.rs` uses the whole module. +#[allow(dead_code)] +mod support; + +use std::time::Duration; + +use dex_loop::{ + Budget, CancellationToken, Engine, Event, Exit, Lexicon, ModelChunk, ModelError, PrincipalId, +}; +use futures_util::{Stream, StreamExt, stream}; +use support::*; + +/// A model that yields text and usage, then a stream that never produces another +/// item and never ends -- the adversarial "stalled stream" case: no error, no +/// natural end, and (in this test) no interrupt ever arrives either. +#[derive(Clone, Default)] +struct StallingModel; + +impl dex_loop::Model for StallingModel { + fn stream<'a>( + &'a self, + _ctx: &'a dex_loop::Context, + _tools: &'a [&'a dex_loop::ToolSpec], + ) -> impl Stream> + Send + 'a { + stream::iter([ + Ok(ModelChunk::Text("partial answer".into())), + usage(2, 3, 5), + ]) + .chain(stream::pending()) + } +} + +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn a_stalled_model_stream_is_bounded_by_the_wall_budget() { + let log = FakeLog::default(); + let tools = FakeTools::new(vec![]); + let engine = Engine::new( + log.clone(), + StallingModel, + tools, + FakeEffects::default(), + Lexicon::default(), + budget_with_wall(Duration::from_millis(50)), + ); + let mut ctx = log.start_turn("t1", "hi"); + let cancel = CancellationToken::new(); + + // Paused virtual time auto-advances once the run future is the only + // thing left to make progress on, so this resolves instantly in real + // time regardless of the 50ms budget; before the fix it never resolves + // at all (the inner loop has no deadline), and this `timeout` -- itself + // generous relative to the 50ms budget -- is what turns that hang into a + // failing test rather than one that never finishes. + let result = tokio::time::timeout(Duration::from_secs(5), engine.run(&mut ctx, &cancel)) + .await + .expect("Engine::run must return once budget.wall elapses, even mid-stream"); + + assert_eq!(result, Ok(Exit::Failed)); + assert_eq!( + log.shapes_after(1), + strings(&[ + "step:1", + "delta:partial answer", + "usage:5", + "abandoned:1", + "error:budget_exhausted:wall budget exhausted: 50ms" + ]) + ); +} + +#[tokio::test(flavor = "current_thread", start_paused = true)] +async fn an_interrupt_still_wins_a_race_with_the_wall_deadline() { + // A wall deadline far in the future must not change ordinary interrupt + // behavior: cancelling still ends the turn as `Interrupted`, not as a + // spurious budget failure. + let log = FakeLog::default(); + let tools = FakeTools::new(vec![]); + let engine = Engine::new( + log.clone(), + StallingModel, + tools, + FakeEffects::default(), + Lexicon::default(), + budget_with_wall(Duration::from_secs(3600)), + ); + let mut ctx = log.start_turn("t1", "hi"); + let cancel = CancellationToken::new(); + + let run = engine.run(&mut ctx, &cancel); + tokio::pin!(run); + // Give the stream a chance to yield its chunks before interrupting. + tokio::task::yield_now().await; + log.host_append(Event::Interrupt { + principal: PrincipalId::new("alice"), + }); + cancel.cancel(); + + let result = tokio::time::timeout(Duration::from_secs(5), run) + .await + .expect("an interrupted stream must not hang"); + assert_eq!(result, Ok(Exit::Interrupted)); +} + +fn budget_with_wall(wall: Duration) -> Budget { + Budget { + max_steps: 10, + max_tokens: u64::MAX, + max_cost_micros: u64::MAX, + wall, + } +}