diff --git a/docs/SURFACE.md b/docs/SURFACE.md index 460f1d27f..4a86b7384 100644 --- a/docs/SURFACE.md +++ b/docs/SURFACE.md @@ -561,6 +561,48 @@ flows resume [--json] [--no-spawn] [--data-dir ] flows observer [--data-dir ] ``` +### Agent sidechannel (initial byte-stream slice) + +Local agent workers (`flows run --local-agent`) open +`/runs//steps//pty.sock` for the raw Claude/Codex +adapters and print `PTY ` on stderr. SDK workers opt in with `dataDir` +and can receive the path through `onPtyReady`. Run and step IDs come from the +kernel dispatch, including for authored `f.agent` calls. + +A subscriber sends `HELLO view\n`, `HELLO drive\n`, or +`HELLO passthrough\n`, then receives live stdout/stderr bytes. View and +passthrough are passive. Only drive forwards subsequent bytes to child stdin. +In this pipe-based slice, a drive greeting must arrive within 100ms of child +startup. Without one, the worker closes stdin so unattended and passive-view +agents receive EOF. Later drive greetings are rejected without marking human +intervention; a closed stdin pipe cannot be reopened. Supporting drive attachment +at arbitrary times requires a future terminal/session transport. Drive readers +pause while child stdin writes flush, preserving input under backpressure. +There is no backlog, terminal resize, or framing after the greeting. Socket +access is restricted to the worker's OS user. Slow, malformed, excess, and +broken subscribers are disconnected independently; absent subscribers or an +unavailable socket do not fail a step or change its lease/deadline handling. + +Operator input stays off journal. Influenced output follows the usual worker +output decoding and completion path. Any accepted drive greeting marks +`step.completed.human_intervention: true`, even without input. Passive and +unattached completions omit the field. The marker persists on failed attempts +as well, and marked completions are excluded from cross-run memoization. + +`flows resume` refuses marked runs with exit 2 and +`REFUSED [human_influenced_run] step ""` before recovery, unless passed +`--allow-human-influenced`. The same flag permits `flows replay` to cross a +marked completion; replay may already have printed earlier entries when it +refuses. `--at` can still inspect a prefix before that boundary. The flag is +per invocation and does not clear the journal marker. + +**Follow-up scope:** this minimal slice forwards the existing process pipes; +it does not yet allocate an actual terminal. True PTY/resize support, +wrapper-session attachment, crash-safe intervention recording before a worker +completion, and the companion `agent-relay attach --external-pty` client are +follow-ups. Long UNIX socket paths and stale socket files disable this optional +channel; they do not prevent agent execution. + `flows observer` prints a single `https://agentrelay.com/observer?key=` URL to stdout using the same mint used by `flows run`. It is daemon-free: no socket is opened, no `relayflowd` binary is invoked, the data dir is not diff --git a/docs/evidence/spec-Q/README.md b/docs/evidence/spec-Q/README.md new file mode 100644 index 000000000..1df21400f --- /dev/null +++ b/docs/evidence/spec-Q/README.md @@ -0,0 +1,72 @@ +# Slice Q verification + +## PR #338 Bugbot follow-up + +Rebased onto `origin/main` at `494f2a11`. Unattended, passive, and incomplete +subscribers now allow stdin EOF after a 100ms startup drive-attachment window. +Late drive greetings are rejected without setting the intervention marker. +Drive input pauses its socket until each child write completes, including +writes that exceed the pipe's buffer capacity. Arbitrary-time drive attachment +remains a terminal/session transport follow-up, documented in SURFACE.md. + +After SDK `npm ci --ignore-scripts` and repository-root +`npm install ./packages/surface --prefix packages/sdk --no-save --ignore-scripts`, +ran from `packages/sdk`: +`npm run typecheck && ./node_modules/.bin/vitest run tests/worker-cli.test.ts tests/pty-sidechannel.test.ts`. +Exit 0; literal command and output: [bugbot-fixes.txt](bugbot-fixes.txt). +This is focused SDK regression evidence, not a new live-provider acceptance run. + +## Original slice evidence + +This is a minimal byte-stream proof for #334, with completion markers and +resume/replay protection. Actual PTY allocation, wrapper-session attachment, +crash-safe intervention recording before completion, and the companion +agent-relay external-pty client remain follow-ups (see SURFACE.md §5). + +Run/step IDs already arrive in AgentWorker's kernel dispatch; no redundant +identity plumbing was added to authored-worker-step.ts. This tree has no +separate resume.ts: resume lives in cli/run.ts and argument parsing in cli.ts. + +All commands ran in this worktree on 2026-09-11. Each linked file is captured +command output. Exit statuses below refer to the completed tool processes. + +## SDK + +From packages/sdk: + +- `npm run typecheck` — exit 0, [output](typecheck.txt). +- `npm run typecheck:tests` — exit 0, [output](typecheck-tests.txt). +- `npm run build` — exit 0, [output](build.txt). +- `npx vitest run tests/pty-sidechannel.test.ts tests/worker-cli.test.ts tests/worker-cli-abort.test.ts tests/cli-replay.test.ts` + — exit 0, 57 passed, [output](sdk-focused.txt). +- `npx vitest run tests/cli.test.ts -t 'human-influenced resume'` + — exit 0, 1 passed / 63 excluded by filter, [output](resume-cli.txt). +- `RELAYFLOWD_BIN=/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/relayflowd npx vitest run` + — exit 1, [initial output](sdk-initial.txt). Bun was absent from PATH, + the live Claude analyzer was unavailable, and one live-kernel test timed out. +- `PATH=/Users/khaliqgant/.bun/bin:$PATH RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 RELAYFLOWD_BIN=/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/relayflowd npx vitest run` + — exit 0, 1215 passed / 4 skipped, [final output](sdk-final.txt). + This is not gate-2/live-analyzer acceptance evidence. The analyzer skip is + explicit; the initial unskipped failure is retained above. + +Dependencies: SDK `npm ci --ignore-scripts`, then root +`npm install ./packages/surface --prefix packages/sdk --no-save --ignore-scripts`. +The local surface needed a build; its pre-existing lockfile was out of sync, +so it was installed with `npm install --ignore-scripts --package-lock=false` +and built with `npm run build`. No dependency manifests/locks changed. + +## Rust + +From the repository root unless otherwise noted: + +- `PATH=/Users/khaliqgant/.cargo/bin:$PATH ops/cargo.sh test --manifest-path kernel/Cargo.toml --workspace` + — exit 0, [output](rust-workspace.txt), including the new durable marker, + refusal, unchanged-journal-on-refusal, and explicit-override test. +- From kernel: `PATH=/Users/khaliqgant/.cargo/bin:$PATH ../ops/cargo.sh build --locked --release -p relayflowd` + — exit 0, [output](rust-release.txt). +- `PATH=/Users/khaliqgant/.cargo/bin:$PATH ops/cargo.sh test --manifest-path kernel/Cargo.toml -p relayflowd-core --test memoization` + — exit 0, [output](rust-memoization.txt). This final focused run includes + the added assertion excluding human-influenced completions from reuse. + +The initial cargo invocation lacked rustc on PATH; the commands above use the +existing toolchain. `git diff --check` also exited 0. diff --git a/docs/evidence/spec-Q/bugbot-fixes.txt b/docs/evidence/spec-Q/bugbot-fixes.txt new file mode 100644 index 000000000..22422b857 --- /dev/null +++ b/docs/evidence/spec-Q/bugbot-fixes.txt @@ -0,0 +1,36 @@ +$ cd packages/sdk +$ npm run typecheck && ./node_modules/.bin/vitest run tests/worker-cli.test.ts tests/pty-sidechannel.test.ts + +> @relayflows/sdk@2.0.8 typecheck +> tsc --noEmit && tsc -p tsconfig.type-tests.json + + + RUN v2.1.9 /Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk + + ✓ tests/pty-sidechannel.test.ts (11 tests) 5016ms + ✓ view attach preserves worker completion and marks only drive 1048ms + ✓ passthrough attach preserves worker completion and marks only drive 800ms + ✓ none attach preserves worker completion and marks only drive 707ms + ✓ none subscriber lets an unattended CLI read EOF 311ms + ✓ rejects drive after EOF without marking human intervention 678ms + ✓ delivers all drive bytes in order across child stdin backpressure 661ms + ✓ tests/worker-cli.test.ts (13 tests) 24222ms + ✓ custom wrapper execution identity > passes an explicit safe environment at identification and execution 381ms + ✓ custom wrapper execution identity > refuses a wrapper symlink retarget before delivering private values 520ms + ✓ custom wrapper execution identity > bounds wrapper execution after acknowledgement 360ms + ✓ custom wrapper execution identity > refuses a duplicate execute protocol frame 400ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a conforming wrapper leaks a stdio pipe to a background helper 1864ms + ✓ custom wrapper execution bounds are reader-owned > resolves when the leaked helper inherits stderr only 2009ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a wrapper leaks a stdio pipe and exits before identifying 3280ms + ✓ custom wrapper execution bounds are reader-owned > journals a completionReason at the default bound when a wrapper leaks a stdio pipe 11264ms + ✓ custom wrapper execution bounds are reader-owned > accepts an execute token and an over-8KiB payload flushed in one write 1289ms + ✓ custom wrapper execution bounds are reader-owned > accepts the same over-8KiB payload whether or not it coalesces with the execute token 896ms + ✓ custom wrapper execution bounds are reader-owned > still bounds an un-terminated handshake buffer and names the bound 377ms + ✓ delivers the journaled memory pack to the real wrapper and excludes its charge from completion usage 1297ms + + Test Files 2 passed (2) + Tests 24 passed (24) + Start at 21:10:32 + Duration 24.48s (transform 50ms, setup 0ms, collect 81ms, tests 29.24s, environment 0ms, prepare 75ms) + +Exit status: 0 diff --git a/docs/evidence/spec-Q/build.txt b/docs/evidence/spec-Q/build.txt new file mode 100644 index 000000000..d4aad672d --- /dev/null +++ b/docs/evidence/spec-Q/build.txt @@ -0,0 +1,4 @@ + +> @relayflows/sdk@2.0.8 build +> tsc && node scripts/make-cli-executable.mjs + diff --git a/docs/evidence/spec-Q/resume-cli.txt b/docs/evidence/spec-Q/resume-cli.txt new file mode 100644 index 000000000..9b8eea726 --- /dev/null +++ b/docs/evidence/spec-Q/resume-cli.txt @@ -0,0 +1,10 @@ + + RUN v2.1.9 /Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk + + ✓ tests/cli.test.ts (64 tests | 63 skipped) 8ms + + Test Files 1 passed (1) + Tests 1 passed | 63 skipped (64) + Start at 20:32:05 + Duration 1.18s (transform 450ms, setup 0ms, collect 792ms, tests 8ms, environment 0ms, prepare 71ms) + diff --git a/docs/evidence/spec-Q/rust-memoization.txt b/docs/evidence/spec-Q/rust-memoization.txt new file mode 100644 index 000000000..cf949166c --- /dev/null +++ b/docs/evidence/spec-Q/rust-memoization.txt @@ -0,0 +1,38 @@ + Compiling syn v3.0.4 + Compiling zerovec-derive v0.11.6 + Compiling displaydoc v0.2.7 + Compiling serde_derive v1.0.229 + Compiling ref-cast-impl v1.0.27 + Compiling thiserror-impl v2.0.20 + Compiling ref-cast v1.0.27 + Compiling zerovec v0.11.8 + Compiling zerotrie v0.2.5 + Compiling thiserror v2.0.20 + Compiling serde v1.0.229 + Compiling tinystr v0.8.4 + Compiling potential_utf v0.1.6 + Compiling icu_collections v2.3.0 + Compiling icu_locale_core v2.3.0 + Compiling icu_provider v2.3.1 + Compiling ahash v0.8.12 + Compiling fluent-uri v0.3.2 + Compiling email_address v0.2.9 + Compiling ulid v1.2.1 + Compiling icu_normalizer v2.3.0 + Compiling icu_properties v2.3.0 + Compiling referencing v0.33.0 + Compiling idna_adapter v1.2.2 + Compiling idna v1.1.0 + Compiling jsonschema v0.33.0 + Compiling relayflowd-core v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd-core) + Finished `test` profile [unoptimized + debuginfo] target(s) in 7.60s + Running tests/memoization.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/memoization-3560adb32830fb1c) + +running 4 tests +test distinct_large_kernel_integers_do_not_alias_through_float_rounding ... ok +test match_reuses_output_with_provenance_and_zero_cost_without_dispatch ... ok +test changed_spec_or_input_dispatches_and_legacy_or_failed_records_miss ... ok +test canonical_corpus_agrees_with_typescript_and_key_permutations ... ok + +test result: ok. 4 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s + diff --git a/docs/evidence/spec-Q/rust-release.txt b/docs/evidence/spec-Q/rust-release.txt new file mode 100644 index 000000000..53a1c1bf0 --- /dev/null +++ b/docs/evidence/spec-Q/rust-release.txt @@ -0,0 +1,127 @@ + Compiling proc-macro2 v1.0.107 + Compiling unicode-ident v1.0.24 + Compiling quote v1.0.47 + Compiling libc v0.2.189 + Compiling stable_deref_trait v1.2.1 + Compiling version_check v0.9.5 + Compiling autocfg v1.5.1 + Compiling cfg-if v1.0.4 + Compiling serde_core v1.0.229 + Compiling getrandom v0.3.4 + Compiling zerocopy v0.8.56 + Compiling smallvec v1.15.2 + Compiling serde v1.0.229 + Compiling litemap v0.8.3 + Compiling num-traits v0.2.19 + Compiling memchr v2.8.3 + Compiling writeable v0.6.4 + Compiling generic-array v0.14.7 + Compiling utf8_iter v1.0.4 + Compiling icu_normalizer_data v2.3.0 + Compiling icu_properties_data v2.3.0 + Compiling parking_lot_core v0.9.12 + Compiling syn v3.0.4 + Compiling syn v2.0.119 + Compiling typenum v1.20.1 + Compiling zmij v1.0.23 + Compiling ref-cast v1.0.27 + Compiling synstructure v0.13.2 + Compiling zerofrom-derive v0.1.7 + Compiling yoke-derive v0.8.2 + Compiling num-integer v0.1.47 + Compiling zerofrom v0.1.8 + Compiling num-bigint v0.4.8 + Compiling aho-corasick v1.1.5 + Compiling ahash v0.8.12 + Compiling zerovec-derive v0.11.6 + Compiling displaydoc v0.2.7 + Compiling serde_derive v1.0.229 + Compiling ref-cast-impl v1.0.27 + Compiling serde_json v1.0.151 + Compiling scopeguard v1.2.0 + Compiling find-msvc-tools v0.1.11 + Compiling shlex v2.0.1 + Compiling regex-syntax v0.8.11 + Compiling num-rational v0.4.2 + Compiling cc v1.4.4 + Compiling lock_api v0.4.14 + Compiling ppv-lite86 v0.2.21 + Compiling yoke v0.8.3 + Compiling num-iter v0.1.46 + Compiling num-complex v0.4.6 + Compiling rand_core v0.9.5 + Compiling pkg-config v0.3.34 + Compiling once_cell v1.21.4 + Compiling regex-automata v0.4.18 + Compiling bit-vec v0.8.0 + Compiling borrow-or-share v0.2.4 + Compiling vcpkg v0.2.15 + Compiling itoa v1.0.18 + Compiling bit-set v0.8.0 + Compiling num v0.4.3 + Compiling rand_chacha v0.9.0 + Compiling parking_lot v0.12.5 + Compiling libsqlite3-sys v0.35.0 + Compiling crypto-common v0.1.7 + Compiling block-buffer v0.10.4 + Compiling percent-encoding v2.3.2 + Compiling zerovec v0.11.8 + Compiling zerotrie v0.2.5 + Compiling uuid v1.26.0 + Compiling fluent-uri v0.3.2 + Compiling vsimd v0.8.0 + Compiling foldhash v0.1.5 + Compiling thiserror v2.0.20 + Compiling outref v0.5.2 + Compiling lazy_static v1.5.0 + Compiling utf8parse v0.2.2 + Compiling fraction v0.15.4 + Compiling anstyle-parse v1.0.0 + Compiling uuid-simd v0.8.0 + Compiling referencing v0.33.0 + Compiling tinystr v0.8.4 + Compiling potential_utf v0.1.6 + Compiling icu_locale_core v2.3.0 + Compiling icu_collections v2.3.0 + Compiling hashbrown v0.15.5 + Compiling email_address v0.2.9 + Compiling digest v0.10.7 + Compiling fancy-regex v0.16.2 + Compiling regex v1.13.1 + Compiling rand v0.9.5 + Compiling thiserror-impl v2.0.20 + Compiling cpufeatures v0.2.17 + Compiling is_terminal_polyfill v1.70.2 + Compiling colorchoice v1.0.5 + Compiling icu_provider v2.3.1 + Compiling icu_normalizer v2.3.0 + Compiling icu_properties v2.3.0 + Compiling num-cmp v0.1.0 + Compiling base64 v0.22.1 + Compiling anstyle v1.0.14 + Compiling bytecount v0.6.9 + Compiling anstyle-query v1.1.5 + Compiling anstream v1.0.0 + Compiling ulid v1.2.1 + Compiling idna_adapter v1.2.2 + Compiling sha2 v0.10.9 + Compiling idna v1.1.0 + Compiling jsonschema v0.33.0 + Compiling hashlink v0.10.0 + Compiling fallible-iterator v0.3.0 + Compiling bitflags v2.13.1 + Compiling clap_lex v1.1.0 + Compiling fallible-streaming-iterator v0.1.9 + Compiling ryu-js v1.0.3 + Compiling anyhow v1.0.104 + Compiling strsim v0.11.1 + Compiling heck v0.5.0 + Compiling clap_builder v4.6.6 + Compiling clap_derive v4.6.4 + Compiling wait-timeout v0.2.1 + Compiling relayflowd-core v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd-core) + Compiling clap v4.6.6 + Compiling rusqlite v0.37.0 + Compiling relayflowd-journal v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd-journal) + Compiling relayflowd v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd) + Finished `release` profile [optimized] target(s) in 34.01s diff --git a/docs/evidence/spec-Q/rust-workspace.txt b/docs/evidence/spec-Q/rust-workspace.txt new file mode 100644 index 000000000..60b6dd604 --- /dev/null +++ b/docs/evidence/spec-Q/rust-workspace.txt @@ -0,0 +1,391 @@ + Compiling relayflowd-core v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd-core) + Compiling relayflowd-journal v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd-journal) + Compiling relayflowd v0.1.0 (/Users/khaliqgant/flows-spec-Q-sidechannel/kernel/relayflowd) + Finished `test` profile [unoptimized + debuginfo] target(s) in 5.36s + Running unittests src/lib.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/relayflowd-331e4e97bba410b4) + +running 41 tests +test engine::remote::worker_failure_detail_tests::a_null_or_blank_output_yields_no_detail ... ok +test engine::remote::worker_failure_detail_tests::a_string_output_is_carried_verbatim_and_trimmed ... ok +test engine::remote::worker_failure_detail_tests::an_output_at_the_boundary_is_not_truncated ... ok +test engine::remote::worker_failure_detail_tests::a_non_string_output_is_rendered_rather_than_dropped ... ok +test engine::remote::worker_failure_detail_tests::truncation_does_not_split_a_multi_byte_char ... ok +test server::client::tests::resume_waits_while_the_heartbeat_renewed_lease_is_live ... ok +test engine::boot_identity_tests::every_engine_in_this_process_shares_one_boot_id ... ok +test server::liveness::tests::sweep_id_buckets_by_the_interval ... ok +test server::channels::tests::unknown_verb_never_falls_through_to_receive ... ok +test exec_det::tests::timeout_has_an_explicit_completion_reason ... ok +test exec_det::tests::captures_deterministic_output ... ok +test server::liveness::tests::sweep_pass_healthy_subscription_is_a_noop ... ok +test server::tests::agent::contract::an_agent_worker_attaching_without_pins_is_refused_at_attach ... ok +test server::liveness::tests::sweep_pass_latches_after_journaling_and_next_bucket_is_empty ... ok +test engine::wake::claim_guard_tests::a_disarmed_guard_leaves_the_claim_alone ... ok +test server::tests::agent::contract::an_oversized_trajectory_tail_is_refused_at_step_complete ... ok +test engine::wake::claim_guard_tests::a_panic_between_claim_and_register_still_releases ... ok +test engine::wake::claim_guard_tests::an_armed_guard_releases_the_claim_when_dropped ... ok +test engine::wake::claim_guard_tests::a_guard_only_releases_its_own_run ... ok +test server::tests::agent::contract::agent_without_a_compatible_worker_parks_without_starting ... ok +test server::tests::agent::contract::an_agent_worker_missing_a_declared_surface_parks_the_run_instead_of_erroring ... ok +test server::tests::hello_enforces_protocol_version ... ok +test server::tests::agent::contract::an_llm_completion_claiming_an_effect_fails_closed_with_the_reason_journaled ... ok +test server::tests::agent::pins::reset_worker_reporting_a_revision_other_than_its_pin_fails_closed_as_worker_error ... ok +test server::tests::a_failed_disconnect_journal_append_is_retained_and_retried_not_dropped ... ok +test server::tests::run_resume_asks_the_registry_instead_of_treating_an_orphan_file_as_a_run ... ok +test server::tests::run_start_fails_closed_on_an_unknown_verification_key ... ok +test server::tests::agent::contract::a_replacement_worker_that_never_reported_the_pinned_surface_is_not_dispatched_to ... ok +test socket_path::tests::deep_data_dir_produces_short_socket_path ... ok +test socket_path::tests::different_data_dirs_yield_different_sockets ... ok +test socket_path::tests::relative_and_absolute_data_dirs_agree ... ok +test socket_path::tests::same_data_dir_yields_same_socket ... ok +test server::tests::run_resume_adopts_a_real_journal_whose_registry_row_is_missing ... ok +test server::tests::run_resume_refuses_a_journal_that_never_recorded_its_run ... ok +test server::tests::agent::contract::human_intervention_is_durable_and_resume_requires_explicit_override ... ok +test server::tests::run_resume_refuses_a_valid_journal_that_belongs_to_another_run ... ok +test server::tests::agent::pins::consecutive_agent_steps_on_different_surfaces_each_start_from_their_own_pins ... ok +test server::tests::stopped_heartbeats_past_the_deadline_journal_lease_expired_and_release_the_step ... ok +test exec_det::tests::timeout_kills_the_whole_process_group ... ok +test server::tests::agent::pins::a_replacement_worker_at_a_different_revision_is_not_dispatched_the_stale_pins ... ok +test server::tests::an_entry_appended_during_watch_registration_is_delivered_exactly_once ... ok + +test result: ok. 41 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.56s + + Running unittests src/main.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/relayflowd-85d3a2820e0dfd5d) + +running 0 tests + +test result: ok. 0 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s + + Running tests/budget_gate.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/budget_gate-e9bdf0830b0a7d74) + +running 3 tests +test daily_windows_reset_and_exact_limits_do_not_refuse ... ok +test crossing_completion_is_durable_and_next_step_is_refused ... ok +test deterministic_spend_and_wallclock_limit_gate_parallel_batch_starts ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.03s + + Running tests/crash_resume.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/crash_resume-f00a2c4b4059afb2) + +running 40 tests +test agent::resume_without_a_worker_parks_immediately_instead_of_timing_out ... ok +test concurrency::cancel_closes_the_lease_and_rejects_a_late_completion ... ok +test agent::rung_c_sigkill_after_final_effect_replays_results_without_redispatch ... ok +test channels::channels_reject_foreign_workers_stale_attempts_and_invalid_acknowledgements ... ok +test concurrency::cancel_and_completion_race_has_one_terminal_fact ... ok +test agent::rung_c_reset_sigkill_mid_edit_restores_pins_dedupes_effect_and_explains_attempts ... ok +test agent::rung_c_sigkill_between_agent_completion_and_final_effect_memoizes_the_agent ... ok +test agent::rung_c_crash_between_effect_election_and_the_provider_call_performs_it_exactly_once ... ok +test concurrency::concurrent_resumes_lease_exactly_one_attempt ... ok +test concurrency::live_resume_leaves_an_active_lease_running ... ok +test concurrency::run_start_dispatches_every_independent_lane_before_any_completion ... ok +test llm::serve_plumbs_watch_events_and_replayable_stream_verbs ... ok +test concurrency::server_restart_recovers_every_parallel_lease_without_duplicate_success ... ok +test llm::llm_verification_exhaustion_is_a_declared_failure_kind ... ok +test llm::failing_llm_verification_schedules_a_durable_retry_and_succeeds ... ok +test agent::rung_c_sigkill_boundaries_resume_only_unfinished_steps_via_real_cli ... ok +test llm::completed_llm_output_is_memoized_when_serve_dies_during_the_next_step ... ok +test llm::sigkill_after_the_final_rung_b_effect_resumes_without_redispatching_llm ... ok +test memory::memory_sigkill_after_injection_replays_pack_and_charges_it_once ... ok +test llm::worker_killed_while_holding_a_lease_is_explained_and_released_on_cli_resume ... ok +test llm::sigkill_under_serve_mid_llm_releases_the_lease_and_finishes_via_cli_resume ... ok +test pin_projection::rejected_completion_cannot_forge_inspect_retry_pins_over_the_real_socket ... ok +test protocol_admission::every_mutating_run_verb_refuses_terminal_before_changing_state ... ok +test parallel_lifecycle::terminal_failure_drains_or_explains_every_live_sibling ... ok +test llm::sigkill_sweep_covers_before_and_between_the_rung_b_steps ... ok +test sigkill_mid_step_replaces_and_explains_the_dead_attempt ... ok +test sigkill_after_cancel_request_resumes_to_one_canceled_fact ... ok +test parallel_lifecycle::overlapping_agent_conflict_survives_server_crash_and_resume ... ok +test surface_identity::aliases_are_rejected_and_external_ancestors_serialize_over_real_sockets ... ok +test sigkill_under_serve_resumes_the_socket_started_run ... ok +test parallel_lifecycle::overlapping_agent_lanes_serialize_while_disjoint_lanes_merge_in_either_order ... ok +test worker_capacity::two_workers_receive_a_deterministic_fair_capacity_bounded_batch ... ok +test worker_capacity::default_capacity_one_reopens_only_after_durable_completion_or_crash ... ok +test workspace_identity::workspace_aliases_are_refused_and_canonical_subtrees_serialize_over_real_sockets ... ok +test channels::channels_sigkill_resume_redelivers_unacked_messages_with_exactly_once_effects ... ok +test placement::declared_placement_keeps_one_source_tree_across_resume ... ok +test sigkill_sweep_covers_every_hello_step_boundary ... ok +test placement::sigkill_before_first_step_preserves_the_submitted_workspace ... ok +test placement::sigkill_mid_step_keeps_the_route_and_source_tree ... ok +test parallel_lifecycle::renewed_parallel_leases_survive_the_original_grant_and_remain_distinct ... ok + +test result: ok. 40 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 37.37s + + Running tests/daemon_lifecycle.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/daemon_lifecycle-ba84c0d831e365f3) + +running 6 tests +test a_second_serve_on_a_served_data_dir_refuses_and_the_first_keeps_serving ... ok +test connection_file_is_published_only_after_the_socket_is_live ... ok +test clean_shutdown_removes_advertisement_and_socket ... ok +test sigkill_leaves_a_stale_file_with_a_dead_pid ... ok +test deep_data_dir_still_binds ... ok +test a_sigkilled_daemons_successor_starts_cleanly ... ok + +test result: ok. 6 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.10s + + Running tests/event_wake.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/event_wake-e0c98dffa0a59265) + +running 3 tests +test two_racing_deliveries_of_one_event_produce_exactly_one_run ... ok +test matching_event_wakes_once_with_fresh_context ... ok +test a_resumed_run_dispatches_the_original_wake_context ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.06s + + Running tests/hn_monitor_integration.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/hn_monitor_integration-19a18ad56a4e0130) + +running 1 test +test hn_story_event_wakes_monitor_once_with_story_context ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.04s + + Running tests/input_binding.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/input_binding-e5900c2c958c1bff) + +running 2 tests +test binding_schema_is_additive_and_fails_closed ... ok +test sigkill_before_consumer_resolves_original_journal_output_without_reexecuting_source ... ok + +test result: ok. 2 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.12s + + Running tests/invalid_schema_preflight.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/invalid_schema_preflight-fb4088ae31bca2af) + +running 3 tests +test invalid_json_schema_is_refused_before_journal_or_command ... ok +test unbounded_json_schema_is_refused_before_journal_or_command ... ok +test legitimately_recursive_json_schema_still_starts ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 3.89s + + Running tests/memoization.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/memoization-b8378bf177dc4caa) + +running 3 tests +test refuses_missing_wrong_flow_and_unreadable_journal_before_creating_run ... ok +test reused_prefix_survives_restart_without_source_and_never_mutates_prior ... ok +test actual_changed_input_invalidates_consumer_even_with_identical_consumer_spec ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.08s + + Running tests/memory.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/memory-2dd999428343e3d3) + +running 5 tests +test rejected_journal_fact_releases_reservation_and_never_dispatches ... ok +test over_budget_and_provider_errors_fail_without_dispatch_or_charge ... ok +test llm_dispatch_receives_same_pack_after_resume_without_provider ... ok +test replay_and_resume_need_no_provider_and_script_receives_recorded_pack ... ok +test semantic_retry_reuses_memory_without_a_second_charge ... ok + +test result: ok. 5 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.07s + + Running tests/memory_epoch.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/memory_epoch-1e598c87f73b4d4d) + +running 1 test +test epoch_carries_pack_and_exact_charge_and_refuses_duplicate_injection ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.02s + + Running tests/parallel_driver.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/parallel_driver-09d57fda35fc3047) + +running 4 tests +test stop_after_one_holds_for_an_independent_deterministic_batch ... ok +test pause_before_second_independent_step_holds_the_driver_boundary ... ok +test backpressured_or_mismatched_lane_does_not_drop_a_later_dispatch ... ok +test crash_boundaries_resume_the_real_driver_with_one_effect_per_lane ... ok + +test result: ok. 4 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.08s + + Running tests/placement_pins.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/placement_pins-d1876318727129de) + +running 3 tests +test unsupported_local_pty_is_refused_before_an_earlier_step_can_run ... ok +test default_worker_pins_the_declared_worktree_base_commit_and_refuses_missing_source ... ok +test a_resumed_attempt_keeps_the_original_pin_after_the_worktree_head_moves ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.51s + + Running tests/placement_routing.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/placement_routing-a5decc4af0d1cbe8) + +running 3 tests +test a_failed_routing_append_never_starts_or_dispatches_work ... ok +test crash_between_routing_and_start_does_not_redecide ... ok +test worker_retry_consumes_the_original_routing_fact ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.02s + + Running tests/routing_diagnostics.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/routing_diagnostics-3fd00989f6b06778) + +running 2 tests +test duplicate_routes_have_a_distinct_diagnostic_and_leave_the_original_fact_intact ... ok +test malformed_routes_name_the_same_field_at_append_replay_and_epoch_replay ... ok + +test result: ok. 2 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.01s + + Running tests/spec_review_routing.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/spec_review_routing-48229baf3a3fa052) + +running 4 tests +test attempt_scoped_route_is_rejected_at_append_and_replay ... ok +test epoch_cannot_drop_or_replace_a_durable_route ... ok +test malformed_epoch_routes_are_rejected_before_commit ... ok +test workspace_pin_peels_tags_and_refuses_non_commit_objects ... ok + +test result: ok. 4 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.41s + + Running tests/subscription_liveness.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/subscription_liveness-b2bd7d0fb2d4da77) + +running 3 tests +test submit_event_upserts_subscription_row_and_sweep_flags_it_stale_after_budget ... ok +test stale_transition_is_journaled_as_subscription_stale_entry_in_the_last_known_run ... ok +test a_fresh_arrival_re_arms_the_latch_and_the_next_silence_can_stale_again ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.02s + + Running tests/trigger_watcher.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/trigger_watcher-dcffffe2d25ba6f1) + +running 3 tests +test retains_bad_and_unregistered_events_while_consuming_filter_nonmatches ... ok +test failed_archive_retries_the_same_durable_run ... ok +test journals_payload_and_filename_key_then_archives_and_dedupes_replay ... ok + +test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.06s + + Running unittests src/lib.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/relayflowd_core-315e59794aff65b0) + +running 60 tests +test clock::tests::simulated_clock_is_explicitly_advanced ... ok +test journal::tests::memory_journal_assigns_sequences_and_rolls_epochs ... ok +test channel::tests::malformed_payloads_and_invalid_new_channel_appends_leave_state_unchanged ... ok +test channel::tests::forged_deliveries_and_acknowledgements_fail_closed ... ok +test channel::tests::send_retry_is_stable_and_conflicting_content_is_rejected ... ok +test channel::tests::delivery_replay_and_independent_acknowledged_offsets ... ok +test machine::parallel_tests::every_declared_mutable_surface_participates_in_conflict_selection ... ok +test machine::parallel_tests::external_ancestor_and_descendant_paths_conflict_but_siblings_do_not ... ok +test machine::parallel_tests::machine_starts_every_runnable_step_in_authored_order ... ok +test machine::tests::all_backing_off_steps_return_timers ... ok +test machine::parallel_tests::failed_run_drains_open_siblings_before_terminal_entry ... ok +test machine::tests::cancel_request_closes_the_active_lease_before_the_terminal_fact ... ok +test machine::tests::every_reason_label_matches_its_serialized_form ... ok +test machine::parallel_tests::parallel_lanes_do_not_cross_the_dependency_barrier_early ... ok +test machine::tests::durable_cancel_request_outranks_crash_recovery ... ok +test machine::parallel_tests::workspace_ancestor_and_descendant_paths_conflict_but_siblings_do_not ... ok +test machine::parallel_tests::overlapping_agent_surfaces_are_serialized_in_authored_order ... ok +test machine::parallel_tests::disjoint_agent_lanes_merge_pins_in_either_completion_order ... ok +test machine::tests::repeated_cancel_request_is_idempotent ... ok +test machine::tests::successful_memo_is_never_scheduled_again ... ok +test machine::tests::machine_starts_runnable_step_with_stable_effect_key ... ok +test machine::parallel_tests::crash_resume_preserves_each_parallel_lease_exactly_once ... ok +test machine::tests::verification_failure_schedules_a_durable_retry ... ok +test machine::tests::worker_reported_failure_without_detail_still_records_a_verification ... ok +test machine::tests::crashed_attempt_does_not_consume_an_iteration ... ok +test machine::tests::reset_recovery_dispatches_the_original_pinned_revision ... ok +test memory::tests::caps_compare_exact_decimals_and_each_token_dimension ... ok +test retry::tests::jitter_is_repeatable_and_bounded ... ok +test machine::tests::inspect_recovery_injects_the_dirty_pin_completion_reason_and_tail ... ok +test machine::tests::manual_recovery_parks_needs_human_and_never_redispatches ... ok +test machine::tests::every_failed_run_terminates_with_declared_completion_reasons ... ok +test schema::tests::in_document_uri_references_resolve_to_the_node_they_name ... ok +test spec::tests::a_misspelled_step_level_key_is_a_parse_error ... ok +test schema::tests::refusal_names_the_cycle_it_found ... ok +test spec::tests::a_misspelled_verification_gate_key_is_a_parse_error_not_a_dropped_gate ... ok +test spec::tests::cycles_are_rejected ... ok +test spec::tests::preflight_data_is_fail_closed ... ok +test spec::tests::spec_version_is_semver_and_gated ... ok +test spec::tests::external_surface_paths_must_have_one_canonical_spelling ... ok +test spec::tests::unknown_root_and_nested_fields_are_rejected ... ok +test spec::tests::workspace_mounts_and_worktrees_must_have_one_canonical_spelling ... ok +test spec::tests::zero_agent_flow_is_valid ... ok +test state::budget::tests::adds_costs_exactly_beyond_machine_decimal_precision ... ok +test state::budget::tests::overflow_and_malformed_cost_leave_total_unchanged ... ok +test schema::tests::references_the_bound_leaves_opaque_are_refused_by_the_engine ... ok +test state::tests::budget_decimal_strings_add_without_floats ... ok +test state::tests::a_completion_that_omits_a_surface_does_not_drop_it_from_the_pin_chain ... ok +test state::tests::journal_replays_data_gate_verdict_without_rerunning_completed_code ... ok +test state::tests::end_pin_chain_is_enforced_and_a_broken_chain_is_a_hard_error ... ok +test verify::tests::an_unbounded_schema_in_a_journal_fails_its_gate_instead_of_aborting ... ok +test verify::tests::deterministic_output_requires_successful_exit_and_content ... ok +test spec::tests::the_full_ladder_parses_in_the_one_dialect ... ok +test verify::tests::json_schema_is_a_control_gate ... ok +test schema::tests::a_property_named_ref_is_not_a_reference ... ok +test schema::tests::shared_declarations_and_boolean_schemas_are_validated ... ok +test schema::tests::every_accepted_corpus_schema_is_accepted ... ok +test schema::tests::every_refused_corpus_schema_compiles_but_is_refused_by_the_bound ... ok +test spec::tests::sdk_boundary_rejects_a_10_000_step_cycle_with_a_typed_error ... ok +test spec::tests::sdk_boundary_accepts_a_valid_10_000_step_reverse_chain ... ok +test schema::tests::deeply_nested_schemas_do_not_overflow_the_checker ... ok + +test result: ok. 60 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.60s + + Running tests/memoization.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/memoization-5083c536bd91f99b) + +running 4 tests +test distinct_large_kernel_integers_do_not_alias_through_float_rounding ... ok +test match_reuses_output_with_provenance_and_zero_cost_without_dispatch ... ok +test changed_spec_or_input_dispatches_and_legacy_or_failed_records_miss ... ok +test canonical_corpus_agrees_with_typescript_and_key_permutations ... ok + +test result: ok. 4 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s + + Running tests/spec_parity.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/spec_parity-fe4dc0a4eb6ad67f) + +running 9 tests +test the_kernel_parses_the_event_triggered_spec_and_stamps_the_same_hash ... ok +test step_memory_has_identical_canonical_bytes_and_hash ... ok +test the_kernel_parses_the_deterministic_rung_and_stamps_the_same_hash ... ok +test placement_requirements_have_identical_canonical_bytes_and_hash ... ok +test memory_declaration_acceptance_matches_the_sdk_corpus ... ok +test the_kernel_parses_the_rung_c_agent_spec_and_stamps_the_same_hash ... ok +test placement_declaration_acceptance_matches_the_sdk_corpus ... ok +test the_kernel_parses_the_sdk_compiled_spec_and_stamps_the_same_hash ... ok +test the_kernel_parses_the_rung_b_spec_and_stamps_the_same_hash ... ok + +test result: ok. 9 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.02s + + Running unittests src/lib.rs (/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/deps/relayflowd_journal-1220f3d7dcad415d) + +running 28 tests +test registry::tests::a_registered_run_dedupes_across_boots ... ok +test registry::tests::a_same_boot_claim_with_no_run_yet_is_a_duplicate_not_wreckage ... ok +test registry::tests::registry_is_a_rebuildable_run_locator ... ok +test registry::tests::releasing_a_claim_lets_the_same_boot_retry ... ok +test registry::tests::a_previous_boots_claim_with_no_run_is_repaired ... ok +test registry::tests::a_pre_migration_registry_gains_boot_id_and_its_claims_are_repairable ... ok +test channel::tests::stale_attempts_and_raw_forged_acknowledgements_cannot_change_offsets ... ok +test subscriptions::tests::last_run_for_subscription_returns_none_before_first_arrival ... ok +test registry::tests::releasing_is_scoped_to_the_claiming_run ... ok +test subscriptions::tests::detect_without_latch_stays_available_for_the_next_sweep ... ok +test subscriptions::tests::prune_sweep_claims_deletes_only_rows_older_than_cutoff ... ok +test channel::tests::channels_cross_segment_boundaries_and_terminal_runs_reject_mutations ... ok +test subscriptions::tests::latch_is_a_no_op_if_a_fresh_event_arrived_between_detect_and_latch ... ok +test subscriptions::tests::last_run_for_subscription_returns_the_lex_greatest_ulid_regardless_of_insertion ... ok +test subscriptions::tests::sweep_ignores_subscriptions_whose_silence_is_still_within_budget ... ok +test subscriptions::tests::sweep_election_gives_the_first_caller_the_result_and_second_gets_empty ... ok +test subscriptions::tests::sweep_does_not_re_emit_the_same_stale_row_on_a_later_tick ... ok +test subscriptions::tests::sweep_marks_row_stale_when_silence_exceeds_budget ... ok +test subscriptions::tests::upsert_after_stale_re_arms_and_next_silence_can_re_emit ... ok +test tests::failed_commit_is_returned_not_swallowed ... ok +test tests::append_is_durable_and_monotonic_after_reopen ... ok +test subscriptions::tests::upsert_is_idempotent_across_bumps_and_preserves_event_type_updates ... ok +test channel::tests::failed_channel_writes_never_expose_delivery_or_advance_acknowledged_offset ... ok +test tests::effects_are_deduplicated_at_the_journal_boundary ... ok +test tests::terminal_run_refuses_every_later_entry_atomically ... ok +test tests::an_unconfirmed_election_is_reclaimed_by_the_next_attempt_not_treated_as_done ... ok +test tests::rollover_is_atomic_scaffolding_for_epoch_resume ... ok +test channel::tests::independent_connections_serialize_send_receive_and_acknowledgement ... ok + +test result: ok. 28 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.13s + + Doc-tests relayflowd + +running 0 tests + +test result: ok. 0 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s + + Doc-tests relayflowd_core + +running 0 tests + +test result: ok. 0 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s + + Doc-tests relayflowd_journal + +running 0 tests + +test result: ok. 0 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s + diff --git a/docs/evidence/spec-Q/sdk-final.txt b/docs/evidence/spec-Q/sdk-final.txt new file mode 100644 index 000000000..37729d723 --- /dev/null +++ b/docs/evidence/spec-Q/sdk-final.txt @@ -0,0 +1,199 @@ + + RUN v2.1.9 /Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk + +stdout | tests/live-kernel.test.ts +LIVE_KERNEL relayflowd=/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/relayflowd +LIVE_KERNEL flows=/Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk/dist/cli.js + + ✓ tests/daemon-lifecycle.test.ts (42 tests) 30ms + ✓ tests/validate.test.ts (68 tests) 15ms + ✓ tests/preflight.test.ts (27 tests) 52ms + ✓ tests/observer-link.test.ts (39 tests) 241ms + ✓ tests/tick-source.test.ts (33 tests) 19ms + ✓ tests/journal-client.test.ts (14 tests) 79ms + ✓ tests/authored-flow.test.ts (25 tests) 675ms + ✓ tests/verb-field-lint.test.ts (78 tests) 393ms + ✓ tests/cloud-run.test.ts (47 tests) 333ms + ✓ tests/authored-flow-lifecycle-executor.test.ts (27 tests) 544ms + ✓ tests/gate-contract.test.ts (20 tests) 89ms + ✓ tests/cli-replay.test.ts (37 tests) 749ms + ✓ flows replay > --json is byte-identical across two CLI invocations (diff) 513ms + ✓ tests/cli-hn-monitor.test.ts (16 tests) 100ms + ✓ tests/tick-runner.test.ts (22 tests) 1520ms + ✓ tests/authored-flow-slack.test.ts (6 tests) 1079ms + ✓ authored Slack helper effects > replays after SIGKILL before confirm with the same token and one successful completion 354ms + ✓ authored Slack helper effects > replays after SIGKILL before complete with the same token and one successful completion 317ms + ✓ tests/backlog-picker.test.ts (14 tests) 60ms + ✓ tests/backlog-picker-flow.test.ts (6 tests) 497ms + ✓ tests/bundle.test.ts (21 tests) 3909ms + ✓ immutable bundles > builds and verifies the canonical YAML fixture through the compiled CLI 629ms + ✓ immutable bundles > emits the ephemeral warning on CLI stderr and uses the default output directory 395ms + ✓ immutable bundles > builds a standalone TS fixture twice with identical executable hashes 1190ms + ✓ tests/authored-flow-operation.test.ts (23 tests) 270ms + ✓ tests/mcp.test.ts (30 tests) 7068ms + ✓ MCP preflight and transports > flows check refuses an undeclared server with exit 2 and no daemon 323ms + ✓ MCP preflight and transports > flows check reports a refusing server and leaves no PID 552ms + ✓ MCP preflight and transports > kills a SIGTERM-resistant silent child after a parent-owned handshake deadline 1310ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with inherit stdio before cleanup finishes 1183ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with ignore stdio before cleanup finishes 2058ms + ✓ authored MCP effects against the real kernel > reports a dropped tool connection as a failed CLI run 362ms + ✓ tests/work-package-consumer.test.ts (13 tests) 184ms + ✓ tests/spec-parity.test.ts (31 tests) 178ms + ✓ tests/stop-process-group.test.ts (6 tests) 6726ms + ✓ every stop reaches the process group, not just the direct child > exits the run after an execution-timeout stop 1025ms + ✓ every stop reaches the process group, not just the direct child > exits the run after a protocol terminate stop 434ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after a protocol terminate stop 1667ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after an execution-timeout stop 2135ms + ✓ every stop reaches the process group, not just the direct child > holds the loop open long enough for the escalation to run 1155ms + ✓ every stop reaches the process group, not just the direct child > terminate() forces a group that outlives SIGTERM 309ms + ✓ tests/worker-lease.test.ts (7 tests) 8ms + ✓ tests/typed-output.test.ts (14 tests) 105ms + ✓ tests/json-schema-bound.test.ts (71 tests) 1474ms + ✓ JSON Schema termination bound > walks a deep schema with an explicit stack rather than recursion 1247ms + ✓ tests/mcp-lifecycle.test.ts (4 tests) 6ms + ✓ tests/model-selection.test.ts (10 tests) 9ms + ✓ tests/flow-executor-chain.test.ts (12 tests) 8369ms + ✓ flow executor LLM and output-binding chain > runs f.llm -> f.agent -> f.run with schema-verified journal output and the exact allowed model 930ms + ✓ flow executor LLM and output-binding chain > fails invalid LLM output before the next step: not JSON 540ms + ✓ flow executor LLM and output-binding chain > fails invalid LLM output before the next step: {"message":7} 535ms + ✓ flow executor LLM and output-binding chain > retains tagged-template text output 628ms + ✓ flow executor LLM and output-binding chain > preserves JSON values without promoting them to process wrappers: null 549ms + ✓ flow executor LLM and output-binding chain > preserves JSON values without promoting them to process wrappers: [1,2] 596ms + ✓ flow executor LLM and output-binding chain > preserves JSON values without promoting them to process wrappers: "hello" 586ms + ✓ flow executor LLM and output-binding chain > runs the authored LLM path through the built flows CLI 982ms + ✓ flow executor LLM and output-binding chain > passes a declarative verified value through an agent into a deterministic artifact 617ms + ✓ flow executor LLM and output-binding chain > journals a missing optional field as a failure before the consuming command executes 567ms + ✓ flow executor LLM and output-binding chain > flows run consumes YAML bindings and resume reuses the original journal output 1806ms + ✓ tests/relayflowd-path.test.ts (10 tests) 2ms + ✓ tests/f-memory.test.ts (7 tests) 738ms + ✓ tests/webhook.test.ts (6 tests) 52ms + ✓ tests/direct-input.test.ts (4 tests) 5334ms + ✓ direct .flow.ts input through the built CLI and live runtime > executes inline and file JSON input through relayflowd 2166ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses missing and malformed input before contacting relayflowd 1739ms + ✓ direct .flow.ts input through the built CLI and live runtime > does not run the authored body before daemon availability 780ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses oversized file input before contacting relayflowd 648ms + ✓ tests/local-dev-ux.test.ts (8 tests) 16ms + ✓ tests/daemon-lifecycle-live.test.ts (9 tests) 10185ms + ✓ flows run against a data dir with no daemon (§6 test 7) > cold start spawns exactly one daemon, the run succeeds, and the daemon outlives the CLI 635ms + ✓ flows run against a data dir with no daemon (§6 test 7) > polls, bounded, for a daemon that holds the lock before it binds 1416ms + ✓ flows run against a data dir with no daemon (§6 test 7) > attaches to a serving daemon that has not published a connection file 631ms + ✓ flows run against a data dir with no daemon (§6 test 7) > a second run attaches to the daemon the first one started, spawning nothing 1936ms + ✓ flows run against a data dir with no daemon (§6 test 7) > detects a stale connection file left by a hard kill and starts a fresh daemon 2037ms + ✓ concurrent invocations against one empty data dir (§6 test 15) > ends with exactly one daemon owning the socket, and both runs succeed 1162ms + ✓ refusals from a spawn that cannot produce a daemon > names relayflowd_not_found rather than falling through to PATH 1068ms + ✓ refusals from a spawn that cannot produce a daemon > names daemon_start_failed and quotes the daemon log when startup dies 648ms + ✓ refusals from a spawn that cannot produce a daemon > refuses a daemon speaking another protocol version instead of binding over it 652ms + ✓ tests/dependency-validation.test.ts (6 tests) 288ms + ✓ tests/cli.test.ts (64 tests) 12650ms + ✓ flows check CLI > binds a checked relative wrapper to the flow directory for worker execution 417ms + ✓ flows check CLI > uses Codex login status and reports a rejected model as unavailable, not unauthenticated 334ms + ✓ flows check CLI > checks the same named-agent contract from declarative JSON 508ms + ✓ flows check CLI > refuses a typo model before probing or contacting relayflowd 1340ms + ✓ flows check CLI > maps every input refusal path to its declared kind without raw exceptions 1706ms + ✓ flows run/resume CLI over the journal protocol > parses run options, submits the kernel dialect, and exits 0 on success 710ms + ✓ flows run/resume CLI over the journal protocol > exits 1 and emits the declared completionReason for a failed run 323ms + ✓ flows run/resume CLI over the journal protocol > exits 3 and names the parked llm step 329ms + ✓ flows run/resume CLI over the journal protocol > classifies a typed hello refusal as a protocol error, not an unreachable daemon 716ms + ✓ flows run/resume CLI over the journal protocol > follows a dispatched worker step instead of reporting a protocol error 1215ms + ✓ flows run/resume CLI over the journal protocol > bounds a worker wait by its lease and reports what it is waiting for 355ms + ✓ flows run/resume CLI over the journal protocol > resumes a parked run from snapshot step types without reading journal sequence one 300ms + ✓ flows run/resume CLI over the journal protocol > maps only run_not_found resumes to exit 2 2537ms + ✓ tests/deterministic-llm.test.ts (5 tests) 41ms + ✓ tests/webhook-live.test.ts (3 tests) 4872ms + ✓ flows serve-webhook writes JSON before the daemon starts, then journals and archives exactly once 1299ms + ✓ resumes the same journal after SIGKILL after spawn and before acknowledgement 3289ms + ✓ tests/input-binding.test.ts (12 tests) 136ms + ✓ tests/scope-compiler.test.ts (25 tests) 9ms + ✓ tests/budget-attribution.test.ts (4 tests) 2ms + ✓ tests/hn-poller.test.ts (6 tests) 4ms + ✓ tests/dir-watcher-poller.test.ts (6 tests) 5ms + ✓ tests/hello-deterministic.test.ts (5 tests) 9ms + ✓ tests/budget-preflight.test.ts (16 tests) 5ms + ✓ tests/work-package-validator.test.ts (7 tests) 3ms + ↓ tests/real-cli-adapters.test.ts (3 tests | 3 skipped) + ✓ tests/model-pricing.test.ts (8 tests) 2ms + ✓ tests/authored-use-loader.test.ts (5 tests) 40ms + ✓ tests/bin.test.ts (7 tests) 1259ms + ✓ tests/parse-json-output.test.ts (7 tests) 1ms + ✓ tests/cli-adapter.test.ts (3 tests) 2ms + ✓ tests/cli-watch.test.ts (10 tests) 10228ms + ✓ flows check --watch > rechecks syntax errors, clears once, and returns the last refusal on Ctrl-C 747ms + ✓ flows check --watch > streams JSON lines without ANSI, recovers after atomic saves, and exits zero after repair 1098ms + ✓ flows check --watch > coalesces 20 concurrent saves into at most two rechecks 1300ms + ✓ flows check --watch > watches transitive relative use imports, cycles, and nearest config changes 1495ms + ✓ flows check --watch > refreshes the import graph and notices missing imports being created 1552ms + ✓ flows check --watch > reloads authored TypeScript instead of reusing the first imported definition 836ms + ✓ flows check --watch > detects a nearer config appearing and falls back after it is deleted 1170ms + ✓ flows check --watch > keeps watching after the target is deleted and recreated 1253ms + ✓ flows check --watch > queues changes during a slow check without overlapping checks 776ms + ✓ tests/journal-client-completion.test.ts (4 tests) 99ms + ✓ tests/direct-run-failure.test.ts (6 tests) 3ms + ✓ tests/memoization.test.ts (57 tests) 697ms + ✓ refuses invalid reuse invocation "run" 655ms + ✓ tests/yaml-local-agent-live.test.ts (7 tests) 3218ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked step CLI and model and journals done 511ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked named CLI and model and journals done 510ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked flow CLI and model and journals done 548ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked project CLI and model and journals done 495ms + ✓ YAML --local-agent through the built CLI and real daemon > still parks without --local-agent 359ms + ✓ YAML --local-agent through the built CLI and real daemon > reports the agent process failure 452ms + ✓ YAML --local-agent through the built CLI and real daemon > preserves declared workspace surfaces that the local worker cannot pin 343ms + ✓ tests/budget-authored-live.test.ts (2 tests) 130ms + ✓ tests/pty-sidechannel.test.ts (5 tests) 2068ms + ✓ view attach preserves worker completion and marks only drive 636ms + ✓ passthrough attach preserves worker completion and marks only drive 619ms + ✓ none attach preserves worker completion and marks only drive 627ms + ✓ tests/placement.test.ts (54 tests) 7ms + ✓ tests/memory.test.ts (18 tests) 3ms + ✓ tests/slack-writeback.test.ts (1 test) 257ms + ✓ tests/worker-platform.test.ts (1 test) 1ms + ✓ tests/classify-outcome.test.ts (2 tests) 2231ms + ✓ classifyOutcome > gives up and reports when a running run never becomes classifiable 2074ms + ✓ tests/cli-progress-wait.test.ts (2 tests) 1503ms + ✓ run starts the wait clock on its first observed lease 769ms + ✓ resume starts the wait clock on its first observed lease 733ms + ✓ tests/worker-cli-abort.test.ts (2 tests) 2635ms + ✓ stops claude and its process group when lease ownership is lost 1330ms + ✓ stops wrapper.mjs and its process group when lease ownership is lost 1305ms + ✓ tests/worker-cli.test.ts (13 tests) 21026ms + ✓ custom wrapper execution identity > passes an explicit safe environment at identification and execution 341ms + ✓ custom wrapper execution identity > refuses a wrapper symlink retarget before delivering private values 321ms + ✓ custom wrapper execution identity > bounds wrapper execution after acknowledgement 533ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a conforming wrapper leaks a stdio pipe to a background helper 2076ms + ✓ custom wrapper execution bounds are reader-owned > resolves when the leaked helper inherits stderr only 1850ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a wrapper leaks a stdio pipe and exits before identifying 3255ms + ✓ custom wrapper execution bounds are reader-owned > journals a completionReason at the default bound when a wrapper leaks a stdio pipe 11259ms + ✓ custom wrapper execution bounds are reader-owned > accepts the same over-8KiB payload whether or not it coalesces with the execute token 455ms + ✓ tests/local-agent-live.test.ts (5 tests) 39666ms + ✓ built CLI local agent against a real daemon > dispatches through the wrapper and keeps --json stdout report-shaped 802ms + ✓ built CLI local agent against a real daemon > runs beyond the initial 30-second lease without a second invocation 35836ms + ✓ built CLI local agent against a real daemon > renders actual agent completion in text output 898ms + ✓ built CLI local agent against a real daemon > returns a failed run when the agent process fails 1060ms + ✓ built CLI local agent against a real daemon > refuses a workspace it cannot pin before invoking the agent 1069ms +stderr | tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +LIVE_ANALYZER_UNAVAILABLE: "/Users/khaliqgant/flows-spec-Q-sidechannel/testdata/preflight/analyze-story-claude-cli auth status" exited 1: analyze-story-claude-cli: "claude -p --model claude-haiku-4-5-20251001" exited 1: — SKIPPING. This skip is diagnostics, not gate-2 acceptance evidence. + +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=47317 run=01M28VX4DQFP16NCJ68N01S0WN while step=two state=Running + + ✓ tests/live-kernel.test.ts (31 tests | 1 skipped) 57575ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 2203ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 5090ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32295ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 1221ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 359ms + ✓ built flows CLI against live relayflowd > f.agent's default flowPath anchors on cwd, not cwd's parent 319ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5575ms + ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 929ms + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 375ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 415ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 1190ms + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 1775ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 884ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 1814ms + + Test Files 69 passed | 1 skipped (70) + Tests 1215 passed | 4 skipped (1219) + Start at 20:32:04 + Duration 58.17s (transform 1.21s, setup 0ms, collect 8.46s, tests 211.79s, environment 6ms, prepare 1.98s) + diff --git a/docs/evidence/spec-Q/sdk-focused.txt b/docs/evidence/spec-Q/sdk-focused.txt new file mode 100644 index 000000000..8374c54ac --- /dev/null +++ b/docs/evidence/spec-Q/sdk-focused.txt @@ -0,0 +1,31 @@ + + RUN v2.1.9 /Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk + + ✓ tests/cli-replay.test.ts (37 tests) 638ms + ✓ flows replay > --json is byte-identical across two CLI invocations (diff) 501ms + ✓ tests/worker-cli-abort.test.ts (2 tests) 2912ms + ✓ stops claude and its process group when lease ownership is lost 1525ms + ✓ stops wrapper.mjs and its process group when lease ownership is lost 1386ms + ✓ tests/pty-sidechannel.test.ts (5 tests) 3152ms + ✓ view attach preserves worker completion and marks only drive 1390ms + ✓ passthrough attach preserves worker completion and marks only drive 948ms + ✓ none attach preserves worker completion and marks only drive 655ms + ✓ tests/worker-cli.test.ts (13 tests) 25865ms + ✓ custom wrapper execution identity > passes an explicit safe environment at identification and execution 704ms + ✓ custom wrapper execution identity > refuses a wrapper symlink retarget before delivering private values 485ms + ✓ custom wrapper execution identity > bounds captured wrapper output 310ms + ✓ custom wrapper execution identity > refuses a duplicate execute protocol frame 514ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a conforming wrapper leaks a stdio pipe to a background helper 1717ms + ✓ custom wrapper execution bounds are reader-owned > resolves when the leaked helper inherits stderr only 1759ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a wrapper leaks a stdio pipe and exits before identifying 3256ms + ✓ custom wrapper execution bounds are reader-owned > journals a completionReason at the default bound when a wrapper leaks a stdio pipe 11257ms + ✓ custom wrapper execution bounds are reader-owned > accepts an execute token and an over-8KiB payload flushed in one write 1958ms + ✓ custom wrapper execution bounds are reader-owned > accepts the same over-8KiB payload whether or not it coalesces with the execute token 1971ms + ✓ custom wrapper execution bounds are reader-owned > still bounds an un-terminated handshake buffer and names the bound 792ms + ✓ delivers the journaled memory pack to the real wrapper and excludes its charge from completion usage 917ms + + Test Files 4 passed (4) + Tests 57 passed (57) + Start at 20:29:40 + Duration 26.18s (transform 404ms, setup 0ms, collect 999ms, tests 32.57s, environment 0ms, prepare 180ms) + diff --git a/docs/evidence/spec-Q/sdk-initial.txt b/docs/evidence/spec-Q/sdk-initial.txt new file mode 100644 index 000000000..f183215fb --- /dev/null +++ b/docs/evidence/spec-Q/sdk-initial.txt @@ -0,0 +1,405 @@ + + RUN v2.1.9 /Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk + +stdout | tests/live-kernel.test.ts +LIVE_KERNEL relayflowd=/Users/khaliqgant/.relayflows-toolchain/target/3923540030/debug/relayflowd +LIVE_KERNEL flows=/Users/khaliqgant/flows-spec-Q-sidechannel/packages/sdk/dist/cli.js + + ✓ tests/daemon-lifecycle.test.ts (42 tests) 45ms + ✓ tests/preflight.test.ts (27 tests) 79ms + ✓ tests/validate.test.ts (68 tests) 88ms + ✓ tests/observer-link.test.ts (39 tests) 275ms + ✓ tests/authored-flow.test.ts (25 tests) 813ms + ✓ tests/journal-client.test.ts (14 tests) 103ms + ✓ tests/tick-source.test.ts (33 tests) 66ms + ✓ tests/cloud-run.test.ts (47 tests) 124ms + ✓ tests/authored-flow-lifecycle-executor.test.ts (27 tests) 542ms + ✓ tests/cli-replay.test.ts (37 tests) 1123ms + ✓ flows replay > --json is byte-identical across two CLI invocations (diff) 780ms + ✓ tests/gate-contract.test.ts (20 tests) 206ms + ✓ tests/verb-field-lint.test.ts (78 tests) 1998ms + ✓ closed per-verb step fields > carries the llm/agent `output` sugar through every path > flows check accepts output on llm 1329ms + ✓ tests/cli-hn-monitor.test.ts (16 tests) 136ms + ✓ tests/tick-runner.test.ts (22 tests) 2653ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms fractional as an invocation error 575ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms exponent notation as an invocation error 347ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms trailing text as an invocation error 300ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms empty as an invocation error 601ms + ✓ CLI argument parsing refuses coercion rather than accepting it > accepts an exact integer and proceeds past parsing 461ms + ✓ tests/authored-flow-slack.test.ts (6 tests) 1748ms + ✓ authored Slack helper effects > replays after SIGKILL before confirm with the same token and one successful completion 624ms + ✓ authored Slack helper effects > replays after SIGKILL before complete with the same token and one successful completion 496ms + ✓ authored Slack helper effects > writes two files for two calls and supports dm, reply, and react 401ms + ✓ tests/backlog-picker.test.ts (14 tests) 69ms + ✓ tests/backlog-picker-flow.test.ts (6 tests) 693ms + ✓ backlog-picker flow > does not emit a stale entry left by a previous run 378ms + ✓ tests/mcp.test.ts (30 tests) 8230ms + ✓ MCP preflight and transports > flows check refuses an undeclared server with exit 2 and no daemon 860ms + ✓ MCP preflight and transports > flows check reports a refusing server and leaves no PID 673ms + ✓ MCP preflight and transports > kills a SIGTERM-resistant silent child after a parent-owned handshake deadline 1310ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with inherit stdio before cleanup finishes 1208ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with ignore stdio before cleanup finishes 2062ms + ✓ MCP preflight and transports > reports malformed connection configuration as config_invalid 498ms + ✓ authored MCP effects against the real kernel > journals one MCP receipt per call with args, result, stable logical key, and a confirmed effect 339ms + ✓ authored MCP effects against the real kernel > reports a dropped tool connection as a failed CLI run 484ms + ✓ tests/authored-flow-operation.test.ts (23 tests) 271ms + ❯ tests/bundle.test.ts (21 tests | 1 failed) 5463ms + ✓ immutable bundles > returns exit 2 naming a byte-flipped payload and refuses to reuse corruption 383ms + ✓ immutable bundles > builds and verifies the canonical YAML fixture through the compiled CLI 1327ms + ✓ immutable bundles > emits the ephemeral warning on CLI stderr and uses the default output directory 456ms + ✓ immutable bundles > preserves quoted asset words and executable permissions 1539ms + × immutable bundles > builds a standalone TS fixture twice with identical executable hashes 293ms + → expected 'REFUSED [bundle_invalid] TypeScript b…' to be '' // Object.is equality + ✓ tests/work-package-consumer.test.ts (13 tests) 202ms + ✓ tests/spec-parity.test.ts (31 tests) 184ms + ✓ tests/stop-process-group.test.ts (6 tests) 10287ms + ✓ every stop reaches the process group, not just the direct child > exits the run after an execution-timeout stop 1658ms + ✓ every stop reaches the process group, not just the direct child > exits the run after a protocol terminate stop 1329ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after a protocol terminate stop 3134ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after an execution-timeout stop 2719ms + ✓ every stop reaches the process group, not just the direct child > holds the loop open long enough for the escalation to run 1152ms + ✓ tests/worker-lease.test.ts (7 tests) 5ms + ✓ tests/typed-output.test.ts (14 tests) 101ms + ✓ tests/json-schema-bound.test.ts (71 tests) 1450ms + ✓ JSON Schema termination bound > walks a deep schema with an explicit stack rather than recursion 1210ms + ✓ tests/mcp-lifecycle.test.ts (4 tests) 7ms + ✓ tests/model-selection.test.ts (10 tests) 9ms + ✓ tests/webhook-live.test.ts (3 tests) 4847ms + ✓ flows serve-webhook writes JSON before the daemon starts, then journals and archives exactly once 1299ms + ✓ resumes the same journal after SIGKILL after spawn and before acknowledgement 3286ms + ✓ tests/relayflowd-path.test.ts (10 tests) 2ms + ✓ tests/direct-input.test.ts (4 tests) 6706ms + ✓ direct .flow.ts input through the built CLI and live runtime > executes inline and file JSON input through relayflowd 2539ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses missing and malformed input before contacting relayflowd 2451ms + ✓ direct .flow.ts input through the built CLI and live runtime > does not run the authored body before daemon availability 843ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses oversized file input before contacting relayflowd 873ms + ✓ tests/f-memory.test.ts (7 tests) 757ms + ✓ tests/webhook.test.ts (6 tests) 55ms + ✓ tests/local-dev-ux.test.ts (8 tests) 14ms + ✓ tests/dependency-validation.test.ts (6 tests) 324ms + ✓ tests/input-binding.test.ts (12 tests) 103ms + ✓ tests/daemon-lifecycle-live.test.ts (9 tests) 13084ms + ✓ flows run against a data dir with no daemon (§6 test 7) > cold start spawns exactly one daemon, the run succeeds, and the daemon outlives the CLI 1357ms + ✓ flows run against a data dir with no daemon (§6 test 7) > polls, bounded, for a daemon that holds the lock before it binds 2432ms + ✓ flows run against a data dir with no daemon (§6 test 7) > attaches to a serving daemon that has not published a connection file 1181ms + ✓ flows run against a data dir with no daemon (§6 test 7) > a second run attaches to the daemon the first one started, spawning nothing 2090ms + ✓ flows run against a data dir with no daemon (§6 test 7) > detects a stale connection file left by a hard kill and starts a fresh daemon 1822ms + ✓ concurrent invocations against one empty data dir (§6 test 15) > ends with exactly one daemon owning the socket, and both runs succeed 1385ms + ✓ refusals from a spawn that cannot produce a daemon > names relayflowd_not_found rather than falling through to PATH 827ms + ✓ refusals from a spawn that cannot produce a daemon > names daemon_start_failed and quotes the daemon log when startup dies 936ms + ✓ refusals from a spawn that cannot produce a daemon > refuses a daemon speaking another protocol version instead of binding over it 1053ms + ✓ tests/deterministic-llm.test.ts (5 tests) 36ms + ✓ tests/scope-compiler.test.ts (25 tests) 5ms + ✓ tests/budget-attribution.test.ts (4 tests) 2ms + ✓ tests/hn-poller.test.ts (6 tests) 3ms + ✓ tests/dir-watcher-poller.test.ts (6 tests) 2ms + ✓ tests/flow-executor-chain.test.ts (12 tests) 13686ms + ✓ flow executor LLM and output-binding chain > runs f.llm -> f.agent -> f.run with schema-verified journal output and the exact allowed model 2334ms + ✓ flow executor LLM and output-binding chain > fails invalid LLM output before the next step: not JSON 1200ms + ✓ flow executor LLM and output-binding chain > fails invalid LLM output before the next step: {"message":7} 1372ms + ✓ flow executor LLM and output-binding chain > retains tagged-template text output 1246ms + ✓ flow executor LLM and output-binding chain > preserves JSON values without promoting them to process wrappers: null 1174ms + ✓ flow executor LLM and output-binding chain > preserves JSON values without promoting them to process wrappers: [1,2] 904ms + ✓ flow executor LLM and output-binding chain > preserves JSON values without promoting them to process wrappers: "hello" 833ms + ✓ flow executor LLM and output-binding chain > runs the authored LLM path through the built flows CLI 1216ms + ✓ flow executor LLM and output-binding chain > passes a declarative verified value through an agent into a deterministic artifact 619ms + ✓ flow executor LLM and output-binding chain > journals a missing optional field as a failure before the consuming command executes 537ms + ✓ flow executor LLM and output-binding chain > flows run consumes YAML bindings and resume reuses the original journal output 2217ms + ✓ tests/hello-deterministic.test.ts (5 tests) 10ms + ✓ tests/budget-preflight.test.ts (16 tests) 5ms + ✓ tests/work-package-validator.test.ts (7 tests) 3ms + ✓ tests/bin.test.ts (7 tests) 1921ms + ✓ built flows binary > refuses through a symlink to the built artifact 441ms + ✓ built flows binary > classifies a signal-terminated auth probe as probe_failed 468ms + ✓ built flows binary > runs one auth probe for three steps sharing a flow CLI 429ms + ✓ tests/yaml-local-agent-live.test.ts (7 tests) 3434ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked step CLI and model and journals done 837ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked named CLI and model and journals done 463ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked flow CLI and model and journals done 482ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked project CLI and model and journals done 472ms + ✓ YAML --local-agent through the built CLI and real daemon > still parks without --local-agent 343ms + ✓ YAML --local-agent through the built CLI and real daemon > reports the agent process failure 485ms + ✓ YAML --local-agent through the built CLI and real daemon > preserves declared workspace surfaces that the local worker cannot pin 351ms + ↓ tests/real-cli-adapters.test.ts (3 tests | 3 skipped) + ✓ tests/model-pricing.test.ts (8 tests) 2ms + ✓ tests/authored-use-loader.test.ts (5 tests) 31ms + ✓ tests/parse-json-output.test.ts (7 tests) 1ms + ✓ tests/cli-adapter.test.ts (3 tests) 2ms + ✓ tests/pty-sidechannel.test.ts (5 tests) 2105ms + ✓ view attach preserves worker completion and marks only drive 635ms + ✓ passthrough attach preserves worker completion and marks only drive 616ms + ✓ none attach preserves worker completion and marks only drive 626ms + ✓ tests/journal-client-completion.test.ts (4 tests) 98ms + ✓ tests/direct-run-failure.test.ts (6 tests) 3ms + ✓ tests/memoization.test.ts (57 tests) 697ms + ✓ refuses invalid reuse invocation "run" 670ms + ✓ tests/cli.test.ts (63 tests) 20258ms + ✓ flows check CLI > binds a checked relative wrapper to the flow directory for worker execution 1068ms + ✓ flows check CLI > uses the raw Claude adapter model flag instead of accepting auth status as model proof 770ms + ✓ flows check CLI > uses Codex login status and reports a rejected model as unavailable, not unauthenticated 1502ms + ✓ flows check CLI > refuses a nonconforming custom wrapper without calling it an authentication failure 910ms + ✓ flows check CLI > accepts an exact allowlisted named-agent model and probes that model 1459ms + ✓ flows check CLI > checks the same named-agent contract from declarative JSON 1445ms + ✓ flows check CLI > refuses a typo model before probing or contacting relayflowd 909ms + ✓ flows check CLI > distinguishes an allowlisted but inaccessible model from broken auth 1044ms + ✓ flows check CLI > refuses cli-unauthenticated.flow.yaml with typed kind cli_unauthenticated and exit 2 870ms + ✓ flows check CLI > resolves a project CLI path relative to the flows.json that declares it 901ms + ✓ flows check CLI > maps every input refusal path to its declared kind without raw exceptions 1856ms + ✓ flows run/resume CLI over the journal protocol > exits 1 and emits the declared completionReason for a failed run 821ms + ✓ flows run/resume CLI over the journal protocol > exits 3 and names the parked llm step 634ms + ✓ flows run/resume CLI over the journal protocol > reports a needs_human agent step as parked for human recovery 631ms + ✓ flows run/resume CLI over the journal protocol > classifies a typed hello refusal as a protocol error, not an unreachable daemon 670ms + ✓ flows run/resume CLI over the journal protocol > follows a dispatched worker step instead of reporting a protocol error 394ms + ✓ flows run/resume CLI over the journal protocol > bounds a worker wait by its lease and reports what it is waiting for 781ms + ✓ flows run/resume CLI over the journal protocol > resumes a parked run from snapshot step types without reading journal sequence one 699ms + ✓ flows run/resume CLI over the journal protocol > maps only run_not_found resumes to exit 2 2254ms + ✓ tests/budget-authored-live.test.ts (2 tests) 127ms + ✓ tests/slack-writeback.test.ts (1 test) 258ms + ✓ tests/placement.test.ts (54 tests) 7ms + ✓ tests/memory.test.ts (18 tests) 4ms + ✓ tests/worker-platform.test.ts (1 test) 2ms + ✓ tests/classify-outcome.test.ts (2 tests) 2224ms + ✓ classifyOutcome > gives up and reports when a running run never becomes classifiable 2066ms + ✓ tests/cli-progress-wait.test.ts (2 tests) 1310ms + ✓ run starts the wait clock on its first observed lease 611ms + ✓ resume starts the wait clock on its first observed lease 699ms + ✓ tests/worker-cli-abort.test.ts (2 tests) 2525ms + ✓ stops claude and its process group when lease ownership is lost 1267ms + ✓ stops wrapper.mjs and its process group when lease ownership is lost 1258ms + ✓ tests/worker-cli.test.ts (13 tests) 26003ms + ✓ custom wrapper execution identity > passes an explicit safe environment at identification and execution 496ms + ✓ custom wrapper execution identity > refuses a wrapper symlink retarget before delivering private values 1072ms + ✓ custom wrapper execution identity > bounds wrapper execution after acknowledgement 986ms + ✓ custom wrapper execution identity > bounds captured wrapper output 1456ms + ✓ custom wrapper execution identity > refuses a duplicate execute protocol frame 1149ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a conforming wrapper leaks a stdio pipe to a background helper 2982ms + ✓ custom wrapper execution bounds are reader-owned > resolves when the leaked helper inherits stderr only 2388ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a wrapper leaks a stdio pipe and exits before identifying 3256ms + ✓ custom wrapper execution bounds are reader-owned > journals a completionReason at the default bound when a wrapper leaks a stdio pipe 11256ms + ✓ custom wrapper execution bounds are reader-owned > accepts the same over-8KiB payload whether or not it coalesces with the execute token 413ms + ❯ tests/cli-watch.test.ts (10 tests | 8 failed) 40824ms + × flows check --watch > rechecks syntax errors, clears once, and returns the last refusal on Ctrl-C 5005ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > streams JSON lines without ANSI, recovers after atomic saves, and exits zero after repair 5002ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > coalesces 20 concurrent saves into at most two rechecks 5003ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > watches transitive relative use imports, cycles, and nearest config changes 5003ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > refreshes the import graph and notices missing imports being created 5003ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > reloads authored TypeScript instead of reusing the first imported definition 5004ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > detects a nearer config appearing and falls back after it is deleted 5002ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + × flows check --watch > keeps watching after the target is deleted and recreated 5003ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + ✓ flows check --watch > queues changes during a slow check without overlapping checks 794ms + ✓ tests/local-agent-live.test.ts (5 tests) 39681ms + ✓ built CLI local agent against a real daemon > dispatches through the wrapper and keeps --json stdout report-shaped 917ms + ✓ built CLI local agent against a real daemon > runs beyond the initial 30-second lease without a second invocation 35883ms + ✓ built CLI local agent against a real daemon > renders actual agent completion in text output 824ms + ✓ built CLI local agent against a real daemon > returns a failed run when the agent process fails 973ms + ✓ built CLI local agent against a real daemon > refuses a workspace it cannot pin before invoking the agent 1082ms +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=42107 run=01M28VTDC2V93DVMKY33GJ0H6H while step=two state=Running + + ❯ tests/live-kernel.test.ts (31 tests | 2 failed) 67786ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 3140ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 6433ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32287ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 1412ms + ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 378ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 450ms + ✓ built flows CLI against live relayflowd > f.agent's default flowPath anchors on cwd, not cwd's parent 464ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5564ms + × built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 5007ms + → Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". + ✓ built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) 377ms + ✓ built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields 371ms + ✓ built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text 364ms + ✓ built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) 420ms + ✓ built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) 413ms + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 515ms + ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 398ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 346ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 393ms + ✓ built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model 489ms + × built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI 1796ms + → LIVE_ANALYZER_UNAVAILABLE: "/Users/khaliqgant/flows-spec-Q-sidechannel/testdata/preflight/analyze-story-claude-cli auth status" exited 1: analyze-story-claude-cli: "claude -p --model claude-haiku-4-5-20251001" exited 1: — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 2191ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 1088ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 2745ms + ✓ a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant 379ms + +⎯⎯⎯⎯⎯⎯ Failed Tests 11 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/bundle.test.ts > immutable bundles > builds a standalone TS fixture twice with identical executable hashes +AssertionError: expected 'REFUSED [bundle_invalid] TypeScript b…' to be '' // Object.is equality + +- Expected ++ Received + ++ REFUSED [bundle_invalid] TypeScript build requires Bun: spawnSync bun ENOENT ++ + + ❯ tests/bundle.test.ts:210:27 + 208| it('builds a standalone TS fixture twice with identical executable h… + 209| const result = invoke(['--out', await temp(), 'packages/sdk/tests/… + 210| expect(result.stderr).toBe(''); expect(result.status).toBe(0); + | ^ + 211| const bundle = result.stdout.trim(); + 212| const second = invoke(['--out', await temp(), 'packages/sdk/tests/… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/11]⎯ + + FAIL tests/cli-watch.test.ts > flows check --watch > rechecks syntax errors, clears once, and returns the last refusal on Ctrl-C + FAIL tests/cli-watch.test.ts > flows check --watch > streams JSON lines without ANSI, recovers after atomic saves, and exits zero after repair + FAIL tests/cli-watch.test.ts > flows check --watch > coalesces 20 concurrent saves into at most two rechecks + FAIL tests/cli-watch.test.ts > flows check --watch > watches transitive relative use imports, cycles, and nearest config changes + FAIL tests/cli-watch.test.ts > flows check --watch > refreshes the import graph and notices missing imports being created + FAIL tests/cli-watch.test.ts > flows check --watch > reloads authored TypeScript instead of reusing the first imported definition + FAIL tests/cli-watch.test.ts > flows check --watch > detects a nearer config appearing and falls back after it is deleted + FAIL tests/cli-watch.test.ts > flows check --watch > keeps watching after the target is deleted and recreated + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked +Error: Test timed out in 5000ms. +If this is a long-running test, pass a timeout value as the last argument or configure it globally with "testTimeout". +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/11]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +Error: LIVE_ANALYZER_UNAVAILABLE: "/Users/khaliqgant/flows-spec-Q-sidechannel/testdata/preflight/analyze-story-claude-cli auth status" exited 1: analyze-story-claude-cli: "claude -p --model claude-haiku-4-5-20251001" exited 1: — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ❯ tests/live-kernel.test.ts:1223:15 + 1221| const notice = `LIVE_ANALYZER_UNAVAILABLE: ${readiness.detail}`; + 1222| if (process.env['RELAYFLOWS_ALLOW_ANALYZER_SKIP'] !== '1') { + 1223| throw new Error( + | ^ + 1224| `${notice} — failing because gate-2 acceptance requires the … + 1225| + 'Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is … + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/11]⎯ + +⎯⎯⎯⎯⎯⎯ Unhandled Errors ⎯⎯⎯⎯⎯⎯ + +Vitest caught 8 unhandled errors during the test run. +This might cause false positive tests. Resolve unhandled errors to make sure your tests are not affected. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-T0QOru/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "rechecks syntax errors, clears once, and returns the last refusal on Ctrl-C". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-zyTD6U/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "streams JSON lines without ANSI, recovers after atomic saves, and exits zero after repair". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-hEvfm5/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "coalesces 20 concurrent saves into at most two rechecks". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-FFk1D6/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "watches transitive relative use imports, cycles, and nearest config changes". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-UI7Vqu/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "refreshes the import graph and notices missing imports being created". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-JWzNP0/authored.flow.ts' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "reloads authored TypeScript instead of reusing the first imported definition". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-K2OPSl/sub/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "detects a nearer config appearing and falls back after it is deleted". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. + +⎯⎯⎯⎯⎯ Uncaught Exception ⎯⎯⎯⎯⎯ +Error: spawn bun ENOENT + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:286:19 + ❯ onErrorNT node:internal/child_process:507:16 + ❯ processTicksAndRejections node:internal/process/task_queues:90:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { errno: -2, code: 'ENOENT', syscall: 'spawn bun', path: 'bun', spawnargs: [ 'run', 'flows', 'check', '--watch', '--json', '/var/folders/_z/f_fpl8j533g_r63706k2xvp00000gn/T/flows-watch-RUVTus/fixture.yaml' ] } +This error originated in "tests/cli-watch.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "keeps watching after the target is deleted and recreated". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ + + Test Files 3 failed | 66 passed | 1 skipped (70) + Tests 11 failed | 1204 passed | 3 skipped (1218) + Errors 8 errors + Start at 20:30:25 + Duration 68.83s (transform 1.68s, setup 0ms, collect 13.49s, tests 285.92s, environment 6ms, prepare 2.86s) + diff --git a/docs/evidence/spec-Q/typecheck-tests.txt b/docs/evidence/spec-Q/typecheck-tests.txt new file mode 100644 index 000000000..daf02e077 --- /dev/null +++ b/docs/evidence/spec-Q/typecheck-tests.txt @@ -0,0 +1,4 @@ + +> @relayflows/sdk@2.0.8 typecheck:tests +> tsc -p tsconfig.tests.json + diff --git a/docs/evidence/spec-Q/typecheck.txt b/docs/evidence/spec-Q/typecheck.txt new file mode 100644 index 000000000..326089087 --- /dev/null +++ b/docs/evidence/spec-Q/typecheck.txt @@ -0,0 +1,4 @@ + +> @relayflows/sdk@2.0.8 typecheck +> tsc --noEmit && tsc -p tsconfig.type-tests.json + diff --git a/kernel/relayflowd-core/src/entry.rs b/kernel/relayflowd-core/src/entry.rs index 645f0cf62..6875ad7ce 100644 --- a/kernel/relayflowd-core/src/entry.rs +++ b/kernel/relayflowd-core/src/entry.rs @@ -251,6 +251,8 @@ pub enum Disposition { #[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] pub struct StepCompletedPayload { + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + pub human_intervention: bool, #[serde(default, skip_serializing_if = "Option::is_none")] pub step_spec_hash: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/kernel/relayflowd-core/src/machine.rs b/kernel/relayflowd-core/src/machine.rs index 0a0282814..0d41a2561 100644 --- a/kernel/relayflowd-core/src/machine.rs +++ b/kernel/relayflowd-core/src/machine.rs @@ -50,6 +50,7 @@ pub enum Action { #[derive(Debug, Clone, PartialEq)] pub struct AttemptResult { + pub human_intervention: bool, pub output: Value, pub budget: Budget, pub completed_by: String, @@ -67,6 +68,7 @@ pub struct AttemptResult { impl AttemptResult { pub fn successful(output: Value, completed_by: impl Into) -> Self { Self { + human_intervention: false, output, budget: Budget::default(), completed_by: completed_by.into(), @@ -415,6 +417,7 @@ pub fn completion_actions( Some(attempt), now_ms, StepCompletedPayload { + human_intervention: result.human_intervention, step_spec_hash: None, input_hash: None, reused_from: None, diff --git a/kernel/relayflowd-core/src/machine/cancel.rs b/kernel/relayflowd-core/src/machine/cancel.rs index 32770ec2e..728d540bd 100644 --- a/kernel/relayflowd-core/src/machine/cancel.rs +++ b/kernel/relayflowd-core/src/machine/cancel.rs @@ -43,6 +43,7 @@ pub(super) fn cancel_run_actions(state: &RunState, now_ms: i64) -> Vec { Some(*attempt), now_ms, StepCompletedPayload { + human_intervention: false, step_spec_hash: None, input_hash: None, reused_from: None, diff --git a/kernel/relayflowd-core/src/machine/parallel_tests.rs b/kernel/relayflowd-core/src/machine/parallel_tests.rs index a85b3fe1e..755bdb3d1 100644 --- a/kernel/relayflowd-core/src/machine/parallel_tests.rs +++ b/kernel/relayflowd-core/src/machine/parallel_tests.rs @@ -436,6 +436,7 @@ fn failed_run_drains_open_siblings_before_terminal_entry() { Some(1), 22, StepCompletedPayload { + human_intervention: false, step_spec_hash: None, input_hash: None, reused_from: None, diff --git a/kernel/relayflowd-core/src/machine/recovery.rs b/kernel/relayflowd-core/src/machine/recovery.rs index 984dc547d..e59c236c4 100644 --- a/kernel/relayflowd-core/src/machine/recovery.rs +++ b/kernel/relayflowd-core/src/machine/recovery.rs @@ -114,6 +114,7 @@ pub fn abandonment_actions( Some(attempt), now_ms, StepCompletedPayload { + human_intervention: false, step_spec_hash: None, input_hash: None, reused_from: None, diff --git a/kernel/relayflowd-core/src/machine/tests.rs b/kernel/relayflowd-core/src/machine/tests.rs index 8d27e322d..2115ccf84 100644 --- a/kernel/relayflowd-core/src/machine/tests.rs +++ b/kernel/relayflowd-core/src/machine/tests.rs @@ -288,6 +288,7 @@ fn all_backing_off_steps_return_timers() { })) .unwrap(); let result = AttemptResult { + human_intervention: false, output: Value::Null, budget: Budget::default(), completed_by: "kernel".to_owned(), @@ -398,6 +399,7 @@ fn inspect_recovery_injects_the_dirty_pin_completion_reason_and_tail() { let started = started_agent(&spec, clean); let running = RunState::fold("run", spec.clone(), std::slice::from_ref(&started)).unwrap(); let result = AttemptResult { + human_intervention: false, output: Value::Null, budget: Budget::default(), completed_by: "worker".to_owned(), @@ -515,6 +517,7 @@ fn worker_reported_failure_without_detail_still_records_a_verification() { })) .unwrap(); let result = AttemptResult { + human_intervention: false, output: Value::Null, budget: Budget::default(), completed_by: "worker".to_owned(), diff --git a/kernel/relayflowd-core/src/memoization.rs b/kernel/relayflowd-core/src/memoization.rs index d16c9732e..a9b174808 100644 --- a/kernel/relayflowd-core/src/memoization.rs +++ b/kernel/relayflowd-core/src/memoization.rs @@ -73,7 +73,8 @@ pub fn candidates(entries: &[JournalEntry]) -> Result, serde_j .filter(|e| e.entry_type == EntryType::StepCompleted) { let payload: StepCompletedPayload = serde_json::from_value(entry.payload.clone())?; - if payload.completion_reason == CompletionReason::Success + if !payload.human_intervention + && payload.completion_reason == CompletionReason::Success && payload.disposition == Disposition::StepDone && payload.step_spec_hash.is_some() && payload.input_hash.is_some() diff --git a/kernel/relayflowd-core/src/state/tests.rs b/kernel/relayflowd-core/src/state/tests.rs index e40c27229..c36d40984 100644 --- a/kernel/relayflowd-core/src/state/tests.rs +++ b/kernel/relayflowd-core/src/state/tests.rs @@ -22,6 +22,7 @@ fn journal_replays_data_gate_verdict_without_rerunning_completed_code() { Some(1), 5, StepCompletedPayload { + human_intervention: false, step_spec_hash: None, input_hash: None, reused_from: None, diff --git a/kernel/relayflowd-core/tests/memoization.rs b/kernel/relayflowd-core/tests/memoization.rs index 4021387cf..b1cfda4e5 100644 --- a/kernel/relayflowd-core/tests/memoization.rs +++ b/kernel/relayflowd-core/tests/memoization.rs @@ -99,6 +99,9 @@ fn changed_spec_or_input_dispatches_and_legacy_or_failed_records_miss() { .unwrap() .remove("step_spec_hash"); assert!(candidates(&[legacy]).unwrap().is_empty()); + let mut influenced = source(&state); + influenced.payload["human_intervention"] = true.into(); + assert!(candidates(&[influenced]).unwrap().is_empty()); let mut failed = source(&state); failed.payload["completionReason"] = "worker_error".into(); assert!(candidates(&[failed]).unwrap().is_empty()); diff --git a/kernel/relayflowd/src/engine.rs b/kernel/relayflowd/src/engine.rs index a005306c3..886b6ce55 100644 --- a/kernel/relayflowd/src/engine.rs +++ b/kernel/relayflowd/src/engine.rs @@ -12,6 +12,21 @@ use relayflowd_journal::{Registry, SqliteJournal}; use sha2::{Digest, Sha256}; use ulid::Ulid; +#[derive(Debug)] +pub struct HumanInfluencedRun { + pub step_id: String, +} +impl std::fmt::Display for HumanInfluencedRun { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "step {:?}; pass --allow-human-influenced to continue", + self.step_id + ) + } +} +impl std::error::Error for HumanInfluencedRun {} + #[derive(Debug)] pub struct RunTerminalError { pub run_id: String, @@ -52,6 +67,7 @@ pub use wake::EventSubmitOutcome; #[derive(Debug, Clone, Default)] #[doc(hidden)] pub struct DriveOptions { + pub allow_human_influenced: bool, pub stop_after: Option, /// Test/debug hook: pause immediately before this runnable step starts. pub pause_before_step: Option, @@ -243,6 +259,22 @@ impl Engine { lease_is_active: &dyn Fn(&str, u32) -> bool, ) -> Result { let mut journal = self.open_run(run_id)?; + if !options.allow_human_influenced { + for entry in journal.scan_all()? { + if entry.entry_type == EntryType::StepCompleted + && entry + .payload + .get("human_intervention") + .and_then(serde_json::Value::as_bool) + == Some(true) + { + return Err(HumanInfluencedRun { + step_id: entry.step_id.unwrap_or_default(), + } + .into()); + } + } + } let registry = self.registry()?; if registry.lookup(run_id)?.is_none() { registry diff --git a/kernel/relayflowd/src/engine/remote.rs b/kernel/relayflowd/src/engine/remote.rs index 3956b7251..ff36aa78c 100644 --- a/kernel/relayflowd/src/engine/remote.rs +++ b/kernel/relayflowd/src/engine/remote.rs @@ -17,6 +17,7 @@ use crate::worker::LeaseProbe; #[derive(Debug, Clone)] pub struct OutOfBandCompletion { + pub human_intervention: bool, pub attempt: u32, pub idempotency_key: String, pub completion_reason: CompletionReason, @@ -127,6 +128,7 @@ impl Engine { completion.end_pins }; let result = AttemptResult { + human_intervention: completion.human_intervention, output: completion.output, budget: completion.budget, completed_by: completion.completed_by, @@ -153,10 +155,24 @@ impl Engine { /// valid, heartbeating lease are left running; only genuinely dead /// attempts (worker detached, or lease deadline passed) are recovered. pub fn resume_live(&self, run_id: &str, leases: &dyn LeaseProbe) -> Result { + self.resume_live_with_human_influence(run_id, leases, false) + } + + pub fn resume_live_with_human_influence( + &self, + run_id: &str, + leases: &dyn LeaseProbe, + allow_human_influenced: bool, + ) -> Result { let now_ms = self.clock.now_ms(); - self.resume_filtered(run_id, DriveOptions::default(), &|step_id, attempt| { - leases.lease_active(run_id, step_id, attempt, now_ms) - }) + self.resume_filtered( + run_id, + DriveOptions { + allow_human_influenced, + ..DriveOptions::default() + }, + &|step_id, attempt| leases.lease_active(run_id, step_id, attempt, now_ms), + ) } /// Explain a dead leased attempt (worker disconnect or lease expiry), then diff --git a/kernel/relayflowd/src/exec_det.rs b/kernel/relayflowd/src/exec_det.rs index d8a541224..9a10a83b8 100644 --- a/kernel/relayflowd/src/exec_det.rs +++ b/kernel/relayflowd/src/exec_det.rs @@ -114,6 +114,7 @@ pub(crate) fn execute_placed_with_input( "stderr_tail": tail(&stderr), }); AttemptResult { + human_intervention: false, output, budget: Budget::default(), completed_by: "kernel".to_owned(), @@ -155,6 +156,7 @@ fn tail(bytes: &[u8]) -> String { fn worker_error(detail: &str) -> AttemptResult { AttemptResult { + human_intervention: false, output: json!({"error": detail}), budget: Budget::default(), completed_by: "kernel".to_owned(), diff --git a/kernel/relayflowd/src/main.rs b/kernel/relayflowd/src/main.rs index fc4274743..03c0050b8 100644 --- a/kernel/relayflowd/src/main.rs +++ b/kernel/relayflowd/src/main.rs @@ -82,6 +82,7 @@ fn main() -> Result<()> { stop_after, pause_before_step, pause_before_completion, + ..DriveOptions::default() }, )?; println!("{}", serde_json::to_string(&outcome)?); diff --git a/kernel/relayflowd/src/server.rs b/kernel/relayflowd/src/server.rs index 7bcbbb784..38bc7cb95 100644 --- a/kernel/relayflowd/src/server.rs +++ b/kernel/relayflowd/src/server.rs @@ -182,7 +182,7 @@ fn handle_request( ) } "run.resume" => { - let params: RunIdParams = decode_params(request.params)?; + let params: RunResumeParams = decode_params(request.params)?; let registry = relayflowd_journal::Registry::open(data_dir.join("relayflowd.sqlite3")) .map_err(|error| internal_error(error.into()))?; if registry @@ -257,8 +257,21 @@ fn handle_request( // Live resume: attempts with a valid, heartbeating lease on this // hub stay running; only genuinely dead attempts are recovered. let outcome = engine - .resume_live(¶ms.run_id, hub.as_ref()) - .map_err(internal_error)?; + .resume_live_with_human_influence( + ¶ms.run_id, + hub.as_ref(), + params.allow_human_influenced, + ) + .map_err(|error| { + if error + .downcast_ref::() + .is_some() + { + ("human_influenced_run", error.to_string()) + } else { + internal_error(error) + } + })?; if outcome.completion_reason.is_some() { hub.finish_run(¶ms.run_id); } @@ -371,6 +384,7 @@ fn handle_request( ¶ms.run_id, ¶ms.step_id, OutOfBandCompletion { + human_intervention: params.human_intervention, attempt: params.attempt, idempotency_key: params.idempotency_key, completion_reason: params.completion_reason, diff --git a/kernel/relayflowd/src/server/tests/agent/contract.rs b/kernel/relayflowd/src/server/tests/agent/contract.rs index 8c8a9e05b..92149ee0f 100644 --- a/kernel/relayflowd/src/server/tests/agent/contract.rs +++ b/kernel/relayflowd/src/server/tests/agent/contract.rs @@ -309,3 +309,72 @@ fn a_replacement_worker_that_never_reported_the_pinned_surface_is_not_dispatched "an undispatchable attempt parks; it must not be marked as leased" ); } + +#[test] +fn human_intervention_is_durable_and_resume_requires_explicit_override() { + let directory = tempdir().unwrap(); + let data_dir = directory.path(); + let hub = Arc::new(ProtocolHub::default()); + let worker_peer = attach_llm_worker(data_dir, &hub, 91); + let mut reader = BufReader::new(worker_peer); + let (writer, _peer) = shared_writer(); + let started = request( + data_dir, + &hub, + 92, + &writer, + &json!({ + "id":"start", "verb":"run.start", "params":{"spec":{"steps":[ + {"id":"influenced","type":"llm","prompt":"hi"} + ]}} + }) + .to_string(), + ); + assert!(started.ok, "{:?}", started.error); + let run_id = started.result.unwrap()["run_id"] + .as_str() + .unwrap() + .to_owned(); + let dispatch = read_frame(&mut reader)["data"].clone(); + let completed = request( + data_dir, + &hub, + 91, + &writer, + &json!({ + "id":"complete", "verb":"step.complete", "params":{ + "run_id":run_id, "step_id":"influenced", "attempt":1, + "idempotency_key":dispatch["idempotency_key"], "completionReason":"success", + "output":"operator-influenced result", "human_intervention":true + } + }) + .to_string(), + ); + assert!(completed.ok, "{:?}", completed.error); + assert!(step_completions(data_dir, &run_id)[0].human_intervention); + let engine = Engine::new(data_dir); + let before = engine.journal_entries(&run_id, 1, usize::MAX).unwrap(); + assert!(engine.resume(&run_id, None).is_err()); + let refused = request( + data_dir, + &hub, + 92, + &writer, + &json!({ + "id":"resume", "verb":"run.resume", "params":{"run_id":run_id} + }) + .to_string(), + ); + let error = refused.error.unwrap(); + assert_eq!(error.code, "human_influenced_run"); + assert!(error.message.contains("step \"influenced\"")); + assert_eq!( + engine.journal_entries(&run_id, 1, usize::MAX).unwrap(), + before + ); + let allowed = request(data_dir, &hub, 92, &writer, &json!({ + "id":"resume", "verb":"run.resume", "params":{"run_id":run_id,"allow_human_influenced":true} + }).to_string()); + assert!(allowed.ok, "{:?}", allowed.error); + assert_eq!(allowed.result.unwrap()["status"], "completed"); +} diff --git a/kernel/relayflowd/src/server/wire.rs b/kernel/relayflowd/src/server/wire.rs index 757d2d232..a19b443a3 100644 --- a/kernel/relayflowd/src/server/wire.rs +++ b/kernel/relayflowd/src/server/wire.rs @@ -42,6 +42,14 @@ pub(super) struct RunStartParams { pub reuse_from_run_id: Option, } +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct RunResumeParams { + pub run_id: String, + #[serde(default)] + pub allow_human_influenced: bool, +} + #[derive(Deserialize)] #[serde(deny_unknown_fields)] pub(super) struct RunIdParams { @@ -75,6 +83,8 @@ pub(super) struct StepHeartbeatParams { #[derive(Deserialize)] #[serde(deny_unknown_fields)] pub(super) struct StepCompleteParams { + #[serde(default)] + pub human_intervention: bool, pub run_id: String, pub step_id: String, pub attempt: u32, diff --git a/kernel/relayflowd/tests/budget_gate.rs b/kernel/relayflowd/tests/budget_gate.rs index 4f810c849..f1eaacb4b 100644 --- a/kernel/relayflowd/tests/budget_gate.rs +++ b/kernel/relayflowd/tests/budget_gate.rs @@ -42,6 +42,7 @@ fn crossing_completion_is_durable_and_next_step_is_refused() { &started.run_id, "first", OutOfBandCompletion { + human_intervention: false, attempt: d.attempt, idempotency_key: d.idempotency_key, completion_reason: CompletionReason::Success, diff --git a/kernel/relayflowd/tests/parallel_driver.rs b/kernel/relayflowd/tests/parallel_driver.rs index b5299c1b0..af9f82ba3 100644 --- a/kernel/relayflowd/tests/parallel_driver.rs +++ b/kernel/relayflowd/tests/parallel_driver.rs @@ -166,6 +166,7 @@ fn complete(engine: &Engine, run_id: &str, dispatch: &StepDispatch) -> RunStatus run_id, &dispatch.step_id, OutOfBandCompletion { + human_intervention: false, attempt: dispatch.attempt, idempotency_key: dispatch.idempotency_key.clone(), completion_reason: CompletionReason::Success, diff --git a/packages/sdk/src/cli.ts b/packages/sdk/src/cli.ts index c6437bda9..8135dfc90 100644 --- a/packages/sdk/src/cli.ts +++ b/packages/sdk/src/cli.ts @@ -50,8 +50,8 @@ type ParsedArgs = | { command: 'serve-webhook'; dataDir: string; port: number } | { command: 'cloud-run'; value: string; json: boolean; wait: boolean } | { command: 'check'; json: boolean; watch: boolean; value: string } - | { command: 'run'; bucket: string | undefined; reuseFromRunId: string | undefined; localAgent: boolean; dataDir: string; input: string | undefined; json: boolean; spawn: boolean; noObserverLink: boolean; value: string } - | { command: 'resume'; dataDir: string; json: boolean; spawn: boolean; noObserverLink: boolean; value: string } + | { command: 'run'; bucket: string | undefined; reuseFromRunId: string | undefined; localAgent: boolean; dataDir: string; input: string | undefined; json: boolean; spawn: boolean; noObserverLink: boolean; allowHumanInfluenced: boolean; value: string } + | { command: 'resume'; dataDir: string; json: boolean; spawn: boolean; noObserverLink: boolean; allowHumanInfluenced: boolean; value: string } | { command: 'observer'; dataDir: string } | { command: 'hn-monitor'; sub: 'start'; dataDir: string; specPath: string; pollIntervalMs: number | undefined } | { command: 'tick'; sub: 'start'; dataDir: string; specPath: string; scheduleId: string; @@ -71,8 +71,8 @@ const USAGE = [ 'flows run --cloud [--json] [--wait] ', 'flows run [--json] [--no-spawn] [--no-observer-link] [--data-dir ] [--local-agent] --input ', 'flows tick start --schedule-id --interval-ms [--epoch-ms ] [--max-catch-up ] [--poll-interval-ms ] [--data-dir ] ', - 'flows resume [--json] [--no-spawn] [--no-observer-link] [--data-dir ] ', - 'flows replay [--json] [--data-dir ] [--at ]', + 'flows resume [--allow-human-influenced] [--json] [--no-spawn] [--no-observer-link] [--data-dir ] ', + 'flows replay [--allow-human-influenced] [--json] [--data-dir ] [--at ]', 'flows observer [--data-dir ]', 'flows hn-monitor start [--data-dir ] [--poll-interval-ms ] ', ].join('\n'); @@ -183,6 +183,8 @@ export async function runCli( }; const lifecycle = { ...(parsed.command === 'run' ? { bucket: parsed.bucket } : {}), + allowHumanInfluenced: parsed.allowHumanInfluenced, + onPtyReady: (path: string) => io.stderr(`PTY ${path}`), ...(parsed.command === 'run' && parsed.reuseFromRunId !== undefined ? { reuseFromRunId: parsed.reuseFromRunId } : {}), localAgent: parsed.command === 'run' && parsed.localAgent, onProgress: showProgress, @@ -423,6 +425,7 @@ function parseArgs(args: readonly string[]): ParsedArgs | undefined { let cloud = false; let wait = false; let localAgent = false; + let allowHumanInfluenced = false; let dataDir = DEFAULT_DATA_DIR; let sawDataDir = false; let spawn = true; @@ -440,6 +443,11 @@ function parseArgs(args: readonly string[]): ParsedArgs | undefined { else wait = true; continue; } + if (argument === '--allow-human-influenced') { + if (command === 'check' || allowHumanInfluenced) return undefined; + allowHumanInfluenced = true; + continue; + } if (argument === '--local-agent') { if (command !== 'run' || localAgent) return undefined; localAgent = true; @@ -509,7 +517,7 @@ function parseArgs(args: readonly string[]): ParsedArgs | undefined { // local run -- an inline input, a data dir, a suppressed daemon, a local // agent, a local observer-link opt-out -- describes nothing there and is // refused rather than ignored. - if (sawInput || sawDataDir || !spawn || localAgent || noObserverLink || reuseFromRunId !== undefined) return undefined; + if (allowHumanInfluenced || sawInput || sawDataDir || !spawn || localAgent || noObserverLink || reuseFromRunId !== undefined) return undefined; return { command: 'cloud-run', value: positionals[0]!, json, wait }; } if (wait) return undefined; @@ -519,8 +527,8 @@ function parseArgs(args: readonly string[]): ParsedArgs | undefined { return command === 'check' ? { command, json, watch, value: positionals[0]! } : command === 'run' - ? { command, bucket, reuseFromRunId, localAgent, dataDir, input, json, spawn, noObserverLink, value: positionals[0]! } - : { command, dataDir, json, spawn, noObserverLink, value: positionals[0]! }; + ? { command, bucket, reuseFromRunId, localAgent, dataDir, input, json, spawn, noObserverLink, allowHumanInfluenced, value: positionals[0]! } + : { command, dataDir, json, spawn, noObserverLink, allowHumanInfluenced, value: positionals[0]! }; } function parseHnMonitorArgs(rest: readonly string[]): ParsedArgs | undefined { diff --git a/packages/sdk/src/cli/direct-run.ts b/packages/sdk/src/cli/direct-run.ts index 80fbf7199..17c041750 100644 --- a/packages/sdk/src/cli/direct-run.ts +++ b/packages/sdk/src/cli/direct-run.ts @@ -64,7 +64,7 @@ export async function runDirectFlow( try { const { handle, getDefinition } = checked.loaded; if (options.localAgent) { - localAgent = await attachLocalAgent(client); + localAgent = await attachLocalAgent(client, dataDir, options.onPtyReady); // A session owns one worker registration. Keep the workspace-free LLM // worker on its own connection so it cannot replace the agent worker. llmClient = new JournalClient(socketPath); diff --git a/packages/sdk/src/cli/replay.ts b/packages/sdk/src/cli/replay.ts index 80c9a6128..d6924de34 100644 --- a/packages/sdk/src/cli/replay.ts +++ b/packages/sdk/src/cli/replay.ts @@ -8,16 +8,21 @@ export interface ReplayArgs { json: boolean; dataDir: string; at?: string; + allowHumanInfluenced?: boolean; } export function parseReplayArgs(args: readonly string[]): ReplayArgs | undefined { let json = false; + let allowHumanInfluenced = false; let dataDir: string | undefined; let at: string | undefined; const positionals: string[] = []; for (let index = 0; index < args.length; index += 1) { const argument = args[index]!; - if (argument === '--json') { + if (argument === '--allow-human-influenced') { + if (allowHumanInfluenced) return undefined; + allowHumanInfluenced = true; + } else if (argument === '--json') { if (json) return undefined; json = true; } else if (argument === '--data-dir' || argument === '--at') { @@ -37,7 +42,7 @@ export function parseReplayArgs(args: readonly string[]): ReplayArgs | undefined } } if (positionals.length !== 1) return undefined; - return { command: 'replay', value: positionals[0]!, json, dataDir: dataDir ?? '.relayflowd', at }; + return { allowHumanInfluenced, command: 'replay', value: positionals[0]!, json, dataDir: dataDir ?? '.relayflowd', at }; } export async function replayJournal(args: ReplayArgs, io: CliIo): Promise<0 | 1 | 2> { @@ -46,6 +51,10 @@ export async function replayJournal(args: ReplayArgs, io: CliIo): Promise<0 | 1 for await (const event of walkJournal(args.value, args.dataDir, { at: args.at })) { const payload = event.payload !== null && typeof event.payload === 'object' && !Array.isArray(event.payload) ? event.payload as Record : {}; + if (event.entry_type === 'step.completed' && payload['human_intervention'] === true && !args.allowHumanInfluenced) { + io.stderr(`REFUSED [human_influenced_run] step ${JSON.stringify(event.step_id)}; pass --allow-human-influenced to continue`); + return 2; + } io.stdout(args.json ? canonicalize({ step_id: event.step_id, kind: event.entry_type, diff --git a/packages/sdk/src/cli/run.ts b/packages/sdk/src/cli/run.ts index 86a29d78d..c68bbd227 100644 --- a/packages/sdk/src/cli/run.ts +++ b/packages/sdk/src/cli/run.ts @@ -68,6 +68,8 @@ export interface RunProgress { export interface RunLifecycleOptions { bucket?: string; + allowHumanInfluenced?: boolean; + onPtyReady?: (path: string) => void; reuseFromRunId?: string; onProgress?: (event: ProgressEvent) => void; localAgent?: boolean; @@ -114,7 +116,7 @@ async function executeCheckedFlow( const spec = toKernelSpec(checked.flow!); // Use the checked CLI/model and declared surfaces unchanged. The worker // advertises its existing pins; the daemon still owns surface matching. - if (options.localAgent) localAgent = await attachLocalAgent(client); + if (options.localAgent) localAgent = await attachLocalAgent(client, dataDir, options.onPtyReady); const outcome = await client.runStart(spec, options.reuseFromRunId); const execution = await classifyOutcome(client, 'run', outcome, base, socketPath, options); if (options.reuseFromRunId !== undefined) { @@ -149,10 +151,16 @@ export async function resumeFlow( if (connected !== undefined) return connected; try { - await resumeSlackEffect(client, runId, dataDir); - const outcome = await client.runResume(runId); + let outcome = await client.runResume(runId, options.allowHumanInfluenced); + if (await resumeSlackEffect(client, runId, dataDir)) { + outcome = await client.runResume(runId, options.allowHumanInfluenced); + } return await classifyOutcome(client, 'resume', outcome, base, socketPath, options); } catch (error) { + if (error instanceof JournalProtocolError && error.code === 'human_influenced_run') { + return { exitCode: 2, report: { ...base, runId, socketPath, + diagnostics: [{ severity: 'refusal', kind: 'human_influenced_run', message: error.message.replace(/^human_influenced_run: /, '') }] } }; + } if (error instanceof AuthoredFlowExecutionError && (error.code === 'helper_slack.credential_missing' || error.code === 'helper_slack.mount_required')) { return { exitCode: 2, report: { ...base, runId, socketPath, @@ -280,11 +288,11 @@ export async function classifyOutcome( } if (inspection?.runningStep !== undefined) { await waitForRunningStep(client, current.run_id, inspection.runningStep, options); - current = await client.runResume(current.run_id); + current = await client.runResume(current.run_id, command === 'run' || options.allowHumanInfluenced); continue; } if (inspection?.status === 'completed' || inspection?.status === 'failed') { - current = await client.runResume(current.run_id); + current = await client.runResume(current.run_id, command === 'run' || options.allowHumanInfluenced); continue; } // The run is still RUNNING but no step is identifiable at this instant. @@ -312,7 +320,7 @@ export async function classifyOutcome( // is already progressing -- it returns the current state rather than // re-dispatching. This loop leans on that up to MAX_UNCLASSIFIED_POLLS // times while the daemon is mid-transition. - current = await client.runResume(current.run_id); + current = await client.runResume(current.run_id, command === 'run' || options.allowHumanInfluenced); continue; } // Fail closed rather than loop forever: if it never resolves, the original diff --git a/packages/sdk/src/failure-kinds.ts b/packages/sdk/src/failure-kinds.ts index 7fbf59584..e4b2d574b 100644 --- a/packages/sdk/src/failure-kinds.ts +++ b/packages/sdk/src/failure-kinds.ts @@ -74,6 +74,7 @@ export const RUN_FAILURE_KINDS = [ 'bucket_unreachable', 'bundle_signature_invalid', 'bundle_unsupported', + 'human_influenced_run', 'reuse_spec_mismatch', 'reuse_run_not_found', 'reuse_journal_read_failed', diff --git a/packages/sdk/src/journal-client.ts b/packages/sdk/src/journal-client.ts index 5240208de..10659d41a 100644 --- a/packages/sdk/src/journal-client.ts +++ b/packages/sdk/src/journal-client.ts @@ -209,8 +209,8 @@ export class JournalClient extends EventEmitter { } /** §3 memoized resume. */ - runResume(runId: string): Promise { - return this.request('run.resume', { run_id: runId }, null); + runResume(runId: string, allowHumanInfluenced = false): Promise { + return this.request('run.resume', { run_id: runId, ...(allowHumanInfluenced ? { allow_human_influenced: true } : {}) }, null); } /** Durably request cancellation and return the terminal run fact. */ @@ -367,6 +367,7 @@ export class JournalClient extends EventEmitter { end_pins?: Pins; effects?: EffectRef[]; trajectory_tail?: unknown; + human_intervention?: boolean; } = {}, ): Promise { return this.request('step.complete', { diff --git a/packages/sdk/src/local-agent.ts b/packages/sdk/src/local-agent.ts index fb68ef2ef..4fe5ad5e3 100644 --- a/packages/sdk/src/local-agent.ts +++ b/packages/sdk/src/local-agent.ts @@ -3,7 +3,7 @@ import type { JournalClient } from './journal-client.js'; import { AgentWorker } from './worker.js'; /** A local worker for stream-only steps; no workspace recovery is claimed. */ -export async function attachLocalAgent(client: JournalClient): Promise<{ +export async function attachLocalAgent(client: JournalClient, dataDir?: string, onPtyReady?: (path: string) => void): Promise<{ stream: string; readonly failure: unknown; close(): Promise; @@ -14,6 +14,7 @@ export async function attachLocalAgent(client: JournalClient): Promise<{ const worker = new AgentWorker(client, { workerId: stream, capacity: 1, + dataDir, onPtyReady, pins: { workspace: [], streams: [{ stream, read_offset: 0 }] }, }); let failure: unknown; diff --git a/packages/sdk/src/protocol.ts b/packages/sdk/src/protocol.ts index 0679b5eb6..447acee70 100644 --- a/packages/sdk/src/protocol.ts +++ b/packages/sdk/src/protocol.ts @@ -98,6 +98,7 @@ export interface RunOutcome { export type RunStartResult = RunOutcome; export interface RunResumeParams { + allow_human_influenced?: boolean; run_id: string; } export type RunResumeResult = RunOutcome; @@ -297,6 +298,7 @@ export interface EffectConfirmResult { } export interface StepCompleteParams { + human_intervention?: boolean; run_id: string; step_id: string; attempt: number; diff --git a/packages/sdk/src/pty-sidechannel.ts b/packages/sdk/src/pty-sidechannel.ts new file mode 100644 index 000000000..c0a0210b7 --- /dev/null +++ b/packages/sdk/src/pty-sidechannel.ts @@ -0,0 +1,92 @@ +import { createServer, type Socket } from 'node:net'; +import { chmod, mkdir } from 'node:fs/promises'; +import { dirname, join, resolve } from 'node:path'; + +export interface SidechannelContext { + dataDir: string; + runId: string; + stepId: string; + onReady?: (path: string) => void; + onDrive: () => void; +} + +export function ptySocketPath(context: Pick): string { + for (const id of [context.runId, context.stepId]) { + if (!id || id === '.' || id === '..' || /[/\\\0]/.test(id)) throw new Error('Invalid sidechannel path component'); + } + return join(resolve(context.dataDir), 'runs', context.runId, 'steps', context.stepId, 'pty.sock'); +} + +/** Best-effort live bytes. Slow peers are dropped; they never pause execution. */ +export async function openSidechannel( + context: SidechannelContext, + input: (bytes: Buffer) => boolean | Promise, + canDrive: () => boolean = () => true, +) { + const peers = new Map(); + let closed = false; + const server = createServer(socket => { + if (peers.size >= 16) { socket.destroy(); return; } + peers.set(socket, false); + let hello = Buffer.alloc(0); + let mode: string | undefined; + socket.setTimeout(2_000, () => socket.destroy()); + socket.on('error', () => socket.destroy()); + socket.on('close', () => peers.delete(socket)); + socket.on('data', (bytes: Buffer) => { + if (mode === undefined) { + hello = Buffer.concat([hello, bytes]); + const end = hello.indexOf(10); + if (end < 0) { if (hello.length > 32) socket.destroy(); return; } + const line = hello.subarray(0, end).toString('utf8'); + if (!['HELLO view', 'HELLO drive', 'HELLO passthrough'].includes(line)) { socket.destroy(); return; } + mode = line.slice(6); + if (mode === 'drive' && !canDrive()) { socket.destroy(); return; } + socket.setTimeout(0); + peers.set(socket, true); + // Passthrough is a passive raw-byte view in this initial slice. + if (mode === 'drive') context.onDrive(); + bytes = hello.subarray(end + 1); + hello = Buffer.alloc(0); + } + if (mode === 'drive' && bytes.length > 0) { + socket.pause(); + void Promise.resolve().then(() => input(bytes)).then(accepted => { + if (!accepted) socket.destroy(); + else if (!socket.destroyed) socket.resume(); + }, () => socket.destroy()); + } + }); + }); + server.on('error', () => { for (const peer of peers.keys()) peer.destroy(); }); + try { + const path = ptySocketPath(context); + // Restrict traversal as well as socket access to the worker's OS user. + await mkdir(dirname(path), { recursive: true, mode: 0o700 }); + await chmod(dirname(path), 0o700); + await new Promise((resolve, reject) => { + server.once('error', reject); + server.listen(path, () => { server.off('error', reject); resolve(); }); + }); + await chmod(path, 0o600); + context.onReady?.(path); + } catch { + // Never unlink a pre-existing socket: it may belong to a live attempt. + if (server.listening) server.close(); + return undefined; + } + return { + publish(bytes: Buffer) { + if (closed) return; + for (const [peer, ready] of peers) { + if (ready && !peer.write(bytes)) peer.destroy(); + } + }, + close() { + if (closed) return; + closed = true; + for (const peer of peers.keys()) peer.destroy(); + server.close(); + }, + }; +} diff --git a/packages/sdk/src/worker-cli.ts b/packages/sdk/src/worker-cli.ts index f5d7ffb8d..d9481b541 100644 --- a/packages/sdk/src/worker-cli.ts +++ b/packages/sdk/src/worker-cli.ts @@ -1,4 +1,5 @@ import { decodeProviderResult, decodeWrapperResult, requirePricedUsage } from './worker-usage.js'; +import { openSidechannel, type SidechannelContext } from './pty-sidechannel.js'; import { spawn } from 'node:child_process'; import { childStop, ownsProcessGroup } from './child-stop.js'; import { @@ -39,6 +40,7 @@ export async function runAgentCli( wrapperLimits?: Partial, signal?: AbortSignal, mode: 'agent' | 'llm' = 'agent', + sidechannel?: SidechannelContext, ): Promise { signal?.throwIfAborted(); if (signal !== undefined && process.platform === 'win32') { @@ -79,21 +81,44 @@ export async function runAgentCli( // Structured provider output carries the authoritative token counts. const args = [...invocation.args]; args.splice(args.length - 1, 0, ...(kind === 'claude' ? ['--output-format', 'json'] : ['--json'])); - return requirePricedUsage(decodeProviderResult(await spawnInvocation(cli, { ...invocation, args }, env, signal), kind), model); + return requirePricedUsage(decodeProviderResult(await spawnInvocation(cli, { ...invocation, args }, env, signal, sidechannel), kind), model); } -function spawnInvocation( +async function spawnInvocation( cli: string, invocation: CliInvocation, env: NodeJS.ProcessEnv, signal?: AbortSignal, + sidechannel?: SidechannelContext, ): Promise { + let writeInput: (bytes: Buffer) => Promise = async () => false; + let canDrive = () => false; + let driven = false; + const channel = sidechannel === undefined ? undefined : await openSidechannel({ + ...sidechannel, + onDrive() { driven = true; sidechannel.onDrive(); }, + }, bytes => writeInput(bytes), () => canDrive()); + if (signal?.aborted) { channel?.close(); signal.throwIfAborted(); } return new Promise((resolve) => { const ownsGroup = ownsProcessGroup(signal); const child = spawn(cli, invocation.args, { - stdio: ['ignore', 'pipe', 'pipe'], env, + stdio: ['pipe', 'pipe', 'pipe'], env, detached: ownsGroup, }); + child.stdin.on('error', () => {}); + if (channel === undefined) child.stdin.end(); + canDrive = () => !child.stdin.destroyed && !child.stdin.writableEnded; + // A pipe cannot be reopened after EOF. Give startup subscribers a bounded + // chance to opt into drive, then let unattended/view-only CLIs read EOF. + const inputTimer = channel === undefined ? undefined : setTimeout(() => { + if (!driven) child.stdin.end(); + }, 100); + writeInput = bytes => new Promise(resolve => { + if (!canDrive()) { resolve(false); return; } + // write(false) still accepts the bytes. The completion callback waits + // until they flush; the sidechannel pauses its reader in the meantime. + child.stdin.write(bytes, error => resolve(!error)); + }); const stop = childStop(child, ownsGroup); const stdout: Buffer[] = []; const stderr: Buffer[] = []; @@ -102,6 +127,8 @@ function spawnInvocation( const finish = (result: WorkerCliResult): void => { if (settled) return; settled = true; + if (inputTimer !== undefined) clearTimeout(inputTimer); + channel?.close(); if (timer !== undefined) clearTimeout(timer); signal?.removeEventListener('abort', onAbort); resolve(result); @@ -124,8 +151,8 @@ function spawnInvocation( }; signal?.addEventListener('abort', onAbort, { once: true }); if (signal?.aborted) onAbort(); - child.stdout.on('data', (chunk: Buffer) => stdout.push(chunk)); - child.stderr.on('data', (chunk: Buffer) => stderr.push(chunk)); + child.stdout.on('data', (chunk: Buffer) => { stdout.push(chunk); channel?.publish(chunk); }); + child.stderr.on('data', (chunk: Buffer) => { stderr.push(chunk); channel?.publish(chunk); }); child.once('error', (error) => finishOnChildExit({ exit_code: null, stdout_tail: Buffer.concat(stdout).toString('utf8'), diff --git a/packages/sdk/src/worker.ts b/packages/sdk/src/worker.ts index 1cbb13e3c..a931d5d65 100644 --- a/packages/sdk/src/worker.ts +++ b/packages/sdk/src/worker.ts @@ -14,6 +14,8 @@ export interface AgentWorkerOptions { workerId: string; pins: Pins; capacity?: number; + dataDir?: string; + onPtyReady?: (path: string) => void; } /** @@ -94,9 +96,13 @@ export class AgentWorker extends EventEmitter { private async execute(dispatch: StepDispatchEvent): Promise { const spec = dispatch.spec as Partial; + let humanIntervention = false; const completed: WorkerCliResult = await withWorkerLease(this.client, dispatch, signal => typeof spec.cli === 'string' && typeof spec.instruction === 'string' - ? runAgentCli(spec.cli, workerInstruction(spec.instruction, dispatch), dispatch.wake_context, spec.model, undefined, signal) + ? runAgentCli(spec.cli, workerInstruction(spec.instruction, dispatch), dispatch.wake_context, spec.model, undefined, signal, 'agent', this.options.dataDir === undefined ? undefined : { + dataDir: this.options.dataDir, runId: dispatch.run_id, stepId: dispatch.step_id, + onReady: this.options.onPtyReady, onDrive: () => { humanIntervention = true; }, + }) : Promise.resolve({ exit_code: null, stdout_tail: '', stderr_tail: 'agent step has no declared CLI' })); const { result, usage } = workerSpend(completed, spec.model); const completionReason = result.exit_code === 0 ? 'success' : 'worker_error'; @@ -124,6 +130,7 @@ export class AgentWorker extends EventEmitter { completionReason, { output, + ...(humanIntervention ? { human_intervention: true } : {}), ...(usage !== undefined ? { usage } : {}), started_pins: dispatch.pins, end_pins: dispatch.pins, diff --git a/packages/sdk/tests/cli-replay.test.ts b/packages/sdk/tests/cli-replay.test.ts index d74ebbb13..58aae3753 100644 --- a/packages/sdk/tests/cli-replay.test.ts +++ b/packages/sdk/tests/cli-replay.test.ts @@ -294,6 +294,16 @@ describe('flows replay', () => { expect(output.stderr[0]).toContain('REFUSED [journal_read_failed]'); }); + it('refuses human-influenced replay unless explicitly allowed', async () => { + const { dataDir, writer } = fixture(); + writer.exec("UPDATE entries SET payload = json_set(payload, '$.human_intervention', json('true')) WHERE entry_type = 'step.completed'"); + writer.close(); + const refused = await replay(dataDir); + expect(refused.code).toBe(2); + expect(refused.stderr.join('\n')).toContain('REFUSED [human_influenced_run] step'); + expect((await replay(dataDir, ['--allow-human-influenced'])).code).toBe(0); + }); + it.each([ [], [RUN_ID, '--at'], [RUN_ID, '--data-dir'], [RUN_ID, '--no-spawn'], [RUN_ID, '--json', '--json'], [RUN_ID, '--at', 'x', '--at', 'y'], diff --git a/packages/sdk/tests/cli.test.ts b/packages/sdk/tests/cli.test.ts index 87389b040..ae8c4dfaf 100644 --- a/packages/sdk/tests/cli.test.ts +++ b/packages/sdk/tests/cli.test.ts @@ -959,6 +959,24 @@ describe('flows run/resume CLI over the journal protocol', () => { expect(output.stderr.join('\n')).not.toContain('journal.read'); }); + it('refuses human-influenced resume with exit 2 and forwards the explicit override', async () => { + const dataDir = temporaryProject('flows-human-'); + await startCliLoopback(dataDir, { + hello: sendOk, + 'run.resume': (ctx, params) => { + if (params['allow_human_influenced'] === true) { + sendResult(ctx, { run_id: 'human-run', status: 'completed', completion_reason: 'success', completed_steps: 1 }); + } else { + ctx.send({ id: ctx.id, ok: false, error: { code: 'human_influenced_run', message: 'step "agent"' } }); + } + }, + }); + const refused = capture(); + expect(await runCli(['resume', '--no-observer-link', '--data-dir', dataDir, 'human-run'], refused.io)).toBe(2); + expect(refused.stderr.join('\n')).toContain('REFUSED [human_influenced_run] step "agent"'); + expect(await runCli(['resume', '--allow-human-influenced', '--no-observer-link', '--data-dir', dataDir, 'human-run'], capture().io)).toBe(0); + }); + it('maps only run_not_found resumes to exit 2', async () => { const dataDir = temporaryProject('flows-resume-'); await startCliLoopback(dataDir, { diff --git a/packages/sdk/tests/pty-sidechannel.test.ts b/packages/sdk/tests/pty-sidechannel.test.ts new file mode 100644 index 000000000..3091577b0 --- /dev/null +++ b/packages/sdk/tests/pty-sidechannel.test.ts @@ -0,0 +1,168 @@ +import { EventEmitter } from 'node:events'; +import { createHash } from 'node:crypto'; +import { mkdtempSync, writeFileSync, rmSync, existsSync } from 'node:fs'; +import { connect, type Socket } from 'node:net'; +import { join } from 'node:path'; +import { afterEach, expect, it } from 'vitest'; +import { AgentWorker } from '../src/worker.js'; +import { openSidechannel } from '../src/pty-sidechannel.js'; +import { runAgentCli } from '../src/worker-cli.js'; +import type { JournalClient } from '../src/journal-client.js'; + +const dirs: string[] = []; +afterEach(() => { for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true, force: true }); }); +function dir() { const path = mkdtempSync('/tmp/q-'); dirs.push(path); return path; } + +it.each(['view', 'drive', 'passthrough', 'none'])('%s attach preserves worker completion and marks only drive', async mode => { + const dataDir = dir(); + const cli = join(dataDir, 'claude'); + writeFileSync(cli, `#!/usr/bin/env node +process.stdin.on('data', b => { process.stdout.write('injected:' + b); process.exit(0); }); +setTimeout(() => process.stdout.write('visible\\n'), 150); +setTimeout(() => process.exit(0), 450); +`, { mode: 0o755 }); + const client = new EventEmitter() as EventEmitter & { workerAttach: () => Promise; stepComplete: (...args: unknown[]) => Promise; stepHeartbeat: () => Promise<{ lease_deadline_ms: number }> }; + client.workerAttach = async () => {}; + client.stepHeartbeat = async () => ({ lease_deadline_ms: Date.now() + 60_000 }); + let finish!: (args: unknown[]) => void; + const done = new Promise(resolve => { finish = resolve; }); + client.stepComplete = async (...args) => { finish(args); }; + let socket: Socket | undefined; + let received = ''; + let socketPath = ''; + const worker = new AgentWorker(client as unknown as JournalClient, { + workerId: 'test', pins: { workspace: [], streams: [] }, dataDir, + onPtyReady(path) { + socketPath = path; + if (mode === 'none') return; + socket = connect(path, () => { + socket!.write('HEL'); + socket!.write(`LO ${mode}\n${mode === 'drive' ? 'operator\n' : 'ignored\n'}`); + }); + socket.on('data', bytes => { received += bytes.toString(); }); + }, + }); + worker.on('error', error => { throw error; }); + await worker.attach(); + client.emit('step.dispatch', { + run_id: 'r', step_id: 's', step_type: 'agent', attempt: 1, + idempotency_key: 'k', lease_id: 'lease', lease_deadline_ms: Date.now() + 60_000, + pins: { workspace: [], streams: [] }, spec: { cli, instruction: 'test' }, + }); + try { + const args = await done; + expect(args[4]).toBe('success'); + const completion = args[5] as { human_intervention?: boolean; output: { stdout_tail: string } }; + expect(completion.human_intervention).toBe(mode === 'drive' ? true : undefined); + expect(completion.output.stdout_tail).toContain(mode === 'drive' ? 'injected:operator' : 'visible'); + if (mode !== 'none' && mode !== 'drive') expect(received).toContain('visible'); + expect(existsSync(socketPath)).toBe(false); + } finally { socket?.destroy(); await worker.close(); } +}); + +it('broken handshake and occupied socket do not affect the sidechannel owner', async () => { + const context = { dataDir: dir(), runId: 'r', stepId: 's', onDrive: () => { throw new Error('must not drive'); } }; + let path = ''; + const first = await openSidechannel({ ...context, onReady: value => { path = value; } }, () => true); + expect(first).toBeDefined(); + try { + expect(await openSidechannel(context, () => true)).toBeUndefined(); + const peer = connect(path); + await new Promise(resolve => { peer.once('close', () => resolve()); peer.write('garbage\n'); }); + expect(existsSync(path)).toBe(true); + } finally { first?.close(); } +}); + +it.each(['none', 'view', 'passthrough', 'incomplete'])('%s subscriber lets an unattended CLI read EOF', async mode => { + const dataDir = dir(); + const cli = join(dataDir, 'claude'); + writeFileSync(cli, `#!/usr/bin/env node +const watchdog = setTimeout(() => process.exit(91), 2000); +process.stdin.resume(); +process.stdin.on('end', () => { clearTimeout(watchdog); process.stdout.write('eof'); }); +`, { mode: 0o755 }); + let peer: Socket | undefined; + let driven = false; + try { + const result = await runAgentCli(cli, 'test', undefined, undefined, undefined, undefined, 'agent', { + dataDir, runId: 'r', stepId: 's', onDrive: () => { driven = true; }, + onReady(path) { + if (mode === 'none') return; + peer = connect(path, () => peer!.write(mode === 'incomplete' ? 'HELLO dri' : `HELLO ${mode}\n`)); + }, + }); + expect(result.exit_code).toBe(0); + expect(result.stdout_tail).toBe('eof'); + expect(driven).toBe(false); + } finally { peer?.destroy(); } +}); + +it('rejects drive after EOF without marking human intervention', async () => { + const dataDir = dir(); + const cli = join(dataDir, 'claude'); + writeFileSync(cli, `#!/usr/bin/env node +process.stdin.resume(); +process.stdin.on('end', () => { + process.stdout.write('eof'); + setTimeout(() => process.exit(0), 300); +}); +`, { mode: 0o755 }); + let peer: Socket | undefined; + let driven = false; + let attempted = false; + let rejectedBeforeExit = false; + try { + const result = await runAgentCli(cli, 'test', undefined, undefined, undefined, undefined, 'agent', { + dataDir, runId: 'r', stepId: 's', onDrive: () => { driven = true; }, + onReady(path) { + peer = connect(path, () => peer!.write('HELLO view\n')); + peer.once('data', () => { + attempted = true; + const late = connect(path, () => late.write('HELLO drive\nignored')); + late.once('close', () => { rejectedBeforeExit = true; }); + }); + }, + }); + expect(result.exit_code).toBe(0); + expect(attempted).toBe(true); + expect(rejectedBeforeExit).toBe(true); + expect(driven).toBe(false); + } finally { peer?.destroy(); } +}); + +it('delivers all drive bytes in order across child stdin backpressure', async () => { + const dataDir = dir(); + const cli = join(dataDir, 'claude'); + // Several pipe buffers, with distinct content to catch loss or retries. + const payload = Buffer.concat(Array.from({ length: 64 }, (_, i) => Buffer.alloc(64 * 1024, i))); + const digest = createHash('sha256').update(payload).digest('hex'); + writeFileSync(cli, `#!/usr/bin/env node +const hash = require('node:crypto').createHash('sha256'); +let count = 0; +setTimeout(() => { + process.stdin.on('data', bytes => { + hash.update(bytes); + count += bytes.length; + if (count >= ${payload.length}) { + process.stdout.write(count + ':' + hash.digest('hex')); + process.exit(0); + } + }); +}, 250); +`, { mode: 0o755 }); + let peer: Socket | undefined; + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), 4000); + try { + const result = await runAgentCli(cli, 'test', undefined, undefined, undefined, controller.signal, 'agent', { + dataDir, runId: 'r', stepId: 's', onDrive() {}, + onReady(path) { + peer = connect(path, () => { peer!.write('HELLO drive\n'); peer!.write(payload); }); + peer.on('error', () => {}); + peer.resume(); + }, + }); + expect(result.exit_code).toBe(0); + expect(result.stdout_tail).toBe(`${payload.length}:${digest}`); + } finally { clearTimeout(timeout); peer?.destroy(); } +});