Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 4 additions & 2 deletions docs/architecture/rfcs/loopx-overall-roadmap-v0.md
Original file line number Diff line number Diff line change
Expand Up @@ -911,8 +911,10 @@ outer Turn and native Goal driver on one binding.

Heartbeat command guidance preserves the selected registry through guard, action
selection, recovery and settlement, independently of the caller's working
directory and runtime root. Conflicting-registry CLI fixtures qualify this
bounded routing repair; they do not qualify host latency or fleet scale.
directory and runtime root, including contract rebuilds for retained selections
and required capability reads. File/SQLite fixtures execute generated settlement
commands from a conflicting-registry directory and verify one-spend replay.
This qualifies bounded routing; it does not qualify host latency or fleet scale.

## 7. Execution and Review Contract

Expand Down
11 changes: 11 additions & 0 deletions docs/development/testing-and-quality.md
Original file line number Diff line number Diff line change
Expand Up @@ -782,6 +782,17 @@ TS RFC 管语义 owner 与跨语言成本,shared-authority RFC 管后端容量

### Budget Failure Decisions

Complete registry routes exposed an existing brief heartbeat regression on the
unchanged 128-character-root fixture. Commit
`fb20c153ba2c2a7376efe68091a76ca470effbde` added those routes: its immediate parent
emitted 10,417 JSON / 8,846 Markdown characters; the same workload then emitted
12,107 / 10,198 against 10,500 / 9,000 ceilings. The additional JSON cost is 1,690
characters across eight command fields and two task-body commands. Keep the
original failure, full paths, fixture, consumer fields and thresholds while
qualifying a separate compaction or justified budget change. This history is
not a waived regression or a frozen SLO pass. An unchanged base failure must be
attributed before it is distinguished from a new candidate regression.

Classify the limit by its owning contract before deciding how to repair a
failure. This applies to output size/structure and latency regression budgets;
it does not grant execution quota, spending, or provider authority.
Expand Down
12 changes: 9 additions & 3 deletions examples/install-local-smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
from datetime import datetime, timedelta, timezone
import json
import os
import shlex
import subprocess
import sys
import tempfile
Expand Down Expand Up @@ -809,7 +810,7 @@ def main() -> int:
assert payload["ok"] is True, payload
assert payload["schema_version"] == "heartbeat_agent_input_v1", payload
expected_quota_guard = (
'loopx --format json --registry "$HOME/.loopx/registry.global.json" '
f'loopx --registry {shlex.quote(str((home / ".loopx" / "registry.global.json").resolve()))} --format json '
'quota should-run --goal-id installer-smoke-goal '
'--turn-instance-id "${LOOPX_TURN:?}"'
)
Expand Down Expand Up @@ -860,8 +861,13 @@ def main() -> int:
canary_payload = json.loads(canary_cli.stdout)
assert canary_payload["cli_bin"] == "loopx-canary", canary_payload
assert "loopx-canary doctor" in canary_payload["cli_preflight"], canary_payload
assert "loopx-canary --format json" in canary_payload["quota_guard_command"], canary_payload
assert "loopx-canary heartbeat-prompt --compact" in canary_payload["task_body"], canary_payload
guard_argv = shlex.split(canary_payload["quota_guard_command"])
assert guard_argv[0] == "loopx-canary", canary_payload
assert guard_argv[guard_argv.index("--format") + 1] == "json", canary_payload
assert guard_argv[guard_argv.index("--registry") + 1] == str(
(home / ".loopx" / "registry.global.json").resolve()
), canary_payload
assert canary_payload["compact_prompt_command"] in canary_payload["task_body"], canary_payload
canary_task_body = canary_payload["task_body"]
# Brief mode renders one bounded guard block: it deliberately omits the
# accountable refresh/spend pair, which belongs to the full and compact
Expand Down
12 changes: 12 additions & 0 deletions loopx/control_plane/quota/live_decision.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,7 @@ def _apply_retained_action_selection_reentry(
Mapping[str, Any] | SchedulerExecutionContextResolution | None
),
turn_instance_id: str | None,
registry_path: Path,
runtime_root: Path,
) -> None:
"""Fence a no-argument reentry with its last explicit Todo choice."""
Expand Down Expand Up @@ -127,6 +128,7 @@ def _apply_retained_action_selection_reentry(
available_capabilities=available_capabilities,
scheduler_execution_context=scheduler_execution_context,
turn_instance_id=turn_instance_id,
registry_path=str(registry_path),
runtime_root=str(runtime_root),
)

Expand Down Expand Up @@ -190,6 +192,7 @@ def _project_turn_start_required_reads(
Mapping[str, Any] | SchedulerExecutionContextResolution | None
),
turn_instance_id: str | None,
registry_path: Path,
runtime_root: Path,
) -> bool:
"""Order evidence before work and report whether the decision changed."""
Expand Down Expand Up @@ -236,6 +239,7 @@ def _project_turn_start_required_reads(
available_capabilities=available_capabilities,
scheduler_execution_context=scheduler_execution_context,
turn_instance_id=turn_instance_id,
registry_path=str(registry_path),
runtime_root=str(runtime_root),
)
return True
Expand Down Expand Up @@ -281,6 +285,8 @@ def _apply_pending_capability_intent_precedence(
Mapping[str, Any] | SchedulerExecutionContextResolution | None
) = None,
turn_instance_id: str | None = None,
registry_path: Path | None = None,
runtime_root: Path | None = None,
) -> bool:
"""Apply intent precedence and report whether the decision changed."""

Expand Down Expand Up @@ -344,6 +350,8 @@ def _apply_pending_capability_intent_precedence(
available_capabilities=available_capabilities,
scheduler_execution_context=scheduler_execution_context,
turn_instance_id=turn_instance_id,
registry_path=str(registry_path) if registry_path is not None else None,
runtime_root=str(runtime_root) if runtime_root is not None else None,
)
return True

Expand Down Expand Up @@ -630,6 +638,7 @@ def build_live_quota_should_run_decision(
available_capabilities=available_capabilities,
scheduler_execution_context=resolved_context,
turn_instance_id=turn_instance_id,
registry_path=registry_path,
runtime_root=runtime_root,
)
remembered_runtime = (payload.get("agent_identity") or {}).get(
Expand All @@ -652,6 +661,7 @@ def build_live_quota_should_run_decision(
available_capabilities=available_capabilities,
scheduler_execution_context=resolved_context,
turn_instance_id=turn_instance_id,
registry_path=registry_path,
runtime_root=runtime_root,
)
hook_dispatch = dispatch_interaction_projection_hooks(interaction_projection_hooks)
Expand All @@ -663,6 +673,8 @@ def build_live_quota_should_run_decision(
available_capabilities=available_capabilities,
scheduler_execution_context=resolved_context,
turn_instance_id=turn_instance_id,
registry_path=registry_path,
runtime_root=runtime_root,
)
interaction = payload.get("interaction_contract")
if isinstance(interaction, dict):
Expand Down
54 changes: 51 additions & 3 deletions tests/control_plane/test_heartbeat_registry_route.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@

from loopx.heartbeat_prompt import build_heartbeat_prompt, build_heartbeat_prompt_error_payload
from loopx.control_plane.heartbeat.budget import build_interface_budget
from loopx.control_plane.quota.settlement import read_heartbeat_settlement


def _route(command: str, registry: Path, runtime: Path) -> None:
Expand Down Expand Up @@ -70,17 +71,28 @@ def _execute(argv: list[str], cwd: Path):
env = dict(os.environ, PYTHONPATH=str(cli.REPO_ROOT))
result = subprocess.run([sys.executable, "-m", "loopx.cli", "--format", "json", *argv],
cwd=cwd, env=env, text=True, capture_output=True, check=False)
assert result.stdout, result.stderr
return result.returncode, json.loads(result.stdout)


def test_cli_generated_guard_selection_and_settlement_ignore_conflicting_cwd_registry(tmp_path):
project, runtime, original, _, _ = journey._source(tmp_path, provider="file")
registry = tmp_path / "selected authority" / "registry.json"
@pytest.mark.parametrize("provider", ["file", "sqlite"])
def test_cli_generated_guard_selection_and_settlement_ignore_conflicting_cwd_registry(tmp_path, provider):
project, runtime, original, _, _ = journey._source(tmp_path, provider=provider)
registry = tmp_path / "selected authority's directory" / "registry.json"
registry.parent.mkdir()
registry.write_text(original.read_text())
conflict = json.loads(original.read_text())
conflict["goals"][0]["coordination"]["registered_agents"] = ["another-worker"]
original.write_text(json.dumps(conflict))
code, remembered = _execute([
"--registry", str(registry), "--runtime-root", str(runtime),
"semantic-preference", "agent", "remember",
"--goal-id", cli.GOAL_ID, "--agent-id", cli.AGENT_ID,
"--key", "review.collaboration", "--statement", "Use the designated reviewer.",
"--source-ref", "owner-message-1", "--source-quote", "Use the designated reviewer.",
"--expected-revision", "none", "--operation-id", "remember-1", "--execute",
], project)
assert code == 0 and remembered["status"] == "applied", remembered
code, prompt = _execute([
"--registry", str(registry), "--runtime-root", str(runtime), "heartbeat-prompt",
"--goal-id", cli.GOAL_ID, "--agent-id", cli.AGENT_ID, "--codex-app", "--full",
Expand Down Expand Up @@ -108,6 +120,12 @@ def test_cli_generated_guard_selection_and_settlement_ignore_conflicting_cwd_reg
code, guard = _execute(shlex.split(selection.replace("{todo_id}", cli.TODO_ID))[1:], project)
assert code == 0, guard
assert guard["selected_todo"]["todo_id"] == cli.TODO_ID
assert any(read["kind"] == "agent_preferences" for read in guard["required_reads"])
identity = guard["heartbeat_receipt"]["settlement_identity"]
# A later no-argument reentry must retain the original choice and authority.
code, guard = _execute(guard_argv, project)
assert code == 0, guard
assert guard["heartbeat_receipt"]["settlement_identity"] == identity
channel = guard["interaction_contract"]["cli_channel"]
for command in channel["next_cli_actions"]:
if command.startswith("loopx "):
Expand All @@ -118,6 +136,36 @@ def test_cli_generated_guard_selection_and_settlement_ignore_conflicting_cwd_reg
_route(step["command_template"], registry, runtime)
assert cli._spend_run_count(runtime) == 0

refresh = next(c for c in channel["next_cli_actions"] if "refresh-state" in c)
replacements = {
"<validated_progress>": "validated_progress", "<scale>": "implementation",
"<outcome>": "outcome_progress",
}
for placeholder, value in replacements.items():
refresh = refresh.replace(placeholder, value)
code, refreshed = _execute(shlex.split(refresh)[1:] + [
"--vision-state", "vision_on_track",
"--vision-summary", "Keep validating the selected work.",
"--vision-acceptance", "The same Turn settles once against the selected authority.",
"--no-global-sync", "--suppress-external-sinks",
], project)
assert code == 0 and refreshed["settlement_result"]["ok"], refreshed
spend = next(c for c in channel["next_cli_actions"] if "spend-slot" in c)
for replay in (False, True):
code, spent = _execute(shlex.split(spend)[1:], project)
assert code == 0 and spent["settlement_result"]["ok"], spent
if replay:
assert spent["idempotent_replay"] and not spent["appended"]
assert cli._spend_run_count(runtime) == 1
code, settled = _execute(guard_argv, project)
assert code == 0 and settled["effective_action"] == "heartbeat_settled_skip", settled
assert settled["heartbeat_receipt"]["settlement_identity"] == identity
readback = read_heartbeat_settlement(
runtime, goal_id=cli.GOAL_ID, agent_id=cli.AGENT_ID,
todo_id=cli.TODO_ID, turn_instance_id=cli.TURN_ID,
)
assert readback is not None and readback.replay_phase.value == "settled"


def test_invalid_selected_registry_does_not_fall_back_to_valid_local_roster(tmp_path):
project, runtime, original = cli._write_fixture(tmp_path)
Expand Down
6 changes: 6 additions & 0 deletions tests/control_plane/test_prompt_upgrade_hook.py
Original file line number Diff line number Diff line change
Expand Up @@ -176,6 +176,12 @@ def test_upgrade_read_projection_preserves_work_authority(tmp_path, monkeypatch,
assert read["command"] == hint["command"]
assert read["ordering"] == "before_work"
assert read["prompt_budget_bytes"] == 1536
# Rebuilding for a required read must not lose the selected authority.
for command in pending["interaction_contract"]["cli_channel"]["next_cli_actions"]:
if command.startswith("loopx "):
argv = shlex.split(command)
assert argv[argv.index("--registry") + 1] == str(registry)
assert argv[argv.index("--runtime-root") + 1] == str(root)
# The dispatch names the hook that produced the read, and the projected hint
# must still carry every field of that read unchanged.
assert read["hook_id"] == "heartbeat.prompt_upgrade"
Expand Down
Loading