From 049c3af34c19278219135f124dc1a061431f4af4 Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 02:17:20 +0800 Subject: [PATCH 1/7] feat(replan): support six-turn benchmark cadence and evidence reuse Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- .../personal-workspace/capability-localization.ts | 4 ++-- benchmark/edgebench/run.py | 4 ++-- benchmark/runtime/codex.py | 4 ++-- benchmark/runtime/harbor.py | 4 ++-- benchmark/runtime/sforge.py | 4 ++-- benchmark/tests/test_sforge_runtime.py | 10 +++++----- benchmark/tests/test_shared_codex_runtime.py | 4 ++-- loopx/capabilities/configuration_ui.py | 6 ++++-- .../todo_replan_cadence/machine_defaults.py | 5 +++-- loopx/cli_commands/registry_admin_configure.py | 4 +++- loopx/control_plane/goals/goal_vision_policy.py | 11 +++++++++++ .../control_plane/work_items/replan_history_codec.py | 8 ++------ loopx/control_plane/work_items/replan_semantics.ts | 2 +- loopx/execution_profile.py | 8 ++++---- tests/control_plane/test_effective_turn_replan.py | 6 +++--- tests/control_plane_ts/replan_semantics.test.ts | 6 ++++++ 16 files changed, 54 insertions(+), 36 deletions(-) diff --git a/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts b/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts index 91aa7f4a19..df82a8e38e 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts +++ b/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts @@ -159,7 +159,7 @@ const fieldCopy: Record = { executor_model: { label: "Model", description: "Optional model for the selected executor. Leave blank to keep the executor's own default." }, executor_reasoning_effort: { label: "Reasoning effort", description: "Optional reasoning effort for the selected executor. Leave blank to keep the executor's own default." }, count_unit: { label: "Review after", description: "Work Turns require accepted settlement; polls and retries do not count.", options: { completed_todos: "Completed Todos", effective_turns: "Settled work Turns" } }, - count: { label: "Number between reviews", description: "From 1 to 5, using the selected unit. Goal overrides take precedence over device defaults." }, + count: { label: "Number between reviews", description: "Settled work Turns: 1–6; completed Todos: 1–5. Goal overrides take precedence over device defaults." }, allowed_domains: { label: "Allowed responsibility domains", description: "Enter one bounded, public-safe domain per line." }, coordinator_agent_id: { label: "Coordinator Agent", description: "Use an already registered Agent id; leave blank to disable coordination." }, enabled: { label: "Enabled" }, @@ -191,7 +191,7 @@ const fieldCopy: Record = { executor_model: { label: "模型", description: "所选执行器使用的模型,可留空;留空表示沿用执行器自身的默认模型。" }, executor_reasoning_effort: { label: "推理档位", description: "所选执行器使用的推理档位,可留空;留空表示沿用执行器自身的默认档位。" }, count_unit: { label: "复核计数依据", description: "有效工作 Turn 须完成结算;轮询和重复重试不计数。", options: { completed_todos: "已完成 Todo", effective_turns: "已结算工作 Turn" } }, - count: { label: "两次复核间的数量", description: "按所选单位计数,范围 1–5;Goal 显式设置优先于设备默认值。" }, + count: { label: "两次复核间的数量", description: "已结算工作 Turn 为 1–6,已完成 Todo 为 1–5;Goal 显式设置优先于设备默认值。" }, allowed_domains: { label: "允许的职责域", description: "每行填写一个有边界、可公开的职责域。" }, coordinator_agent_id: { label: "协调 Agent", description: "填写一个已经注册的 Agent ID;留空表示关闭协调。" }, enabled: { label: "启用" }, diff --git a/benchmark/edgebench/run.py b/benchmark/edgebench/run.py index b99a41eca1..3b26dc7c74 100644 --- a/benchmark/edgebench/run.py +++ b/benchmark/edgebench/run.py @@ -109,8 +109,8 @@ def main(argv=None): parser.add_argument("--task-entry", choices=TASK_ENTRIES, help="Heartbeat default: loopx-planned; seeded-todo is an explicit ablation") cadence = parser.add_mutually_exclusive_group() - cadence.add_argument("--replan-after-turns", type=int, choices=range(1, 6), - help="Heartbeat default: 3 settled effective work Turns") + cadence.add_argument("--replan-after-turns", type=int, choices=range(1, 7), + help="Heartbeat default: 6 settled effective work Turns") cadence.add_argument("--replan-after-todos", type=int, choices=range(1, 6), help="Explicit completed-Todo cadence ablation for heartbeat profiles") parser.add_argument("--eval-interval", type=int, diff --git a/benchmark/runtime/codex.py b/benchmark/runtime/codex.py index 1d27e69e50..25bae55f13 100644 --- a/benchmark/runtime/codex.py +++ b/benchmark/runtime/codex.py @@ -9,8 +9,8 @@ from pathlib import Path -# Benchmarks deliberately review earlier than the product default of five. -DEFAULT_REPLAN_AFTER_TURNS = 3 +# New benchmark runs review less often; the product default remains five. +DEFAULT_REPLAN_AFTER_TURNS = 6 MODES = ("plain", "native-goal", "heartbeat", "turn", "loopx-goal") diff --git a/benchmark/runtime/harbor.py b/benchmark/runtime/harbor.py index cd20668b87..d6051842a5 100644 --- a/benchmark/runtime/harbor.py +++ b/benchmark/runtime/harbor.py @@ -84,8 +84,8 @@ def __init__( raise ValueError("Choose replan_after_turns or replan_after_todos, not both") if replan_after_turns is not None: if (type(replan_after_turns) is not int or - not 1 <= replan_after_turns <= 5): - raise ValueError("replan_after_turns must be an integer between 1 and 5") + not 1 <= replan_after_turns <= 6): + raise ValueError("replan_after_turns must be an integer between 1 and 6") if not self.execution.uses_loopx: raise ValueError("replan_after_turns requires a LoopX execution mode") if replan_after_turns is None and replan_after_todos is None and self.execution.uses_loopx: diff --git a/benchmark/runtime/sforge.py b/benchmark/runtime/sforge.py index 3f2c12a928..c49a36ff6d 100644 --- a/benchmark/runtime/sforge.py +++ b/benchmark/runtime/sforge.py @@ -118,8 +118,8 @@ def __init__(self, config, *, profile: str, cwd: str, raise ValueError("replan_after_todos requires a heartbeat profile") if replan_after_turns is not None: if (type(replan_after_turns) is not int or - not 1 <= replan_after_turns <= 5): - raise ValueError("replan_after_turns must be an integer between 1 and 5") + not 1 <= replan_after_turns <= 6): + raise ValueError("replan_after_turns must be an integer between 1 and 6") if not profile.startswith("heartbeat-"): raise ValueError("replan_after_turns requires a heartbeat profile") if (replan_after_turns is None and replan_after_todos is None diff --git a/benchmark/tests/test_sforge_runtime.py b/benchmark/tests/test_sforge_runtime.py index bbed53a38c..681cc8ff71 100644 --- a/benchmark/tests/test_sforge_runtime.py +++ b/benchmark/tests/test_sforge_runtime.py @@ -174,7 +174,7 @@ async def installed(self, environment): env = worker.runtime._worker_env(cwd="/task") assert float(env["LOOPX_CODEX_TURN_TIMEOUT_SEC"]) == expected assert worker.runtime.scheduler_timeout == total - expected_cadence = ({"replan_after_effective_turns": turns or 3} + expected_cadence = ({"replan_after_effective_turns": turns or 6} if profile.startswith("heartbeat-") else {"replan_after_completed_todos": 3}) assert worker.runtime._replan_receipt() == expected_cadence @@ -396,7 +396,7 @@ async def installed(self, environment): assert json.loads((tmp_path / "worker-profile.json").read_text())["task_entry"] == entry @pytest.mark.parametrize('profile', ['heartbeat-resume', 'heartbeat-explore']) @pytest.mark.parametrize('enabled', [False, True]) -@pytest.mark.parametrize('cadence', [None, 2]) +@pytest.mark.parametrize('cadence', [None, 2, 6]) def test_envelope_treatment_reaches_shared_worker_and_receipts(tmp_path, monkeypatch, profile, enabled, cadence): pytest.importorskip('sforge') pytest.importorskip('harbor') @@ -413,7 +413,7 @@ async def installed(self, environment): env = worker.runtime._worker_env(cwd='/task') assert env.get('LOOPX_TURN_ENVELOPE') == ('1' if enabled else None) assert worker.runtime.execution.turn_envelope is enabled - assert worker.runtime.replan_after_turns == (cadence or 3) + assert worker.runtime.replan_after_turns == (cadence or 6) receipt = json.loads((tmp_path / 'worker-profile.json').read_text()) assert receipt['outer_resume'] is False assert receipt.get('turn_envelope') is (True if enabled else None) @@ -462,7 +462,7 @@ def test_edgebench_rejects_invalid_profile_settings_before_creating_trial(tmp_pa @pytest.mark.parametrize("cadence_args,field,count", [ - ([], "replan_after_effective_turns", 3), + ([], "replan_after_effective_turns", 6), (["--replan-after-turns", "2"], "replan_after_effective_turns", 2), (["--replan-after-todos", "3"], "replan_after_completed_todos", 3), ]) @@ -519,7 +519,7 @@ def stop_before_solver(**kwargs): assert receipt["status"] == "runner_failed" -@pytest.mark.parametrize("value", [0, 6, True, 2.5, "3"]) +@pytest.mark.parametrize("value", [0, 7, True, 2.5, "3"]) def test_effective_turn_cadence_rejects_invalid_values_before_install(tmp_path, monkeypatch, value): pytest.importorskip("sforge") pytest.importorskip("harbor") diff --git a/benchmark/tests/test_shared_codex_runtime.py b/benchmark/tests/test_shared_codex_runtime.py index 5765fc57aa..edc3dddbd0 100644 --- a/benchmark/tests/test_shared_codex_runtime.py +++ b/benchmark/tests/test_shared_codex_runtime.py @@ -359,12 +359,12 @@ def test_baseline_and_treatment_use_same_harbor_entry(tmp_path): assert env["LOOPX_EXECUTION_MODE"] == mode assert env["LOOPX_PROJECT"] == "/workspace" assert env["MODEL_NAME"] == "fixture" - assert agent.replan_after_turns == (3 if agent.execution.uses_loopx else None) + assert agent.replan_after_turns == (6 if agent.execution.uses_loopx else None) @pytest.mark.parametrize("existing", [False, True]) @pytest.mark.parametrize("settings,field,value", [ - ({}, "replan_after_effective_turns", 3), + ({}, "replan_after_effective_turns", 6), ({"replan_after_turns": 2}, "replan_after_effective_turns", 2), ({"replan_after_todos": 3}, "replan_after_completed_todos", 3), ]) diff --git a/loopx/capabilities/configuration_ui.py b/loopx/capabilities/configuration_ui.py index fc45b53e4f..b88a7ad7ef 100644 --- a/loopx/capabilities/configuration_ui.py +++ b/loopx/capabilities/configuration_ui.py @@ -5,6 +5,7 @@ from typing import Any from ..configuration_transaction import configuration_payload_revision +from ..control_plane.goals.goal_vision_policy import MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD from .progress_review.policy import ( PROGRESS_REVIEW_MAX_DRIFT_THRESHOLD, @@ -110,8 +111,9 @@ def capability_configuration_editor( _field("count_unit", "Count between reviews", "select", options=["completed_todos", "effective_turns"], required=True, description="Existing completed-Todo values keep their units until explicitly changed."), - _field("count", "Review interval", "number", minimum=1, maximum=5, required=True, - description="Effective Turns require accepted work settlement. Retries and observation-only polls do not count."), + _field("count", "Review interval", "number", minimum=1, + maximum=MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD, required=True, + description="Effective Turns: 1–6; completed Todos: 1–5. Effective Turns require accepted work settlement. Retries and observation-only polls do not count."), ], }, "periodic_report": { diff --git a/loopx/capabilities/todo_replan_cadence/machine_defaults.py b/loopx/capabilities/todo_replan_cadence/machine_defaults.py index a6d286d05f..82abeccdd6 100644 --- a/loopx/capabilities/todo_replan_cadence/machine_defaults.py +++ b/loopx/capabilities/todo_replan_cadence/machine_defaults.py @@ -9,6 +9,7 @@ from ...control_plane.goals.goal_vision_policy import ( DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD, normalize_completed_todo_replan_threshold, + normalize_effective_turn_replan_threshold, ) from ..machine_configuration.contract import ( MACHINE_CONFIGURATION_SCHEMA, @@ -45,8 +46,8 @@ def normalize_replan_cadence_configuration(raw: Mapping[str, Any]) -> dict[str, count = raw["count"] if unit is ReplanCadenceUnit.COMPLETED_TODOS: count = normalize_completed_todo_replan_threshold(count) - elif type(count) is not int or not 1 <= count <= 5: - raise ValueError("effective Turn count must be an integer from 1 to 5") + else: + count = normalize_effective_turn_replan_threshold(count) return {"count_unit": unit.value, "count": count} diff --git a/loopx/cli_commands/registry_admin_configure.py b/loopx/cli_commands/registry_admin_configure.py index 0c1e1e5cbb..f23af44788 100644 --- a/loopx/cli_commands/registry_admin_configure.py +++ b/loopx/cli_commands/registry_admin_configure.py @@ -3,6 +3,7 @@ import argparse from ..execution_profile import TURN_GRANULARITY_CHOICES +from ..control_plane.goals.goal_vision_policy import MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD from ..orchestration import EXPLORE_HARNESS_PROFILES from .registry_admin_peer import ( register_peer_runtime_arguments, @@ -54,7 +55,8 @@ def register_configure_goal_command(subparsers: argparse._SubParsersAction) -> N ), ) configure_goal_parser.add_argument( - "--execution-replan-after-turns", type=int, choices=range(1, 6), + "--execution-replan-after-turns", type=int, + choices=range(1, MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD + 1), help="Review direction after this many settled work Turns, independently of Todo completion (product default: 5).", ) configure_goal_parser.add_argument( diff --git a/loopx/control_plane/goals/goal_vision_policy.py b/loopx/control_plane/goals/goal_vision_policy.py index 5b475f7d90..67f96f3259 100644 --- a/loopx/control_plane/goals/goal_vision_policy.py +++ b/loopx/control_plane/goals/goal_vision_policy.py @@ -15,12 +15,23 @@ class GoalVisionAdvancementPolicy(str, Enum): # Default review cadence counts settled work Turns, not Todo size. DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD = 5 +MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD = 6 # Explicit completed-Todo cadence remains bounded by the retained evidence # window. This is a legacy-unit limit, not the product default. COMPLETED_TODO_CHAIN_REPLAN_THRESHOLD = 5 +def normalize_effective_turn_replan_threshold(value: Any) -> int: + # Effective Turns use settlement receipts rather than the five-Todo window. + if type(value) is not int or not 1 <= value <= MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD: + raise ValueError( + "effective Turn count must be an integer from 1 to " + f"{MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD}" + ) + return value + + def normalize_completed_todo_replan_threshold(value: Any) -> int: # The durable agent projection retains five completions. Larger thresholds # would be unreachable without widening that evidence window. diff --git a/loopx/control_plane/work_items/replan_history_codec.py b/loopx/control_plane/work_items/replan_history_codec.py index fbba558d9b..c4e6ad26ae 100644 --- a/loopx/control_plane/work_items/replan_history_codec.py +++ b/loopx/control_plane/work_items/replan_history_codec.py @@ -15,6 +15,7 @@ from ..effect_runtime import MAX_REQUEST_BYTES, EffectRuntimeRejected, effect_runtime_result from ..runtime.time import parse_timestamp +from ..goals.goal_vision_policy import normalize_effective_turn_replan_threshold from ..todos.contract import normalize_todo_claimed_by, normalize_todo_id, normalize_todo_id_list from ..todos.resume_planning import build_todo_resume_planning_request @@ -36,12 +37,7 @@ def effective_turn_cadence_context( threshold = profile.get("replan_after_effective_turns") if threshold is None: return None - if ( - isinstance(threshold, bool) - or not isinstance(threshold, int) - or not 1 <= threshold <= 5 - ): - raise ValueError("replan_after_effective_turns must be an integer from 1 to 5") + threshold = normalize_effective_turn_replan_threshold(threshold) if runtime_root is None or not goal.get("id"): raise ValueError("effective Turn cadence requires the Goal settlement runtime") if goal_ref is None and goal.get("goal_instance_id"): diff --git a/loopx/control_plane/work_items/replan_semantics.ts b/loopx/control_plane/work_items/replan_semantics.ts index 8595ccbff8..571b5dc255 100644 --- a/loopx/control_plane/work_items/replan_semantics.ts +++ b/loopx/control_plane/work_items/replan_semantics.ts @@ -201,7 +201,7 @@ function writebackProjection(required: SemanticOutcome[], externalReview: boolea vision_authoring: visionAuthoringContract(), required_fields: ["vision_patch.acceptance_summary", "path_delta.outcome", "path_delta.evidence_refs"], path_outcomes: [...FRESH_PATH_DISPOSITIONS], - rule: "Author the JSON file from observed evidence, then execute the bound refresh and spend. This path requires an acceptance summary and evidence-linked path outcome; an unchanged reason alone is insufficient. Other required_any_of exits remain subject to their typed contracts.", + rule: "Review available evidence against the current source and acceptance first. Reuse applicable evidence; run additional checks when evidence is missing, stale or insufficient to decide the triggered obligation. Do not repeat unchanged checks solely to create a new review artifact. Author an acceptance summary and evidence-linked path outcome, then execute the bound refresh and spend; an unchanged reason alone is insufficient. Explicit validation gates and other required_any_of exits retain their typed contracts.", }, }; } diff --git a/loopx/execution_profile.py b/loopx/execution_profile.py index 94cbd750a8..987d9c92ea 100644 --- a/loopx/execution_profile.py +++ b/loopx/execution_profile.py @@ -6,6 +6,7 @@ from .control_plane.goals.goal_vision_policy import ( completed_todo_replan_threshold, normalize_completed_todo_replan_threshold, + normalize_effective_turn_replan_threshold, ) from .control_plane.work_items.delivery_outcome import DeliveryOutcome @@ -156,10 +157,9 @@ def compact_execution_profile(value: Any) -> dict[str, Any]: if "replan_after_effective_turns" in value and "replan_after_completed_todos" in value: raise ValueError("choose one review cadence unit") if "replan_after_effective_turns" in value: - effective = value["replan_after_effective_turns"] - if isinstance(effective, bool) or not isinstance(effective, int) or not 1 <= effective <= 5: - raise ValueError("replan_after_effective_turns must be an integer from 1 to 5") - profile["replan_after_effective_turns"] = effective + profile["replan_after_effective_turns"] = normalize_effective_turn_replan_threshold( + value["replan_after_effective_turns"] + ) if "replan_after_completed_todos" in value: # Retain explicit legacy-unit overrides, including five. Omitting one diff --git a/tests/control_plane/test_effective_turn_replan.py b/tests/control_plane/test_effective_turn_replan.py index 58b7037b59..3b30e49cef 100644 --- a/tests/control_plane/test_effective_turn_replan.py +++ b/tests/control_plane/test_effective_turn_replan.py @@ -29,7 +29,7 @@ @pytest.mark.parametrize("threshold_override,device_count,threshold", [ - (None, None, 5), (2, 3, 2), (None, 3, 3), + (None, None, 5), (2, 3, 2), (None, 3, 3), (6, None, 6), (None, 6, 6), ]) def test_open_todo_settled_turn_cadence_and_evidence_linked_review( tmp_path: Path, threshold_override: int | None, device_count: int | None, @@ -289,11 +289,11 @@ def periodic(): ) -@pytest.mark.parametrize("invalid", [0, 6, True, 2.5, "2"]) +@pytest.mark.parametrize("invalid", [0, 7, True, 2.5, "2"]) def test_effective_cadence_rejects_invalid_units(invalid): from loopx.execution_profile import compact_execution_profile - with pytest.raises(ValueError, match="integer from 1 to 5"): + with pytest.raises(ValueError, match="integer from 1 to 6"): compact_execution_profile({"replan_after_effective_turns": invalid}) diff --git a/tests/control_plane_ts/replan_semantics.test.ts b/tests/control_plane_ts/replan_semantics.test.ts index 47000165e3..ec1726df59 100644 --- a/tests/control_plane_ts/replan_semantics.test.ts +++ b/tests/control_plane_ts/replan_semantics.test.ts @@ -112,6 +112,12 @@ for (const reviewKind of ["long_todo_chain", "periodic_review_due"]) { assert.match(String(projection.cli_semantic_args), /--agent-vision-json/); assert.equal(projectReplanSemantics({operation: "qualify", obligation: chain, agent_vision: vision}).accepted, true); + // Reusing applicable work evidence can keep the route without inventing + // another probe or Todo; a current evidence-linked decision is still needed. + assert.equal(projectReplanSemantics({operation: "qualify", obligation: chain, + agent_vision: {...vision, path_delta: {...vision.path_delta, outcome: "no_change"}}, + observation_delta: {delta_kinds: [], evidence_novel: false}}).accepted, true); + assert.match(String((projection.writeback_contract as JsonObject).rule), /Reuse applicable evidence/); for (const outcome of ["new_surface", "new_hypothesis", "new_probe_family", "new_runnable_successor"]) { assert.equal(projectReplanSemantics({operation: "qualify", obligation: chain, observation_delta: {delta_kinds: [outcome]}}).accepted, true); From 3d5aed156637b746087aa09b3c93d5287b3da177 Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 02:17:26 +0800 Subject: [PATCH 2/7] docs(replan): explain six-turn benchmark defaults and review boundaries Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- benchmark/edgebench/README.md | 5 +++-- benchmark/runtime/RUNTIME.md | 20 ++++++++++++-------- benchmark/runtime/SETTINGS.md | 25 +++++++++++++++---------- docs/quota-allocation.md | 8 ++++++-- 4 files changed, 36 insertions(+), 22 deletions(-) diff --git a/benchmark/edgebench/README.md b/benchmark/edgebench/README.md index c87f22849c..51be117ee1 100644 --- a/benchmark/edgebench/README.md +++ b/benchmark/edgebench/README.md @@ -368,9 +368,10 @@ and credentials belong outside the public repository. ## Default replan cadence -New `heartbeat-resume` and `heartbeat-explore` trials replan after **3 settled +New `heartbeat-resume` and `heartbeat-explore` trials replan after **6 settled effective work Turns** by default. Use `--replan-after-turns N` to set a different -count (1–5), or `--replan-after-todos 3` for the previous completed-Todo cadence +count (1–6); `--replan-after-turns 3` retains the previous benchmark default. +Use `--replan-after-todos 3` for the previous completed-Todo cadence as an explicit ablation. The two options are mutually exclusive and require a heartbeat profile. This default is independent of `--task-entry`, feedback mode, and `--turn-envelope`. Official, single and native-goal profiles are unchanged. diff --git a/benchmark/runtime/RUNTIME.md b/benchmark/runtime/RUNTIME.md index 6b713bcee9..ef20db5a56 100644 --- a/benchmark/runtime/RUNTIME.md +++ b/benchmark/runtime/RUNTIME.md @@ -30,7 +30,7 @@ agents: codex_sandbox: danger-full-access turn_timeout_sec: null scheduler_timeout_sec: 5080 - replan_after_turns: 3 + replan_after_turns: 6 ``` For an explicitly selected **heartbeat-only** context experiment, add @@ -266,20 +266,24 @@ by selecting this option for a new run. ### Default effective-Turn cadence Harbor LoopX modes (`heartbeat`, `turn`, `loopx-goal`) default to -`replan_after_turns: 3`; native EdgeBench `heartbeat-resume` and -`heartbeat-explore` default to `--replan-after-turns 3`. This passes the existing Goal option +`replan_after_turns: 6`; native EdgeBench `heartbeat-resume` and +`heartbeat-explore` default to `--replan-after-turns 6`. This passes the existing Goal option `--execution-replan-after-turns` and verifies the persisted `replan_after_effective_turns` value before execution. The shared TypeScript control plane still owns which settled work Turns count; adapters do not count records or completed Todos themselves. -Omitting both cadence options now selects three effective work Turns instead -of three completed Todos. Idle wakes, tool calls and the planning checkpoint do +Omitting both cadence options selects six effective work Turns; the +product default remains five. The previous benchmark default was three effective work Turns. +Idle wakes, tool calls and the planning checkpoint do not count as effective work Turns; this is a deterministic threshold rather than a per-wake probability, and other replan triggers can act sooner. -To retain the old cadence in a new trial, pass `replan_after_todos: 3` in Harbor -or `--replan-after-todos 3` in EdgeBench. Explicit Turn/Todo settings are mutually -exclusive and accept counts from one through five. Resolved runtime and worker +To retain the previous benchmark default in a new trial, pass +`replan_after_turns: 3` in Harbor or `--replan-after-turns 3` in EdgeBench. +For completed-Todo cadence, use `replan_after_todos` / `--replan-after-todos`. +Explicit Turn/Todo settings are mutually +exclusive; effective-Turn counts accept one through six and completed-Todo counts +accept one through five. Resolved runtime and worker receipts name the selected unit even when no flag was supplied. Non-LoopX profiles retain their existing behavior. No active attempt, task, scoring, feedback, spawn permission, or total-budget change is implied. diff --git a/benchmark/runtime/SETTINGS.md b/benchmark/runtime/SETTINGS.md index 45fd6bd922..817fa5e64a 100644 --- a/benchmark/runtime/SETTINGS.md +++ b/benchmark/runtime/SETTINGS.md @@ -1,8 +1,7 @@ # Runner defaults and controlled ablations New LoopX executions default to **loopx-planned** task entry and replanning -after **3 settled effective work Turns**, deliberately earlier than the product -default of five. This applies to +after **6 settled effective work Turns**, while the product default remains five. This applies to Harbor heartbeat, Turn and LoopX Goal modes, and EdgeBench heartbeat-resume and heartbeat-explore profiles. Planned entry replaces the seeded-todo default; pass `task_entry: seeded-todo` / `--task-entry seeded-todo` to retain the previous @@ -22,16 +21,20 @@ use matched repetitions to test its effect. | Task entry | LoopX modes: loopx-planned | Record planned or seeded for every arm | | Explore | Off in heartbeat-resume; on in heartbeat-explore | Keep resume as the reference | | Turn envelope | Off | Enable only in its ablation | -| Replan cadence | 3 settled effective work Turns | Use explicit completed-Todo cadence or another Turn count for ablation | +| Replan cadence | 6 settled effective work Turns | Use explicit completed-Todo cadence or another Turn count for ablation | | Iteration context | Harbor: fresh; EdgeBench heartbeat: resume | Freeze the provider and context within a comparison | | Evaluator feedback | EdgeBench: best-only (strict new-best snapshot notifications) | Native and blind remain explicit controls; freeze feedback mode for every comparison | | Model and effort | Caller-selected | Pin both; never infer them from a profile name | | Time and sampling | EdgeBench task defaults in [task settings](../edgebench/README.md#trial-timeouts); explicit flags override | Pin resolved seconds in the study manifest | `replan_after_turns` counts settled effective work turns through the shared -control-plane contract, not tool calls or idle heartbeat wakes. The threshold is 3 by +control-plane contract, not tool calls or idle heartbeat wakes. The threshold is 6 by default, independent of task entry and TurnEnvelope. This is a deterministic threshold, not a random per-wake probability; other replan triggers may act sooner. +An accepted replan resets the same Agent's periodic and repeated-progress history +windows. A failed or unaccepted replan does not; current acceptance gaps remain +independent triggers. Six reduces periodic interruptions relative to the previous +default of three, but does not promise lower total replan time or better scores. Task timeouts bound attempts, not a requirement to consume every second. ## Recommended small study @@ -41,11 +44,11 @@ These are recommendations, not automatically launched experiments. | Arm | EdgeBench flags relative to the reference | Question | | --- | --- | --- | -| Reference | `--worker heartbeat-resume --task-entry loopx-planned --replan-after-turns 3` | Planned entry with effective-turn replanning | +| Reference | `--worker heartbeat-resume --task-entry loopx-planned --replan-after-turns 6` | Planned entry with effective-turn replanning | | Seeded | Replace only `--task-entry` with `seeded-todo` | Does initial task decomposition help? | | Explore | Replace only `--worker` with `heartbeat-explore` | Are recorded evidence and subsequent route choices useful? | | Short envelope | Add `--turn-envelope` | Does progressive context loading reduce overhead without losing decisions? | -| Todo cadence | Replace `--replan-after-turns 3` with `--replan-after-todos 3` | Does effective-turn cadence avoid postponing replans on long Todos? | +| Todo cadence | Replace `--replan-after-turns 6` with `--replan-after-todos 3` | How do units and thresholds change interruptions? This is a combined ablation. | | No LoopX | `--worker official`; omit LoopX-specific flags | What is the net effect of the whole LoopX treatment? | The no-LoopX comparison changes several mechanisms; do not attribute its delta @@ -64,7 +67,7 @@ python -m benchmark.edgebench.run \ --task portfolio_risk_calibration --tasks-dir "$TASKS_DIR" \ --log-dir "$RUNS_DIR" --run-id "$NEW_ATTEMPT_ID" \ --worker heartbeat-resume --task-entry loopx-planned \ - --replan-after-turns 3 --feedback best-only \ + --replan-after-turns 6 --feedback best-only \ --model "$MODEL" --effort xhigh --timeout 43200 --eval-interval 300 \ --judge-url "$JUDGE_URL" --api-proxy-url "$API_PROXY_URL" ``` @@ -78,8 +81,10 @@ recorded service setting, not inherited from another task. Harbor uses the same task-entry owner. In the agent kwargs, set `execution_mode: heartbeat`, `task_entry: loopx-planned`, -`replan_after_turns: 3`, `iteration_context: resume` and `turn_envelope: false` -for an equivalent mechanism reference. For Todo cadence, remove +`replan_after_turns: 6`, `iteration_context: resume` and `turn_envelope: false` +for an equivalent mechanism reference. For the previous benchmark default, set +`replan_after_turns: 3`. +For Todo cadence, remove `replan_after_turns` and set `replan_after_todos: 3`; the two cadence fields are mutually exclusive. The named Explore profile in this matrix is EdgeBench-specific; do not assume an equivalent Harbor kwargs switch. Keep the dataset, provider and validation @@ -98,6 +103,6 @@ cannot isolate an individual fix. Old attempts are immutable references. A rerun after multiple fixes measures the combined revision change unless each fix has a matched control. Roll back the entry default through explicit seeded entry and the cadence default through -`--replan-after-todos 3` / `replan_after_todos: 3` in a new attempt. Do not rewrite +`--replan-after-turns 3` / `replan_after_turns: 3` in a new attempt. Do not rewrite old receipts or change an active worker. None of these settings grants model, submission, credential or launch authority. diff --git a/docs/quota-allocation.md b/docs/quota-allocation.md index 5ed9ad90b6..d7a3f6a483 100644 --- a/docs/quota-allocation.md +++ b/docs/quota-allocation.md @@ -93,7 +93,7 @@ depending on the executor: The product defaults to review after **5 settled effective work Turns** in both standard and fine-grained modes. This replaces the previous default of five completed Todos, so review can occur while a long Todo remains open. Benchmark -runners deliberately use **3** effective Turns for earlier experimental review. +runners configure **6** effective Turns to reduce periodic interruptions. TurnEnvelope is not required. These are deterministic thresholds, not random averages or counts of model messages, tool calls, or heartbeat wakeups. @@ -104,12 +104,16 @@ writebacks and missing settlement receipts cannot. At the threshold, the next quota evaluation creates `periodic_review_due` through the existing replan path. An evidence-linked review may retain the current approach; it does not require inventing a new plan, completing the open Todo, or declaring the Goal achieved. +Review evidence from settled work for current source and acceptance applicability +before running another probe. Reuse applicable evidence; missing, stale or +insufficient evidence needs targeted verification. Do not repeat unchanged checks +only to produce another review artifact. Explicit validation gates still apply. The cadence does not interrupt a host, schedule a Turn, spend quota or grant authority. The executor must settle work and re-enter quota before starting the next work Turn for the review obligation to take effect. ```bash -# Preview, apply, and read back an explicit Goal threshold (supported: 1–5). +# Preview, apply, and read back an explicit Goal threshold (supported: 1–6). loopx configure-goal --goal-id example --execution-replan-after-turns 5 loopx configure-goal --goal-id example --execution-replan-after-turns 5 --execute loopx configure-goal --goal-id example From b9fdb0f0380562411078382dfee9d4f1a7392f5a Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 02:22:12 +0800 Subject: [PATCH 3/7] refine(replan): keep evidence-first guidance within packet budgets Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- loopx/control_plane/work_items/replan_semantics.ts | 2 +- tests/control_plane_ts/replan_semantics.test.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/loopx/control_plane/work_items/replan_semantics.ts b/loopx/control_plane/work_items/replan_semantics.ts index 571b5dc255..c150dd370f 100644 --- a/loopx/control_plane/work_items/replan_semantics.ts +++ b/loopx/control_plane/work_items/replan_semantics.ts @@ -201,7 +201,7 @@ function writebackProjection(required: SemanticOutcome[], externalReview: boolea vision_authoring: visionAuthoringContract(), required_fields: ["vision_patch.acceptance_summary", "path_delta.outcome", "path_delta.evidence_refs"], path_outcomes: [...FRESH_PATH_DISPOSITIONS], - rule: "Review available evidence against the current source and acceptance first. Reuse applicable evidence; run additional checks when evidence is missing, stale or insufficient to decide the triggered obligation. Do not repeat unchanged checks solely to create a new review artifact. Author an acceptance summary and evidence-linked path outcome, then execute the bound refresh and spend; an unchanged reason alone is insufficient. Explicit validation gates and other required_any_of exits retain their typed contracts.", + rule: "First reuse observed evidence valid for current source/acceptance; probe missing, stale or insufficient evidence. Write a JSON file with acceptance summary and evidence-linked path outcome, then run bound refresh and spend. An unchanged reason alone is insufficient. Explicit validation gates and other required_any_of contracts apply.", }, }; } diff --git a/tests/control_plane_ts/replan_semantics.test.ts b/tests/control_plane_ts/replan_semantics.test.ts index ec1726df59..34d96f669a 100644 --- a/tests/control_plane_ts/replan_semantics.test.ts +++ b/tests/control_plane_ts/replan_semantics.test.ts @@ -117,7 +117,7 @@ for (const reviewKind of ["long_todo_chain", "periodic_review_due"]) { assert.equal(projectReplanSemantics({operation: "qualify", obligation: chain, agent_vision: {...vision, path_delta: {...vision.path_delta, outcome: "no_change"}}, observation_delta: {delta_kinds: [], evidence_novel: false}}).accepted, true); - assert.match(String((projection.writeback_contract as JsonObject).rule), /Reuse applicable evidence/); + assert.match(String((projection.writeback_contract as JsonObject).rule), /First reuse observed evidence valid for current source\/acceptance/); for (const outcome of ["new_surface", "new_hypothesis", "new_probe_family", "new_runnable_successor"]) { assert.equal(projectReplanSemantics({operation: "qualify", obligation: chain, observation_delta: {delta_kinds: [outcome]}}).accepted, true); From 848fd3870d3c66a5661c168e83adaf89bf5668fb Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 02:50:16 +0800 Subject: [PATCH 4/7] fix(replan): unify product and benchmark defaults at six Turns Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- benchmark/runtime/codex.py | 2 +- .../todo_replan_cadence/machine_defaults.py | 2 +- loopx/cli_commands/registry_admin_configure.py | 2 +- loopx/control_plane/goals/goal_vision_policy.py | 2 +- .../test_machine_configuration_goal_defaults.py | 10 +++++----- tests/control_plane/test_effective_turn_replan.py | 4 ++-- 6 files changed, 11 insertions(+), 11 deletions(-) diff --git a/benchmark/runtime/codex.py b/benchmark/runtime/codex.py index 25bae55f13..5ca9764a50 100644 --- a/benchmark/runtime/codex.py +++ b/benchmark/runtime/codex.py @@ -9,7 +9,7 @@ from pathlib import Path -# New benchmark runs review less often; the product default remains five. +# New benchmark runs and the product both default to six settled work Turns. DEFAULT_REPLAN_AFTER_TURNS = 6 diff --git a/loopx/capabilities/todo_replan_cadence/machine_defaults.py b/loopx/capabilities/todo_replan_cadence/machine_defaults.py index 82abeccdd6..5ca81990a6 100644 --- a/loopx/capabilities/todo_replan_cadence/machine_defaults.py +++ b/loopx/capabilities/todo_replan_cadence/machine_defaults.py @@ -108,7 +108,7 @@ def todo_replan_cadence_machine_configuration_namespace() -> ( title="Goal review cadence", description=( "Live review threshold for Goals without an explicit cadence override. " - "Defaults to five settled work Turns; completed Todos remain an explicit option. " + "Defaults to six settled work Turns; completed Todos remain an explicit option. " "It does not create " "turns, spend quota, or grant authority." ), diff --git a/loopx/cli_commands/registry_admin_configure.py b/loopx/cli_commands/registry_admin_configure.py index f23af44788..09f955426b 100644 --- a/loopx/cli_commands/registry_admin_configure.py +++ b/loopx/cli_commands/registry_admin_configure.py @@ -32,7 +32,7 @@ def register_configure_goal_command(subparsers: argparse._SubParsersAction) -> N choices=TURN_GRANULARITY_CHOICES, help=( "Set sticky goal turn granularity. fine plans small checkpoints within " - "a coherent work slice; review defaults to 5 settled work Turns in both modes." + "a coherent work slice; review defaults to 6 settled work Turns in both modes." ), ) configure_goal_parser.add_argument( diff --git a/loopx/control_plane/goals/goal_vision_policy.py b/loopx/control_plane/goals/goal_vision_policy.py index 67f96f3259..6f2f9c4548 100644 --- a/loopx/control_plane/goals/goal_vision_policy.py +++ b/loopx/control_plane/goals/goal_vision_policy.py @@ -14,7 +14,7 @@ class GoalVisionAdvancementPolicy(str, Enum): ) # Default review cadence counts settled work Turns, not Todo size. -DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD = 5 +DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD = 6 MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD = 6 # Explicit completed-Todo cadence remains bounded by the retained evidence diff --git a/tests/capabilities/test_machine_configuration_goal_defaults.py b/tests/capabilities/test_machine_configuration_goal_defaults.py index 8a9b719c04..95a8307eae 100644 --- a/tests/capabilities/test_machine_configuration_goal_defaults.py +++ b/tests/capabilities/test_machine_configuration_goal_defaults.py @@ -287,7 +287,7 @@ def test_v1_cadence_rejects_ambiguous_units(configuration) -> None: @pytest.mark.parametrize("mode", ["standard", "fine"]) -def test_product_default_and_namespace_removal_use_five_settled_turns(tmp_path, mode): +def test_product_default_and_namespace_removal_use_six_settled_turns(tmp_path, mode): from loopx.capabilities.goal_inspection import inspect_goal_capabilities from loopx.control_plane.work_items.replan_history_codec import effective_turn_cadence_context @@ -306,17 +306,17 @@ def test_product_default_and_namespace_removal_use_five_settled_turns(tmp_path, raw = json.loads(registry_path.read_text())["goals"][0] assert "replan_after_effective_turns" not in raw["execution_profile"] projected = _history_goal(registry_path, runtime_root) - assert projected["execution_profile"]["replan_after_effective_turns"] == 5 + assert projected["execution_profile"]["replan_after_effective_turns"] == 6 assert effective_turn_cadence_context( resolve_todo_replan_cadence_goal(raw, runtime_root), runtime_root, - )["threshold"] == 5 + )["threshold"] == 6 inspected = inspect_goal_capabilities(registry_path=registry_path, runtime_root=runtime_root, goal_id=GOAL_ID)["configuration"] cadence = next(c for c in inspected["capability_catalog"]["capabilities"] if c["capability_id"] == "todo_replan_cadence") assert cadence["effective_configuration"]["source"] == "capability_default" assert cadence["effective_configuration"]["configuration"]["count_unit"] == "effective_turns" - assert cadence["effective_configuration"]["configuration"]["count"] == 5 + assert cadence["effective_configuration"]["configuration"]["count"] == 6 # Even the old default is an explicit, sticky override; compacting must # never silently switch it back to the new Turn unit. @@ -328,4 +328,4 @@ def test_product_default_and_namespace_removal_use_five_settled_turns(tmp_path, assert effective_turn_cadence_context(raw, runtime_root) is None configure_goal(registry_path=registry_path, goal_id=GOAL_ID, clear_execution_replan_after_todos=True, execute=True) - assert _history_goal(registry_path, runtime_root)["execution_profile"]["replan_after_effective_turns"] == 5 + assert _history_goal(registry_path, runtime_root)["execution_profile"]["replan_after_effective_turns"] == 6 diff --git a/tests/control_plane/test_effective_turn_replan.py b/tests/control_plane/test_effective_turn_replan.py index 3b30e49cef..9234de8eed 100644 --- a/tests/control_plane/test_effective_turn_replan.py +++ b/tests/control_plane/test_effective_turn_replan.py @@ -29,7 +29,7 @@ @pytest.mark.parametrize("threshold_override,device_count,threshold", [ - (None, None, 5), (2, 3, 2), (None, 3, 3), (6, None, 6), (None, 6, 6), + (None, None, 6), (2, 3, 2), (None, 3, 3), (5, 6, 5), (None, 5, 5), (6, None, 6), (None, 6, 6), ]) def test_open_todo_settled_turn_cadence_and_evidence_linked_review( tmp_path: Path, threshold_override: int | None, device_count: int | None, @@ -285,7 +285,7 @@ def periodic(): json.loads(registry.read_text())["goals"][0], runtime, ), runtime, )["threshold"] - == (device_count or 5) + == (device_count or 6) ) From 59cfaf4c9cdd7ee2cb82ca9dc85180f63df205f1 Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 02:50:16 +0800 Subject: [PATCH 5/7] docs(replan): disclose six-Turn default and explicit override retention Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- benchmark/runtime/RUNTIME.md | 5 ++-- benchmark/runtime/SETTINGS.md | 2 +- docs/quota-allocation.md | 24 ++++++++++--------- .../goal-vision-replan-contract-v0.md | 7 +++--- 4 files changed, 21 insertions(+), 17 deletions(-) diff --git a/benchmark/runtime/RUNTIME.md b/benchmark/runtime/RUNTIME.md index ef20db5a56..3ccfa9b610 100644 --- a/benchmark/runtime/RUNTIME.md +++ b/benchmark/runtime/RUNTIME.md @@ -273,8 +273,9 @@ Harbor LoopX modes (`heartbeat`, `turn`, `loopx-goal`) default to control plane still owns which settled work Turns count; adapters do not count records or completed Todos themselves. -Omitting both cadence options selects six effective work Turns; the -product default remains five. The previous benchmark default was three effective work Turns. +Omitting both cadence options selects six effective work Turns, matching the +product default. The previous benchmark default was three effective work Turns; +the previous product default was five. Explicit settings retain their values. Idle wakes, tool calls and the planning checkpoint do not count as effective work Turns; this is a deterministic threshold rather than a per-wake probability, and other replan triggers can act sooner. diff --git a/benchmark/runtime/SETTINGS.md b/benchmark/runtime/SETTINGS.md index 817fa5e64a..f54c257749 100644 --- a/benchmark/runtime/SETTINGS.md +++ b/benchmark/runtime/SETTINGS.md @@ -1,7 +1,7 @@ # Runner defaults and controlled ablations New LoopX executions default to **loopx-planned** task entry and replanning -after **6 settled effective work Turns**, while the product default remains five. This applies to +after **6 settled effective work Turns**, matching the product default. This applies to Harbor heartbeat, Turn and LoopX Goal modes, and EdgeBench heartbeat-resume and heartbeat-explore profiles. Planned entry replaces the seeded-todo default; pass `task_entry: seeded-todo` / `--task-entry seeded-todo` to retain the previous diff --git a/docs/quota-allocation.md b/docs/quota-allocation.md index d7a3f6a483..e8bd0f2502 100644 --- a/docs/quota-allocation.md +++ b/docs/quota-allocation.md @@ -90,10 +90,10 @@ depending on the executor: ## Goal Review Cadence -The product defaults to review after **5 settled effective work Turns** in both -standard and fine-grained modes. This replaces the previous default of five -completed Todos, so review can occur while a long Todo remains open. Benchmark -runners configure **6** effective Turns to reduce periodic interruptions. +The product and benchmark runners default to review after **6 settled effective +work Turns** in both standard and fine-grained modes. This changes the previous +product threshold of five Turns and benchmark threshold of three Turns. Review +can occur while a long Todo remains open; it does not count completed Todos. TurnEnvelope is not required. These are deterministic thresholds, not random averages or counts of model messages, tool calls, or heartbeat wakeups. @@ -131,10 +131,11 @@ writeback validation use the same effective configuration. Precedence is **explicit Goal override → live device setting → product default**. A Goal override is a complete unit/count value, not a field merge. Device writes -are revision-locked; removing the namespace restores five effective Turns. -Existing v0 device records and explicit Goal values retain their completed-Todo -meaning, including an explicit value of five. Reading or upgrading does not -rewrite them. Clearing the Goal override restores inheritance. Goals without +are revision-locked; removing the namespace restores six effective Turns. +Existing explicit Goal and device values retain their unit and count, including +five effective Turns and legacy completed-Todo settings. To retain the previous +product cadence, set `--execution-replan-after-turns 5` explicitly. Reading or +upgrading does not rewrite overrides. Clearing the Goal override restores inheritance. Goals without any override adopt the new default on their next evaluation after upgrade; this is a disclosed default behavior change, not a migration of historical data. Already running benchmark attempts retain their frozen code and settings. @@ -166,10 +167,11 @@ long-open-Todo chains and Monitor-specific thresholds. The former 20-material-run periodic fallback remains on the explicit completed-Todo path; the effective-Turn path uses verified settlements instead. -产品默认从完成 5 个 Todo 改为 5 个已结算有效工作 Turn;benchmark 默认 3 个。 +产品与 benchmark 默认统一为 6 个已结算有效工作 Turn,此前分别为 5 个与 3 个。 长 Todo 未结束也能触发方向复核,不依赖 TurnEnvelope,不按工具调用或空轮询计数。 -Goal 显式设置优先,其次是设备设置;清除两层覆盖才恢复产品默认。旧的 Todo -配置保留原单位,升级不改写;没有覆盖的 Goal 则采用新默认。复核仍可有证据地 +Goal 显式设置优先,其次是设备设置;清除两层覆盖才恢复产品默认。显式 Turn 与 +旧 Todo 配置保留原单位和次数,升级不改写;显式设置 5 个 Turn 可保留旧产品周期。 +没有覆盖的 Goal 则采用新默认。复核仍可有证据地 保留当前路线,不要求机械换方向,也不增加执行、配额或目标验收权限。 ### Governed Turn Execution diff --git a/docs/reference/protocols/goal-vision-replan-contract-v0.md b/docs/reference/protocols/goal-vision-replan-contract-v0.md index 9ebb644515..2d98bd24ed 100644 --- a/docs/reference/protocols/goal-vision-replan-contract-v0.md +++ b/docs/reference/protocols/goal-vision-replan-contract-v0.md @@ -29,9 +29,10 @@ budgeting, dreaming, or product-specific replan logic. ## Default review cadence -Goals without an explicit Goal or device cadence use five settled effective work -Turns per Agent. This replaces the implicit five-completed-Todo default; existing -explicit Todo values keep their unit. Benchmark runners pin three effective Turns. +Goals without an explicit Goal or device cadence use six settled effective work +Turns per Agent, matching the benchmark runner default. The previous defaults +were five and three Turns respectively. Existing explicit Turn and Todo values +keep their unit and count; upgrading does not rewrite these overrides. The existing TypeScript replan history owner counts verified settlement receipts; no TurnEnvelope opt-in is needed. Quota and writeback resolve the same live configuration. See [Goal review cadence](../../quota-allocation.md#goal-review-cadence) From 2477b8b20954efe189e2acc58bdb28f2112f8293 Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 03:08:54 +0800 Subject: [PATCH 6/7] Align cadence catalog guidance and commands with six-Turn default Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- loopx/configuration_catalog.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/loopx/configuration_catalog.py b/loopx/configuration_catalog.py index b8e2e25d9b..df5801d930 100644 --- a/loopx/configuration_catalog.py +++ b/loopx/configuration_catalog.py @@ -167,7 +167,7 @@ def build_goal_configuration_catalog( "default": default_replan_cadence_configuration(), "consider_when": "Direction needs regular review even while the same Todo remains open.", "effect": ( - "Choose 1–5 completed Todos or settled work Turns per Agent. " + "Choose 1–6 settled work Turns or 1–5 completed Todos per Agent. " "Legacy settings retain their completed-Todo units until explicitly changed." ), "does_not": [ @@ -177,10 +177,12 @@ def build_goal_configuration_catalog( ], "commands": { "preview_enable": _configure_command( - goal_id, "--execution-replan-after-turns", "5" + goal_id, "--execution-replan-after-turns", + str(default_replan_cadence_configuration()["count"]), ), "apply_enable": _configure_command( - goal_id, "--execution-replan-after-turns", "5", execute=True + goal_id, "--execution-replan-after-turns", + str(default_replan_cadence_configuration()["count"]), execute=True ), "preview_disable": _configure_command( goal_id, "--clear-execution-replan-after-todos", "--clear-execution-replan-after-turns" From 7c4ea7b90da7222f105fd2d938d5ad20dc0f90cc Mon Sep 17 00:00:00 2001 From: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> Date: Sat, 10 Oct 2026 03:11:32 +0800 Subject: [PATCH 7/7] Correct cadence help and retain existing editor regression coverage Signed-off-by: LoopX Agent <337587101+loopx-agent@users.noreply.github.com> --- loopx/cli_commands/registry_admin_configure.py | 13 ++++++++++--- tests/control_plane/test_todo_replan_cadence.py | 2 +- 2 files changed, 11 insertions(+), 4 deletions(-) diff --git a/loopx/cli_commands/registry_admin_configure.py b/loopx/cli_commands/registry_admin_configure.py index 09f955426b..8690b8215f 100644 --- a/loopx/cli_commands/registry_admin_configure.py +++ b/loopx/cli_commands/registry_admin_configure.py @@ -3,7 +3,10 @@ import argparse from ..execution_profile import TURN_GRANULARITY_CHOICES -from ..control_plane.goals.goal_vision_policy import MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD +from ..control_plane.goals.goal_vision_policy import ( + DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD, + MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD, +) from ..orchestration import EXPLORE_HARNESS_PROFILES from .registry_admin_peer import ( register_peer_runtime_arguments, @@ -32,7 +35,8 @@ def register_configure_goal_command(subparsers: argparse._SubParsersAction) -> N choices=TURN_GRANULARITY_CHOICES, help=( "Set sticky goal turn granularity. fine plans small checkpoints within " - "a coherent work slice; review defaults to 6 settled work Turns in both modes." + "a coherent work slice; review defaults to " + f"{DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD} settled work Turns in both modes." ), ) configure_goal_parser.add_argument( @@ -57,7 +61,10 @@ def register_configure_goal_command(subparsers: argparse._SubParsersAction) -> N configure_goal_parser.add_argument( "--execution-replan-after-turns", type=int, choices=range(1, MAX_EFFECTIVE_TURN_REPLAN_THRESHOLD + 1), - help="Review direction after this many settled work Turns, independently of Todo completion (product default: 5).", + help=( + "Review direction after this many settled work Turns, independently of " + f"Todo completion (product default: {DEFAULT_EFFECTIVE_TURN_REPLAN_THRESHOLD})." + ), ) configure_goal_parser.add_argument( "--clear-execution-replan-after-turns", action="store_true", diff --git a/tests/control_plane/test_todo_replan_cadence.py b/tests/control_plane/test_todo_replan_cadence.py index 42d44dae16..3a1eea5671 100644 --- a/tests/control_plane/test_todo_replan_cadence.py +++ b/tests/control_plane/test_todo_replan_cadence.py @@ -54,7 +54,7 @@ def test_cadence_configuration_previews_persists_and_restores_default(tmp_path): assert editor["fields"][0]["input_kind"] == "select" assert editor["fields"][1]["input_kind"] == "number" assert editor["fields"][1]["minimum"] == 1 - assert editor["fields"][1]["maximum"] == 5 + assert editor["fields"][1]["maximum"] == 6 configure_goal( registry_path=registry, goal_id="example",