Skip to content
Merged
Original file line number Diff line number Diff line change
Expand Up @@ -159,7 +159,7 @@ const fieldCopy: Record<WorkspaceLocale, FieldCopy> = {
executor_model: { label: "Model", description: "Optional model for the selected executor. Leave blank to keep the executor's own default." },
executor_reasoning_effort: { label: "Reasoning effort", description: "Optional reasoning effort for the selected executor. Leave blank to keep the executor's own default." },
count_unit: { label: "Review after", description: "Work Turns require accepted settlement; polls and retries do not count.", options: { completed_todos: "Completed Todos", effective_turns: "Settled work Turns" } },
count: { label: "Number between reviews", description: "From 1 to 5, using the selected unit. Goal overrides take precedence over device defaults." },
count: { label: "Number between reviews", description: "Settled work Turns: 1–6; completed Todos: 1–5. Goal overrides take precedence over device defaults." },
allowed_domains: { label: "Allowed responsibility domains", description: "Enter one bounded, public-safe domain per line." },
coordinator_agent_id: { label: "Coordinator Agent", description: "Use an already registered Agent id; leave blank to disable coordination." },
enabled: { label: "Enabled" },
Expand Down Expand Up @@ -191,7 +191,7 @@ const fieldCopy: Record<WorkspaceLocale, FieldCopy> = {
executor_model: { label: "模型", description: "所选执行器使用的模型,可留空;留空表示沿用执行器自身的默认模型。" },
executor_reasoning_effort: { label: "推理档位", description: "所选执行器使用的推理档位,可留空;留空表示沿用执行器自身的默认档位。" },
count_unit: { label: "复核计数依据", description: "有效工作 Turn 须完成结算;轮询和重复重试不计数。", options: { completed_todos: "已完成 Todo", effective_turns: "已结算工作 Turn" } },
count: { label: "两次复核间的数量", description: "按所选单位计数,范围 1–5;Goal 显式设置优先于设备默认值。" },
count: { label: "两次复核间的数量", description: "已结算工作 Turn 为 1–6,已完成 Todo 为 1–5;Goal 显式设置优先于设备默认值。" },
allowed_domains: { label: "允许的职责域", description: "每行填写一个有边界、可公开的职责域。" },
coordinator_agent_id: { label: "协调 Agent", description: "填写一个已经注册的 Agent ID;留空表示关闭协调。" },
enabled: { label: "启用" },
Expand Down
5 changes: 3 additions & 2 deletions benchmark/edgebench/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -368,9 +368,10 @@ and credentials belong outside the public repository.

## Default replan cadence

New `heartbeat-resume` and `heartbeat-explore` trials replan after **3 settled
New `heartbeat-resume` and `heartbeat-explore` trials replan after **6 settled
effective work Turns** by default. Use `--replan-after-turns N` to set a different
count (1–5), or `--replan-after-todos 3` for the previous completed-Todo cadence
count (1–6); `--replan-after-turns 3` retains the previous benchmark default.
Use `--replan-after-todos 3` for the previous completed-Todo cadence
as an explicit ablation. The two options are mutually exclusive and require a
heartbeat profile. This default is independent of `--task-entry`, feedback mode,
and `--turn-envelope`. Official, single and native-goal profiles are unchanged.
Expand Down
4 changes: 2 additions & 2 deletions benchmark/edgebench/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -109,8 +109,8 @@ def main(argv=None):
parser.add_argument("--task-entry", choices=TASK_ENTRIES,
help="Heartbeat default: loopx-planned; seeded-todo is an explicit ablation")
cadence = parser.add_mutually_exclusive_group()
cadence.add_argument("--replan-after-turns", type=int, choices=range(1, 6),
help="Heartbeat default: 3 settled effective work Turns")
cadence.add_argument("--replan-after-turns", type=int, choices=range(1, 7),
help="Heartbeat default: 6 settled effective work Turns")
cadence.add_argument("--replan-after-todos", type=int, choices=range(1, 6),
help="Explicit completed-Todo cadence ablation for heartbeat profiles")
parser.add_argument("--eval-interval", type=int,
Expand Down
21 changes: 13 additions & 8 deletions benchmark/runtime/RUNTIME.md
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ agents:
codex_sandbox: danger-full-access
turn_timeout_sec: null
scheduler_timeout_sec: 5080
replan_after_turns: 3
replan_after_turns: 6
```

For an explicitly selected **heartbeat-only** context experiment, add
Expand Down Expand Up @@ -266,20 +266,25 @@ by selecting this option for a new run.
### Default effective-Turn cadence

Harbor LoopX modes (`heartbeat`, `turn`, `loopx-goal`) default to
`replan_after_turns: 3`; native EdgeBench `heartbeat-resume` and
`heartbeat-explore` default to `--replan-after-turns 3`. This passes the existing Goal option
`replan_after_turns: 6`; native EdgeBench `heartbeat-resume` and
`heartbeat-explore` default to `--replan-after-turns 6`. This passes the existing Goal option
`--execution-replan-after-turns` and verifies the persisted
`replan_after_effective_turns` value before execution. The shared TypeScript
control plane still owns which settled work Turns count; adapters do not count
records or completed Todos themselves.

Omitting both cadence options now selects three effective work Turns instead
of three completed Todos. Idle wakes, tool calls and the planning checkpoint do
Omitting both cadence options selects six effective work Turns, matching the
product default. The previous benchmark default was three effective work Turns;
the previous product default was five. Explicit settings retain their values.
Idle wakes, tool calls and the planning checkpoint do
not count as effective work Turns; this is a deterministic threshold rather than
a per-wake probability, and other replan triggers can act sooner.
To retain the old cadence in a new trial, pass `replan_after_todos: 3` in Harbor
or `--replan-after-todos 3` in EdgeBench. Explicit Turn/Todo settings are mutually
exclusive and accept counts from one through five. Resolved runtime and worker
To retain the previous benchmark default in a new trial, pass
`replan_after_turns: 3` in Harbor or `--replan-after-turns 3` in EdgeBench.
For completed-Todo cadence, use `replan_after_todos` / `--replan-after-todos`.
Explicit Turn/Todo settings are mutually
exclusive; effective-Turn counts accept one through six and completed-Todo counts
accept one through five. Resolved runtime and worker
receipts name the selected unit even when no flag was supplied. Non-LoopX
profiles retain their existing behavior. No active attempt, task, scoring, feedback,
spawn permission, or total-budget change is implied.
Expand Down
25 changes: 15 additions & 10 deletions benchmark/runtime/SETTINGS.md
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
# Runner defaults and controlled ablations

New LoopX executions default to **loopx-planned** task entry and replanning
after **3 settled effective work Turns**, deliberately earlier than the product
default of five. This applies to
after **6 settled effective work Turns**, matching the product default. This applies to
Harbor heartbeat, Turn and LoopX Goal modes, and EdgeBench heartbeat-resume and
heartbeat-explore profiles. Planned entry replaces the seeded-todo default; pass
`task_entry: seeded-todo` / `--task-entry seeded-todo` to retain the previous
Expand All @@ -22,16 +21,20 @@ use matched repetitions to test its effect.
| Task entry | LoopX modes: loopx-planned | Record planned or seeded for every arm |
| Explore | Off in heartbeat-resume; on in heartbeat-explore | Keep resume as the reference |
| Turn envelope | Off | Enable only in its ablation |
| Replan cadence | 3 settled effective work Turns | Use explicit completed-Todo cadence or another Turn count for ablation |
| Replan cadence | 6 settled effective work Turns | Use explicit completed-Todo cadence or another Turn count for ablation |
| Iteration context | Harbor: fresh; EdgeBench heartbeat: resume | Freeze the provider and context within a comparison |
| Evaluator feedback | EdgeBench: best-only (strict new-best snapshot notifications) | Native and blind remain explicit controls; freeze feedback mode for every comparison |
| Model and effort | Caller-selected | Pin both; never infer them from a profile name |
| Time and sampling | EdgeBench task defaults in [task settings](../edgebench/README.md#trial-timeouts); explicit flags override | Pin resolved seconds in the study manifest |

`replan_after_turns` counts settled effective work turns through the shared
control-plane contract, not tool calls or idle heartbeat wakes. The threshold is 3 by
control-plane contract, not tool calls or idle heartbeat wakes. The threshold is 6 by
default, independent of task entry and TurnEnvelope. This is a deterministic
threshold, not a random per-wake probability; other replan triggers may act sooner.
An accepted replan resets the same Agent's periodic and repeated-progress history
windows. A failed or unaccepted replan does not; current acceptance gaps remain
independent triggers. Six reduces periodic interruptions relative to the previous
default of three, but does not promise lower total replan time or better scores.
Task timeouts bound attempts, not a requirement to consume every second.

## Recommended small study
Expand All @@ -41,11 +44,11 @@ These are recommendations, not automatically launched experiments.

| Arm | EdgeBench flags relative to the reference | Question |
| --- | --- | --- |
| Reference | `--worker heartbeat-resume --task-entry loopx-planned --replan-after-turns 3` | Planned entry with effective-turn replanning |
| Reference | `--worker heartbeat-resume --task-entry loopx-planned --replan-after-turns 6` | Planned entry with effective-turn replanning |
| Seeded | Replace only `--task-entry` with `seeded-todo` | Does initial task decomposition help? |
| Explore | Replace only `--worker` with `heartbeat-explore` | Are recorded evidence and subsequent route choices useful? |
| Short envelope | Add `--turn-envelope` | Does progressive context loading reduce overhead without losing decisions? |
| Todo cadence | Replace `--replan-after-turns 3` with `--replan-after-todos 3` | Does effective-turn cadence avoid postponing replans on long Todos? |
| Todo cadence | Replace `--replan-after-turns 6` with `--replan-after-todos 3` | How do units and thresholds change interruptions? This is a combined ablation. |
| No LoopX | `--worker official`; omit LoopX-specific flags | What is the net effect of the whole LoopX treatment? |

The no-LoopX comparison changes several mechanisms; do not attribute its delta
Expand All @@ -64,7 +67,7 @@ python -m benchmark.edgebench.run \
--task portfolio_risk_calibration --tasks-dir "$TASKS_DIR" \
--log-dir "$RUNS_DIR" --run-id "$NEW_ATTEMPT_ID" \
--worker heartbeat-resume --task-entry loopx-planned \
--replan-after-turns 3 --feedback best-only \
--replan-after-turns 6 --feedback best-only \
--model "$MODEL" --effort xhigh --timeout 43200 --eval-interval 300 \
--judge-url "$JUDGE_URL" --api-proxy-url "$API_PROXY_URL"
```
Expand All @@ -78,8 +81,10 @@ recorded service setting, not inherited from another task.

Harbor uses the same task-entry owner. In the agent kwargs, set
`execution_mode: heartbeat`, `task_entry: loopx-planned`,
`replan_after_turns: 3`, `iteration_context: resume` and `turn_envelope: false`
for an equivalent mechanism reference. For Todo cadence, remove
`replan_after_turns: 6`, `iteration_context: resume` and `turn_envelope: false`
for an equivalent mechanism reference. For the previous benchmark default, set
`replan_after_turns: 3`.
For Todo cadence, remove
`replan_after_turns` and set `replan_after_todos: 3`; the two cadence fields are
mutually exclusive. The named Explore profile in this matrix is EdgeBench-specific; do not assume
an equivalent Harbor kwargs switch. Keep the dataset, provider and validation
Expand All @@ -98,6 +103,6 @@ cannot isolate an individual fix.
Old attempts are immutable references. A rerun after multiple fixes measures
the combined revision change unless each fix has a matched control. Roll back
the entry default through explicit seeded entry and the cadence default through
`--replan-after-todos 3` / `replan_after_todos: 3` in a new attempt. Do not rewrite
`--replan-after-turns 3` / `replan_after_turns: 3` in a new attempt. Do not rewrite
old receipts or change an active worker. None of these settings grants model,
submission, credential or launch authority.
4 changes: 2 additions & 2 deletions benchmark/runtime/codex.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,8 @@
from pathlib import Path


# Benchmarks deliberately review earlier than the product default of five.
DEFAULT_REPLAN_AFTER_TURNS = 3
# New benchmark runs and the product both default to six settled work Turns.
DEFAULT_REPLAN_AFTER_TURNS = 6


MODES = ("plain", "native-goal", "heartbeat", "turn", "loopx-goal")
Expand Down
4 changes: 2 additions & 2 deletions benchmark/runtime/harbor.py
Original file line number Diff line number Diff line change
Expand Up @@ -84,8 +84,8 @@ def __init__(
raise ValueError("Choose replan_after_turns or replan_after_todos, not both")
if replan_after_turns is not None:
if (type(replan_after_turns) is not int or
not 1 <= replan_after_turns <= 5):
raise ValueError("replan_after_turns must be an integer between 1 and 5")
not 1 <= replan_after_turns <= 6):
raise ValueError("replan_after_turns must be an integer between 1 and 6")
if not self.execution.uses_loopx:
raise ValueError("replan_after_turns requires a LoopX execution mode")
if replan_after_turns is None and replan_after_todos is None and self.execution.uses_loopx:
Expand Down
4 changes: 2 additions & 2 deletions benchmark/runtime/sforge.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,8 +118,8 @@ def __init__(self, config, *, profile: str, cwd: str,
raise ValueError("replan_after_todos requires a heartbeat profile")
if replan_after_turns is not None:
if (type(replan_after_turns) is not int or
not 1 <= replan_after_turns <= 5):
raise ValueError("replan_after_turns must be an integer between 1 and 5")
not 1 <= replan_after_turns <= 6):
raise ValueError("replan_after_turns must be an integer between 1 and 6")
if not profile.startswith("heartbeat-"):
raise ValueError("replan_after_turns requires a heartbeat profile")
if (replan_after_turns is None and replan_after_todos is None
Expand Down
10 changes: 5 additions & 5 deletions benchmark/tests/test_sforge_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -174,7 +174,7 @@ async def installed(self, environment):
env = worker.runtime._worker_env(cwd="/task")
assert float(env["LOOPX_CODEX_TURN_TIMEOUT_SEC"]) == expected
assert worker.runtime.scheduler_timeout == total
expected_cadence = ({"replan_after_effective_turns": turns or 3}
expected_cadence = ({"replan_after_effective_turns": turns or 6}
if profile.startswith("heartbeat-") else
{"replan_after_completed_todos": 3})
assert worker.runtime._replan_receipt() == expected_cadence
Expand Down Expand Up @@ -396,7 +396,7 @@ async def installed(self, environment):
assert json.loads((tmp_path / "worker-profile.json").read_text())["task_entry"] == entry
@pytest.mark.parametrize('profile', ['heartbeat-resume', 'heartbeat-explore'])
@pytest.mark.parametrize('enabled', [False, True])
@pytest.mark.parametrize('cadence', [None, 2])
@pytest.mark.parametrize('cadence', [None, 2, 6])
def test_envelope_treatment_reaches_shared_worker_and_receipts(tmp_path, monkeypatch, profile, enabled, cadence):
pytest.importorskip('sforge')
pytest.importorskip('harbor')
Expand All @@ -413,7 +413,7 @@ async def installed(self, environment):
env = worker.runtime._worker_env(cwd='/task')
assert env.get('LOOPX_TURN_ENVELOPE') == ('1' if enabled else None)
assert worker.runtime.execution.turn_envelope is enabled
assert worker.runtime.replan_after_turns == (cadence or 3)
assert worker.runtime.replan_after_turns == (cadence or 6)
receipt = json.loads((tmp_path / 'worker-profile.json').read_text())
assert receipt['outer_resume'] is False
assert receipt.get('turn_envelope') is (True if enabled else None)
Expand Down Expand Up @@ -462,7 +462,7 @@ def test_edgebench_rejects_invalid_profile_settings_before_creating_trial(tmp_pa


@pytest.mark.parametrize("cadence_args,field,count", [
([], "replan_after_effective_turns", 3),
([], "replan_after_effective_turns", 6),
(["--replan-after-turns", "2"], "replan_after_effective_turns", 2),
(["--replan-after-todos", "3"], "replan_after_completed_todos", 3),
])
Expand Down Expand Up @@ -519,7 +519,7 @@ def stop_before_solver(**kwargs):
assert receipt["status"] == "runner_failed"


@pytest.mark.parametrize("value", [0, 6, True, 2.5, "3"])
@pytest.mark.parametrize("value", [0, 7, True, 2.5, "3"])
def test_effective_turn_cadence_rejects_invalid_values_before_install(tmp_path, monkeypatch, value):
pytest.importorskip("sforge")
pytest.importorskip("harbor")
Expand Down
4 changes: 2 additions & 2 deletions benchmark/tests/test_shared_codex_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -359,12 +359,12 @@ def test_baseline_and_treatment_use_same_harbor_entry(tmp_path):
assert env["LOOPX_EXECUTION_MODE"] == mode
assert env["LOOPX_PROJECT"] == "/workspace"
assert env["MODEL_NAME"] == "fixture"
assert agent.replan_after_turns == (3 if agent.execution.uses_loopx else None)
assert agent.replan_after_turns == (6 if agent.execution.uses_loopx else None)


@pytest.mark.parametrize("existing", [False, True])
@pytest.mark.parametrize("settings,field,value", [
({}, "replan_after_effective_turns", 3),
({}, "replan_after_effective_turns", 6),
({"replan_after_turns": 2}, "replan_after_effective_turns", 2),
({"replan_after_todos": 3}, "replan_after_completed_todos", 3),
])
Expand Down
Loading
Loading