diff --git a/apps/presentation/site/public/blog/agent-facing-kanban/index.html b/apps/presentation/site/public/blog/agent-facing-kanban/index.html
index 3f2a850eed..de2b75945a 100644
--- a/apps/presentation/site/public/blog/agent-facing-kanban/index.html
+++ b/apps/presentation/site/public/blog/agent-facing-kanban/index.html
@@ -191,7 +191,7 @@
How to read a LoopX board in practice
loopx --format json task-lease inspect --goal-id <goal-id> --todo-id <todo-id>
# Bounded evidence timeline visible to the current agent
-loopx --format json evidence-log --goal-id <goal-id> --agent-id <agent-id> --thin --limit 20
+loopx --format json history --goal-id <goal-id> --agent-id <agent-id> --limit 20
Write operations use lifecycle entry points exposed by the current Todo, Gate, claim or lease, and runtime contract—not by editing a display column or constructing status prose. Read current command help and the returned action contract to learn required parameters, whether an execution identity is needed, and whether work can continue.
The CLI and local frontend expose operations and read models at different levels. Lark has corresponding messages, goal channels, and a Base board adapter, but that does not establish full equivalence across every field and action. The Lark Kanban adapter still identifies itself as a prototype contract: synchronization and triggers require configuration, and an external board does not create another task identity system.
diff --git a/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html b/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html
index ff60b92ba3..82c838bc0a 100644
--- a/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html
+++ b/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html
@@ -189,7 +189,7 @@ 怎样实际读一张 LoopX 看板
loopx --format json task-lease inspect --goal-id <goal-id> --todo-id <todo-id>
# 当前 Agent 可见的有界证据时间线
-loopx --format json evidence-log --goal-id <goal-id> --agent-id <agent-id> --thin --limit 20
+loopx --format json history --goal-id <goal-id> --agent-id <agent-id> --limit 20
写操作通过当前 Todo、Gate、claim/lease 与运行契约提供的生命周期入口完成,不靠改展示列或拼一段状态文本。需要什么参数、是否要执行实例身份,以及当前能否继续,读取当前命令帮助和返回的动作契约。
CLI 与本地前端提供不同粒度的操作与读面;Lark 也有相应消息、目标通道和 Base 看板适配,但不能因此宣称所有界面字段和动作已完全等价。Lark Kanban adapter 目前仍标明 prototype contract:同步和触发需要配置,外部看板不自行创造另一套任务身份。
diff --git a/docs/architecture/rfcs/typescript-control-plane-migration-v0.md b/docs/architecture/rfcs/typescript-control-plane-migration-v0.md
index b53a0c467e..8fa226d176 100644
--- a/docs/architecture/rfcs/typescript-control-plane-migration-v0.md
+++ b/docs/architecture/rfcs/typescript-control-plane-migration-v0.md
@@ -440,6 +440,19 @@ replan source while retaining independently evidenced vision paths. This
corrects false advancement admission; it does not add a blocked-wait settlement
route or certify Goal completion.
+Replan decision context now follows the same boundary: `work_items/replan_context.ts`
+owns scoping, chronology, repetition reduction, result/route diversity and bounded
+coverage projection. Its Python codec reads the existing Goal and compact index,
+normalizes historical observations and reuses the private snapshot transport.
+The core Goal, scoped acceptance contract, evidence and uncovered frontier enter
+one context; the standalone evidence command is retired. Selection reads beyond
+the status display window. The readable view retains up to 24 distinct
+observations, while writeback novelty checks retain complete available history.
+Real CLI references, file/SQLite receipt reentry, wrong-scope/stale reads and
+omitted-old-blocker rejection qualify this S3/S6 slice. Ordinary guard output
+remains unchanged; the required-replan information budget increases explicitly.
+This does not complete T3, qualify ten-day costs or establish S11 score gains.
+
The delivery-history boundary now treats `classification`, `health_check`, and
`recommended_action` as narrative. They cannot create or discharge a
follow-through obligation, prove an outcome, or classify delivery scale.
diff --git a/docs/book/chapters/appendix-reference.md b/docs/book/chapters/appendix-reference.md
index fb52592f2a..0572497be6 100644
--- a/docs/book/chapters/appendix-reference.md
+++ b/docs/book/chapters/appendix-reference.md
@@ -76,7 +76,7 @@ Capability 描述调用者可依赖的 outcome contract,Provider 提供实现
| 精确查看 Todo | `loopx todo list --goal-id --todo-id ` | 不从缺少显示推断不存在 |
| 压缩工作列表 | `loopx todo list --goal-id --thin --format json` | 有界显示不等于完整候选集 |
| 查看当前 lease | `loopx task-lease inspect` | 读回不是获取新执行证明 |
-| 查看 Agent 证据 | `loopx evidence-log --goal-id --agent-id --thin --limit 30` | 历史与当前事实分开 |
+| 查看 Agent 证据 | `loopx history --goal-id --agent-id --limit 30` | 历史与当前事实分开 |
| 请求当前准入 | `loopx quota should-run` | 读完整 contract;相关 Host 路径可能记 receipt |
| 查看受管 Turn journal | `loopx turn inspect-journal` | 保留原 Turn key;诊断不执行恢复 |
| 配置与能力发现 | `loopx capability list/show`、`loopx configure-goal` | 无设置 flag 的读取与 execute 分开 |
diff --git a/docs/book/en/chapters/appendix-reference.md b/docs/book/en/chapters/appendix-reference.md
index b3f8c3469e..6d02b5f317 100644
--- a/docs/book/en/chapters/appendix-reference.md
+++ b/docs/book/en/chapters/appendix-reference.md
@@ -76,7 +76,7 @@ Capability describes a caller outcome contract; Provider supplies implementation
| Read an exact Todo | `loopx todo list --goal-id --todo-id ` | Absence from a display is not absence of work |
| Compact a work list | `loopx todo list --goal-id --thin --format json` | Bounded display is not the whole candidate set |
| Inspect a lease | `loopx task-lease inspect` | Observation does not acquire new proof |
-| Read Agent evidence | `loopx evidence-log --goal-id --agent-id --thin --limit 30` | Separate history from current facts |
+| Read Agent evidence | `loopx history --goal-id --agent-id --limit 30` | Separate history from current facts |
| Request admission | `loopx quota should-run` | Read the full contract; relevant Host paths can record receipts |
| Inspect a managed Turn journal | `loopx turn inspect-journal` | Preserve original Turn key; diagnosis does not recover |
| Discover configuration and capabilities | `loopx capability list/show`, `loopx configure-goal` | Separate reads without setting flags from execution |
diff --git a/docs/concepts/interaction-pattern-catalog.md b/docs/concepts/interaction-pattern-catalog.md
index 2aacd5f2e4..5a87d40bbd 100644
--- a/docs/concepts/interaction-pattern-catalog.md
+++ b/docs/concepts/interaction-pattern-catalog.md
@@ -2993,7 +2993,7 @@ resolved. The two lanes must not collapse into each other.
Replan closeout is semantic and causally bound. A normal validated progress
refresh may record useful work, but it must not silently close the
-`autonomous_replan_obligation_v0`. Quota first projects the evidence-log into a
+`autonomous_replan_obligation_v0`. Quota first projects compact run history into a
compact coverage ledger and emits an opaque `obligation_id`; after the bounded
slice, the agent writes one typed observation. When the result is a runnable
successor, the Todo transition itself is the receipt:
diff --git a/docs/development/control-plane-course/08-evidence-refresh-and-self-repair.md b/docs/development/control-plane-course/08-evidence-refresh-and-self-repair.md
index 79434ae13c..d5ed8310b9 100644
--- a/docs/development/control-plane-course/08-evidence-refresh-and-self-repair.md
+++ b/docs/development/control-plane-course/08-evidence-refresh-and-self-repair.md
@@ -382,7 +382,7 @@ Replan 是当前目标图或覆盖账本上的机器可见语义变化:
Dreaming 是探索未来可能性,可以产生 proposal,但不能替代当前 runnable frontier。
-quota 会把 evidence-log 压成 host-projected coverage ledger,并给出当前
+quota 会把当前 Agent 的 compact run history 压成 host-projected coverage ledger,并给出当前
`obligation_id`。新 surface / hypothesis / probe 用 typed observation 写回;如果
replan 的结果是下一条 runnable direction,则 Todo 本身就是原子语义 receipt:
@@ -857,7 +857,7 @@ CLI/status/quota 并补 smoke;只更新文档不会改变机器的下一次决
- `examples/state-projection-gap-smoke.py`
- `examples/project/goal-vision-replan-contract-smoke.py`
-- `examples/control_plane/agent-scoped-evidence-log-smoke.py`
+- `tests/control_plane/test_replan_context_evidence.py`
- `examples/outcome-followthrough-policy-smoke.py`
- `tests/control_plane_ts/quota_monitor_poll_commit.test.ts`
diff --git a/docs/reference/contracts/interface-budget-contract.md b/docs/reference/contracts/interface-budget-contract.md
index 8e69749c6b..3cf6fc6ec9 100644
--- a/docs/reference/contracts/interface-budget-contract.md
+++ b/docs/reference/contracts/interface-budget-contract.md
@@ -84,7 +84,6 @@ removed without a separately validated caller migration.
| `heartbeat-prompt --thin` | absolute hot path | agent scope, multi-agent fixture matrix, and exact Agent-input field allowlist | Markdown diagnostics, `--compact`, `--full` |
| `todo list` | baseline and growth | todo-count growth and agent filtering semantics | `--thin`, `--limit N`, role/status filters, direct todo-id lifecycle commands |
| `history --limit 5` | explicit-limit cold path | returned-run bound | individual run JSON/Markdown artifacts |
-| `evidence-log --thin --limit 5` | explicit-limit cold path | returned-evidence bound | referenced run-history and rollout-event artifacts |
`quota should-run` uses one repeatable cold-path selector:
`--include-detail scheduler`, `agent-todos`, `user-todos`, `vision`, or
@@ -164,7 +163,10 @@ structural refactor does not become a permanent CI red light.
The receipts contain counts, shape paths, headings, and digests only; they do
not persist raw CLI output. Candidate-only surfaces are allowed after their
absolute characterization passes, while removing a qualified base row fails
-closed.
+closed. The intentional evidence-command retirement is recognized only with a
+qualified replacement replan-context projection and remains a review signal.
+Coverage-only to dense replan context has a measured, one-time allowance on its
+three affected JSON surfaces; dense-to-dense changes retain ordinary limits.
Both budget layers are intentionally about projections, not the full archival
facts. When a surface needs more detail, put that detail behind a queryable
@@ -173,6 +175,21 @@ recurring heartbeat prompt carry it. `nested_keys` counts dictionary keys
through three payload levels and samples at most 20 list items per level; it is
a hot-path structure budget, not an archival record-size budget.
+Required replan is a decision phase with a separate information need. Its
+`replan_context` carries the core Goal and up to 24 distinct observations from
+the full compact index, reducing repetitions before selection. On the unchanged
+crowded public CLI fixture, emitted JSON grows from 26,844 to 33,523 characters
+and 718 to 806 lines. The replan scenario ceiling moves from 30,000/750 to
+34,000/830, with 6,000 fixed semantic growth characters; ordinary and multi-Agent
+non-replan guards retain their previous output and ceilings. Handoff forwarding
+keeps a brief evidence pointer and the full review packet owns the structured
+context, removing duplicate JSON. The explicit cold-path diagnosis retains its
+existing selected-packet plus Goal-array contract; its replan fixture measures
+43,132 characters / 804 lines and uses 44,000 / 850 ceilings with 7,000 fixed
+semantic growth characters. These are output measurements, not model-token,
+latency or long-horizon quality qualifications. Settlement checks use complete
+available history even when the readable context omits older observations.
+
Restraint rules for new fields:
1. Prefer adding evidence to run history, then projecting only the smallest
diff --git a/docs/reference/protocols/agent-management-projection-v0.md b/docs/reference/protocols/agent-management-projection-v0.md
index 0ba5095cd7..a283f0aabe 100644
--- a/docs/reference/protocols/agent-management-projection-v0.md
+++ b/docs/reference/protocols/agent-management-projection-v0.md
@@ -57,8 +57,8 @@ This contract intentionally does not add:
- a tool gateway or agent profile runtime.
State changes still go through existing LoopX lifecycle commands such as
-`loopx todo ...`, `loopx refresh-state ...`, `loopx quota ...`, evidence-log
-writeback, and future APIs that preserve the same event-ledger semantics.
+`loopx todo ...`, `loopx refresh-state ...`, `loopx quota ...`, and future APIs
+that preserve the same event-ledger semantics. Replan context is a read model.
## Shape
diff --git a/docs/reference/protocols/agent-material-frontier-v0.md b/docs/reference/protocols/agent-material-frontier-v0.md
index 6089fa21ca..6b776963b8 100644
--- a/docs/reference/protocols/agent-material-frontier-v0.md
+++ b/docs/reference/protocols/agent-material-frontier-v0.md
@@ -93,7 +93,7 @@ count can never be mistaken for an empty authority registry.
An omitted boundary is also treated as inaccessible rather than implicitly
public.
-Evidence and receipts have different semantics. A run-history or evidence-log
+Evidence and receipts have different semantics. A run-history or replan-evidence
row may prove that an agent changed or validated an artifact, but it does not
prove the agent consumed a material revision. Only a matching
`material_usage_receipt_v0` can move a material from unread or stale to
diff --git a/docs/reference/protocols/agent-scoped-evidence-ledger-v0.md b/docs/reference/protocols/agent-scoped-evidence-ledger-v0.md
index 7d98a0549b..a31d459c77 100644
--- a/docs/reference/protocols/agent-scoped-evidence-ledger-v0.md
+++ b/docs/reference/protocols/agent-scoped-evidence-ledger-v0.md
@@ -1,6 +1,6 @@
# agent_scoped_evidence_ledger_v0
-`agent_scoped_evidence_ledger_v0` defines a thin, chronological read model for
+`agent_scoped_evidence_ledger_v0` defines a decision-oriented read model for
agents that need to replan, hand off, or explain progress without reading raw
rollout logs, private active state, or another agent's detailed working trail.
@@ -21,8 +21,8 @@ jobs:
| `loopx history` | Reads compact run history and run indexes. | It is run-centric and not equivalent to rollout events. |
| `loopx quota should-run --agent-id ...` | Decides whether a specific agent lane should act and projects a compact coverage ledger plus uncovered frontier from the evidence source. | It does not ask the model to reconstruct history or treat a read receipt as progress. |
-The resulting surface is a public-safe, bounded, agent-scoped ledger that the
-host can project into a replan action packet and an operator can inspect in full.
+The resulting surface is one public-safe, bounded replan context. Operators
+retain the existing history view for compact run diagnostics.
## Ownership Boundary
@@ -31,118 +31,76 @@ host can project into a replan action packet and an operator can inspect in full
| Event sources | Durable append-only events, compact run records, ids, timestamps, and public-safe refs. | Prompt-ready planning summaries or cross-agent privacy policy. |
| Status and review packets | Current projections, attention queues, frontier summaries, and operator packets. | Raw chronological replay or write authority. |
| Quota | Lane routing, spend policy, scheduler hints, host context delivery, and the minimal replan action packet. | Storing replan rationale or accepting writeback. |
-| Agent-scoped evidence ledger | Thin chronological rows for the current agent plus compressed frontier for other agents. | Replan selection policy, semantic-delta validation, canonical writes, raw logs, raw trajectories, private documents, or full other-agent traces. |
-| Replan context policy | Builds the coverage ledger, delivery receipt, and uncovered frontier. | Reimplementing typed progress comparison or terminal-closure truth. |
+| Internal event/history adapter | Public-safe historical rows for supervisor and recovery consumers. | Agent-facing read rituals, semantic-delta validation or canonical writes. |
+| Replan context policy | Selects dense scoped evidence and builds core Goal, coverage ledger, delivery receipt, and uncovered frontier. | Reimplementing typed progress comparison, terminal-closure truth or exposing full other-agent traces. |
| Semantic write gate | Validates typed progress, state-grounded successors, fresh vision outcomes, blockers, and coverage-backed terminal results against the current obligation. | Reconstructing the evidence ledger or interpreting classification prose. |
| Acting agent | Selects an uncovered direction from delivered context and submits a typed observation or vision outcome. | Treating context delivery, a manual read, or a legacy ACK alone as progress. |
-## Read Model Shape
-
-The CLI payload uses the shipped `agent_scoped_evidence_log_v0` schema (the
-protocol name describes the ledger concept rather than a second wire schema):
-
-```json
-{
- "schema_version": "agent_scoped_evidence_log_v0",
- "goal_id": "example-goal",
- "agent_id": "codex-evidence-peer",
- "mode": "thin",
- "todo_id": null,
- "since": null,
- "event_kinds": [],
- "limit": 30,
- "matched_count": 1,
- "ledger_count": 1,
- "truncated": false,
- "source_refs": [
- "rollout_event_log.public_safe_view",
- "compact_run_history.public_refs"
- ],
- "ledger": [
- {
- "event_id": "evt_123",
- "recorded_at": "2026-07-05T00:00:00Z",
- "source": "rollout_event_log",
- "event_kind": "todo_update",
- "agent_id": "codex-evidence-peer",
- "todo_id": "todo_123",
- "classification": "implementation_batch",
- "status": "open",
- "summary": "P0 implementation frontier was split into a design contract and a CLI read model."
- }
- ],
- "other_agent_frontier": {
- "schema_version": "other_agent_frontier_v0",
- "policy": "goal_frontier_only",
- "item_count": 1,
- "items": [
- {
- "agent_id": "codex-main-control",
- "source": "run_history",
- "classification": "validated_progress"
- }
- ]
- },
- "boundary": {
- "raw_logs_recorded": false,
- "raw_trajectory_recorded": false,
- "credential_values_recorded": false,
- "absolute_paths_recorded": false,
- "other_agent_event_stream_expanded": false
- }
-}
-```
-
-The schema is intentionally narrow. It should be cheap to produce, cheap to read
-in a prompt, and stable enough for quota/replan tests.
-
-## CLI Contract
-
-The public CLI is read-only:
+## One Replan Entry
+
+Routine replan consumes the host-projected `replan_context_v0` on the current
+obligation returned by `loopx quota should-run --goal-id --agent-id
+`. Full review/handoff packets embed the same read model; handoff-only forwarding
+keeps its readable summaries and the normal quota guard without duplicating the
+structured evidence packet. There is no
+standalone `evidence-log` command or compatibility alias.
+
+The TypeScript work-item owner scopes compact run records, sorts them by
+normalized timestamp, removes exact replays and collapses repeated observations
+while retaining their count and first/latest timestamps. It selects up to 24
+distinct observations: preserve observed result classes, then distinct typed
+surface/hypothesis/probe routes, then recent observations. This is a bounded
+selection rule, not a relevance or best-score oracle. An unchanged score is
+never interpreted as a disproved hypothesis from prose alone.
+
+Actual quota and handoff routes read the complete decoded compact index before
+selection; the ordinary status display limit cannot hide earlier evidence.
+Snapshot-only callers remain limited to their supplied source and are marked by
+`from_full_index=false`. Large inputs reuse the private, digest-checked replan
+snapshot transport rather than truncating at the RPC size limit.
+
+`core_goal` precedes the evidence and carries the current active-state Objective
+(or the registered objective when absent), with explicit source and missing
+state. The existing acceptance owner's scoped objective, criteria, non-goals,
+revision and status remain a separate `acceptance_contract`: a selected-work
+contract and an Agent's current task do not redefine the whole Goal. This read
+model creates no new Goal or acceptance authority.
+
+Evidence retains public-safe validation/action summaries linked to typed
+`coverage_ledger` observations by fingerprint. Other Agents and unattributed
+records cannot supply proof for a scoped Agent. Counts describe the available
+source, not complete Goal acceptance. Omitted observations expose counts, time
+bounds and a concrete scoped history read. Missing required fields, unsupported
+typed versions and malformed evidence fail explicitly; valid empty sources
+produce empty arrays.
+
+Readable evidence has a display budget; writeback validation uses all supplied
+scoped history observations, including omitted blockers and prior claims. A
+truncated coverage projection alone cannot establish novelty. Reading context
+or changing an evidence label never creates permission or semantic progress.
+
+Each evidence row includes a content-bound `evidence_ref` and an exact
+`read_action` using the existing history entry:
```bash
-loopx --format json evidence-log --goal-id --agent-id --thin --limit 30
+loopx --format json history --goal-id --agent-id --evidence-ref
```
-Supported filters:
-
-| Option | Meaning |
-| --- | --- |
-| `--todo-id ` | Filter rollout events by exact todo id and compact runs by bounded todo mention. |
-| `--since ` | Return rows recorded after a timestamp. |
-| `--event-kind ` | Filter rollout event kinds such as `todo_update`, `quota_should_run`, or `validation`. |
-| `--limit ` | Bound rows after filtering. Default should be small enough for an agent prompt. |
-| `--history-limit ` | Bound compact run-history rows scanned before filtering. |
-| `--rollout-limit ` | Bound rollout-event rows scanned from the tail before filtering. |
-| `--thin` | Select the only current public-safe mode; accepted explicitly for readable generated commands. |
-| global `--format json\|markdown` | Select JSON or the compact Markdown rendering. |
-
-The command must fail closed on missing `goal_id` or `agent_id`. A vague
-surface value such as `codex` should not silently fall into `other-agent`
-semantics; callers should pass a registered agent id and, when needed, a
-separate host surface such as `codex-app`, `codex-cli`, `opencode`, or `claude-code`.
-
-## Scoping Rules
-
-The current implementation returns detailed rows under these deterministic
-rules:
-
-- the event has `agent_id` equal to the requested agent id;
-- when `--todo-id` is present, the event has that exact todo id;
-- compact run-history rows have the requested agent id and, when filtered by
- todo, mention that todo in one of the bounded run fields;
-- `--since` and normalized `--event-kind` filters are applied before the final
- newest-first limit.
-
-Other agents should not be shown row by row by default. They should be compressed
-into `other_agent_frontier` from the latest compact run-history row per agent,
-with a maximum of three rows. This lets an agent understand the shared direction
-without inheriting another lane's private scratchpad.
+Execute the supplied action only when more detail is needed. It resolves one
+public-safe compact record in the same Goal and Agent; a changed or unavailable reference fails explicitly and asks the caller to refresh context.
+It never falls back to another Agent or the latest unrelated record. It reads no
+raw transcript and creates no read receipt. Ordinary `history` remains the
+operator's compact run-history view; `--agent-id` scopes that view as well.
+
+The internal rollout/history adapter remains available to supervisor and native
+recovery code. Its legacy receipt decoder preserves historical observability,
+but no product path generates a new `evidence_log_read` event or read ritual.
+Removing the CLI neither deletes event/history files nor changes settlement.
## Replan Integration
When quota or status projects a replan obligation for an agent, the host folds
-the bounded agent-scoped chronology into a compact coverage ledger and delivers
+agent-scoped compact history into the decision evidence and coverage ledger and delivers
it with the current obligation:
```json
@@ -170,7 +128,7 @@ The full obligation also carries `replan_context_v0`: a bounded
`replan_context_delivery_receipt_v0`. The control-plane responsibilities are
deliberately split and causally bound:
-- the evidence log remains the durable public-safe chronology;
+- compact run history and append-only rollout events remain the durable sources;
- quota owns context delivery and does not require a weak protocol-following
model to discover or execute a read ritual;
- `typed_progress_observation_v0` owns work-slice identity and result semantics;
@@ -203,23 +161,19 @@ even when the source acceptance gap remains visible. Terminal coverage inputs
fail at the CLI boundary when their required coverage scope is missing;
`exploration_exhausted` additionally requires explicit coverage completion.
-Every successful diagnostic `loopx evidence-log` execution appends an
-`evidence_log_read` rollout event and returns an
-`evidence_log_read_receipt_v0`. The receipt carries the goal id, agent id,
-bounded read window, canonical public-safe command, and recorded timestamp.
-Receipt events are excluded from an unfiltered ledger view so repeated reads do
-not recursively inflate the chronology. They are observability facts only: a
-read, a failed read, a prose ACK, or a historical repair-delta ACK does not
-close the current obligation. This prevents a receipt for an earlier periodic
-review from masking a later vision/frontier duty.
+Historical `evidence_log_read_receipt_v0` records remain readable. They are
+observability facts only: a read, a failed read, a prose ACK, or a historical
+repair-delta ACK cannot close the current obligation.
### Effect-program boundary
This flow uses the effect-program separation without adding a second settlement
executor. Host context projection is a repeatable read effect; the typed
progress writeback is a separately validated state transition. The delivery
-receipt proves context delivery, while the semantic delta proves use of that
-context. Neither receipt is allowed to impersonate the other.
+receipt records context delivery, while an accepted semantic delta establishes
+an eligible change against the obligation and its evidence. Neither proves
+that the model read or understood every delivered observation, and neither
+receipt is allowed to impersonate the other.
The live behavior qualification tests that causal handoff through an actual
function-tool conversation rather than a testing-only output field. A Doubao
@@ -228,7 +182,7 @@ command against a hermetic public-safe Goal. The harness runs that command
through the real LoopX CLI, returns its actual context/action packet, and asks
the actor to choose the next real tool action. The actor independently qualifies
the selected typed observation, then executes the real `refresh-state` command.
-Evidence-log-only, prose-only, pre-quota, equivalent-fingerprint, and ungrounded
+History-read-only, prose-only, pre-quota, equivalent-fingerprint, and ungrounded
successor actions do not pass. Only temporary fixture state may change, and the
receipt stores bounded command digests and typed outcomes rather than prompts,
packets, or output.
@@ -250,29 +204,27 @@ say that only a compact pointer or count was recorded.
## Current Implementation Status
-The CLI, rollout-event/run-history merge, bounded other-agent frontier,
-host-projected coverage context, minimal action packet, typed repeat detector,
-and shared quota/write-time semantic gate are implemented. Todo and material
-projections remain separate current-state surfaces; they are not copied into
-this chronological ledger. Historical repair ACKs have a bounded read adapter
-for old run rows, but new replan closure has one truth: typed semantic delta.
+The standalone evidence command and its generated read obligations are retired.
+The existing replan context owns the bounded model view; the TypeScript owner
+selects evidence and coverage while Python adapts historical codecs and public
+safety. Handoff embeds that view, and history resolves exact references.
+Supervisor/native recovery retain their internal event source. None of these
+read paths grants execution, writeback, quota or Goal-completion authority.
+
+This closes the duplicate evidence-entry gap in roadmap S3/S6. It does not
+qualify long-horizon score improvement, full-history completeness, or research
+observation settlement; those retain their existing S11 acceptance.
## Acceptance
-A change satisfies this contract only when:
-
-- `loopx evidence-log` returns a bounded JSON packet for a concrete `goal_id` and
- `agent_id`;
-- current-agent rows are detailed while other-agent rows are compressed by
- default;
-- filters behave deterministically and do not require parsing raw JSONL in agent
- prompts;
-- replan-capable quota/status payloads deliver a compact coverage ledger and
- uncovered frontier without requiring a model read ritual;
-- the live function-tool qualification proves that the default model selects
- and executes a semantic next action from a production heartbeat/quota
- exchange rather than merely echoing a test field;
-- the existing status, history, review-packet, and rollout-event-log surfaces
- keep their current responsibilities; and
-- public tests prove the privacy boundary without committing private state,
- local paths, raw logs, or raw trajectories.
+- Real quota and handoff paths deliver readable scoped evidence, typed coverage
+ and the current uncovered frontier without a mandatory read roundtrip.
+- Sorting, replay deduplication and truncation are deterministic and bounded.
+- A projected reference resolves through the real history CLI; stale, changed,
+ wrong-Goal and wrong-Agent references fail explicitly.
+- Contract errors remain distinct from valid empty evidence.
+- Historical receipts do not grant semantic progress; equivalent observations,
+ read-only actions and ungrounded successors still cannot close replan.
+- Public tests use synthetic fixtures and retain the public/private boundary.
+- Long-horizon model utility remains an evaluation result, not a claim inferred
+ from passing deterministic tests.
diff --git a/docs/reference/protocols/goal-vision-replan-contract-v0.md b/docs/reference/protocols/goal-vision-replan-contract-v0.md
index 5eeb7fca16..ad612fac4c 100644
--- a/docs/reference/protocols/goal-vision-replan-contract-v0.md
+++ b/docs/reference/protocols/goal-vision-replan-contract-v0.md
@@ -576,8 +576,9 @@ The audit also exposes a compact deterministic `vision_gap_judge_v0`
instruction packet for the agent. It borrows the strict done-judge stance used
by autonomous goal loops without calling an LLM: the agent is told to compare
the active vision `acceptance_summary` with the host-projected coverage ledger,
-then permitted registry-declared material references. The agent-scoped
-`loopx evidence-log` remains an operator diagnostic, not a mandatory model ritual.
+then permitted registry-declared material references. `replan_context` supplies
+scoped readable evidence and exact history read actions; no separate evidence
+command or mandatory model read ritual remains.
Bounded public web research is the next
fallback when those sources are missing or stale and the gap depends on public
facts. `done=true` is only valid
@@ -847,6 +848,14 @@ source references with the typed observation.
## Write / Correction Mechanism
+For optional history drill-down, `history --goal-id ... --agent-id ... --limit N`
+filters the complete available compact index by Agent before applying `N`.
+Its run lists and latest status refer to that same scoped source; newer Peer
+records cannot hide the requested lane. Goal quota accounting remains Goal-wide.
+This query limit is independent of status/quota's bounded replan decision
+lookback. The generated omitted-evidence read action must recover older distinct
+observations without a context-access receipt or state mutation.
+
After a material milestone, `vision_outcome_checkpoint_required` remains a
completion guard. When the checkpoint is satisfied and current, the path outcome
is `continue`, `no_change`, or `replan`, evidence refs are present, and no
diff --git a/docs/reference/protocols/model-behavior-qualification-v0.md b/docs/reference/protocols/model-behavior-qualification-v0.md
index 8a3eb9aecc..28b6d0babc 100644
--- a/docs/reference/protocols/model-behavior-qualification-v0.md
+++ b/docs/reference/protocols/model-behavior-qualification-v0.md
@@ -296,7 +296,7 @@ target. The terminal-settlement scenario requires the model to validate a real
fixture artifact, then follow the quota-projected `durable_writeback ->
quota_spend -> terminal_closeout` sequence under one stable effect identity; a
premature no-follow-up or spend-before-writeback fails. The replan scenario
-requires the exact agent-scoped evidence-log
+requires the exact agent-scoped replan
context projected by a hermetic typed-repeat replan state, then a real
frontier/source read and either a typed `refresh-state` delta or one
obligation-bound successor `todo add`; the third requires a
diff --git a/docs/status-data-contract.md b/docs/status-data-contract.md
index 7ec8851b8e..2521f98127 100644
--- a/docs/status-data-contract.md
+++ b/docs/status-data-contract.md
@@ -199,10 +199,12 @@ For replan, the guard carries a host-built `replan_context_v0` and the compact
`replan_action_packet_v0`. The context projects a bounded coverage ledger from
the agent-scoped evidence history, an uncovered frontier, and a delivery
receipt. The acting model therefore chooses a direction from delivered context;
-it does not have to discover and execute an evidence-log command as a protocol
-preflight. `loopx evidence-log --goal-id --agent-id --thin`
-remains the cold-path diagnostic chronology, and its read receipt remains useful
-for observability, but a read or legacy ACK cannot close a replan obligation.
+it does not have to discover or execute an evidence command as a protocol
+preflight. The standalone evidence command is removed. Each projected evidence
+row carries an exact Goal/Agent-bound history read action. Missing fields,
+unsupported versions and unavailable references fail explicitly; valid empty
+arrays alone mean no matching evidence. Historical read receipts remain
+observable, but a read or legacy ACK cannot close a replan obligation.
Closure requires a typed semantic delta accepted against the current obligation:
a new surface, hypothesis, probe family, state-grounded runnable successor,
fresh evidence-linked vision path, concrete new blocker, or coverage-backed
diff --git a/examples/cli-help-manpage-smoke.py b/examples/cli-help-manpage-smoke.py
index 993c6fb670..86e3b5343a 100644
--- a/examples/cli-help-manpage-smoke.py
+++ b/examples/cli-help-manpage-smoke.py
@@ -49,7 +49,8 @@ def assert_concise_default_help(output: str) -> None:
assert "Codex App" in output, output
assert "Claude Code" in output, output
assert "loopx commands" in output, output
- assert "evidence-log --goal-id ID --agent-id AGENT --thin" in output, output
+ assert "evidence-log" not in output, output
+ assert "Read this agent's thin ledger" not in output, output
assert "man loopx" in output, output
assert LONG_TAIL_COMMAND not in output, output
assert len(output.splitlines()) <= 39, output
@@ -72,8 +73,8 @@ def assert_command_reference_surface() -> None:
assert result.returncode == 0, (result.returncode, result.stdout, result.stderr)
assert "LoopX command reference" in result.stdout, result.stdout
assert "Daily operator commands" in result.stdout, result.stdout
- assert "required evidence-log reads" in result.stdout, result.stdout
- assert "before replan or handoff" in result.stdout, result.stdout
+ assert "host-projected replan context" in result.stdout, result.stdout
+ assert "evidence-log" not in result.stdout, result.stdout
assert "Loop driver hints" in result.stdout, result.stdout
assert "Claude Code /loop" in result.stdout, result.stdout
assert "Maintainer and adapter commands" in result.stdout, result.stdout
@@ -248,8 +249,8 @@ def assert_installer_manpage_surface() -> None:
assert "Codex App automation" in man_text, man_text
assert "loopx commands" in man_text, man_text
assert "loopx extension" in man_text, man_text
- assert r"loopx evidence\-log \-\-goal\-id" in man_text, man_text
- assert "before replan or handoff" in compact_man_text, man_text
+ assert r"loopx evidence\-log" not in man_text, man_text
+ assert r"host\-projected replan context" in compact_man_text, man_text
assert r"loopx COMMAND \-\-help" in man_text, man_text
profile_text = profile.read_text(encoding="utf-8")
diff --git a/examples/control_plane/agent-scoped-evidence-log-smoke.py b/examples/control_plane/agent-scoped-evidence-log-smoke.py
deleted file mode 100644
index 28022c9db6..0000000000
--- a/examples/control_plane/agent-scoped-evidence-log-smoke.py
+++ /dev/null
@@ -1,301 +0,0 @@
-#!/usr/bin/env python3
-"""Smoke-test the read-only agent-scoped evidence-log CLI."""
-
-from __future__ import annotations
-
-import json
-import subprocess
-import sys
-import tempfile
-from pathlib import Path
-
-
-REPO_ROOT = Path(__file__).resolve().parents[2]
-if str(REPO_ROOT) not in sys.path:
- sys.path.insert(0, str(REPO_ROOT))
-
-from loopx.control_plane.runtime.agent_scoped_evidence_log import ( # noqa: E402
- project_evidence_log_read_receipts,
-)
-from loopx.rollout_event_log import load_rollout_events # noqa: E402
-
-
-GOAL_ID = "agent-evidence-fixture"
-AGENT_ID = "agent-a"
-TODO_ID = "todo_target"
-
-
-def write_jsonl(path: Path, rows: list[dict]) -> None:
- path.parent.mkdir(parents=True, exist_ok=True)
- path.write_text("\n".join(json.dumps(row, ensure_ascii=False, sort_keys=True) for row in rows) + "\n")
-
-
-def write_fixture(root: Path) -> Path:
- runtime = root / "runtime"
- registry_path = root / "registry.json"
- registry_path.write_text(
- json.dumps(
- {
- "schema_version": 1,
- "common_runtime_root": str(runtime),
- "goals": [
- {
- "id": GOAL_ID,
- "domain": "fixture",
- "status": "active-read-only",
- "adapter": {"kind": "fixture", "status": "connected-read-only"},
- }
- ],
- },
- ensure_ascii=False,
- indent=2,
- )
- + "\n",
- encoding="utf-8",
- )
- write_jsonl(
- runtime / "goals" / GOAL_ID / "rollout-event-log.jsonl",
- [
- {
- "schema_version": "loopx_rollout_event_v0",
- "goal_id": GOAL_ID,
- "event_id": "evt-agent-a",
- "event_kind": "todo_update",
- "recorded_at": "2026-07-05T00:00:01Z",
- "agent_id": AGENT_ID,
- "todo_id": TODO_ID,
- "status": "open",
- "summary": "authorization token label is safe as prose",
- "boundary": {
- "raw_task_text_recorded": False,
- "raw_logs_recorded": False,
- "raw_trajectory_recorded": False,
- "raw_session_transcript_recorded": False,
- "credential_values_recorded": False,
- "absolute_paths_recorded": False,
- },
- },
- {
- "schema_version": "loopx_rollout_event_v0",
- "goal_id": GOAL_ID,
- "event_id": "evt-aksk",
- "event_kind": "todo_update",
- "recorded_at": "2026-07-05T00:00:02Z",
- "agent_id": AGENT_ID,
- "todo_id": TODO_ID,
- "status": "open",
- "summary": "sk=should-not-surface",
- "boundary": {
- "raw_task_text_recorded": False,
- "raw_logs_recorded": False,
- "raw_trajectory_recorded": False,
- "raw_session_transcript_recorded": False,
- "credential_values_recorded": False,
- "absolute_paths_recorded": False,
- },
- },
- {
- "schema_version": "loopx_rollout_event_v0",
- "goal_id": GOAL_ID,
- "event_id": "evt-access-secret",
- "event_kind": "todo_update",
- "recorded_at": "2026-07-05T00:00:03Z",
- "agent_id": AGENT_ID,
- "todo_id": TODO_ID,
- "status": "open",
- "summary": "access_key=should-not-surface",
- "boundary": {
- "raw_task_text_recorded": False,
- "raw_logs_recorded": False,
- "raw_trajectory_recorded": False,
- "raw_session_transcript_recorded": False,
- "credential_values_recorded": False,
- "absolute_paths_recorded": False,
- },
- },
- {
- "schema_version": "loopx_rollout_event_v0",
- "goal_id": GOAL_ID,
- "event_id": "evt-other-agent",
- "event_kind": "todo_update",
- "recorded_at": "2026-07-05T00:00:04Z",
- "agent_id": "agent-b",
- "todo_id": TODO_ID,
- "status": "open",
- "summary": "other agent private stream should not expand",
- "boundary": {
- "raw_task_text_recorded": False,
- "raw_logs_recorded": False,
- "raw_trajectory_recorded": False,
- "raw_session_transcript_recorded": False,
- "credential_values_recorded": False,
- "absolute_paths_recorded": False,
- },
- },
- ],
- )
- write_jsonl(
- runtime / "goals" / GOAL_ID / "runs" / "index.jsonl",
- [
- {
- "goal_id": GOAL_ID,
- "generated_at": "2026-07-05T00:00:04+00:00",
- "agent_id": AGENT_ID,
- "classification": "agent_a_progress",
- "recommended_action": f"continue {TODO_ID}",
- "health_check": "compact only",
- },
- {
- "goal_id": GOAL_ID,
- "generated_at": "2026-07-05T00:00:05+00:00",
- "agent_id": "agent-b",
- "classification": "agent_b_old_frontier_should_not_surface",
- "recommended_action": "old other-agent run should not surface",
- "health_check": "compact only",
- },
- {
- "goal_id": GOAL_ID,
- "generated_at": "2026-07-05T00:00:06+00:00",
- "agent_id": "agent-e",
- "classification": "agent_e_frontier",
- "recommended_action": "fourth other-agent run exceeds frontier cap",
- "health_check": "compact only",
- },
- {
- "goal_id": GOAL_ID,
- "generated_at": "2026-07-05T00:00:07+00:00",
- "agent_id": "agent-d",
- "classification": "agent_d_frontier",
- "recommended_action": "other agent d top todo only",
- "health_check": "compact only",
- },
- {
- "goal_id": GOAL_ID,
- "generated_at": "2026-07-05T00:00:08+00:00",
- "agent_id": "agent-c",
- "classification": "agent_c_frontier",
- "recommended_action": "other agent c top todo only",
- "health_check": "compact only",
- },
- {
- "goal_id": GOAL_ID,
- "generated_at": "2026-07-05T00:00:09+00:00",
- "agent_id": "agent-b",
- "classification": "agent_b_frontier",
- "recommended_action": "other agent top todo only",
- "health_check": "compact only",
- },
- ],
- )
- return registry_path
-
-
-def run_cli(registry_path: Path, *, limit: int = 10) -> dict:
- result = subprocess.run(
- [
- sys.executable,
- "-m",
- "loopx.cli",
- "--registry",
- str(registry_path),
- "--format",
- "json",
- "evidence-log",
- "--goal-id",
- GOAL_ID,
- "--agent-id",
- AGENT_ID,
- "--todo-id",
- TODO_ID,
- "--thin",
- "--limit",
- str(limit),
- ],
- cwd=REPO_ROOT,
- check=True,
- text=True,
- capture_output=True,
- )
- return json.loads(result.stdout)
-
-
-def run_status(registry_path: Path) -> dict:
- result = subprocess.run(
- [
- sys.executable,
- "-m",
- "loopx.cli",
- "--registry",
- str(registry_path),
- "--format",
- "json",
- "status",
- ],
- cwd=REPO_ROOT,
- check=False,
- text=True,
- capture_output=True,
- )
- return json.loads(result.stdout)
-
-
-def main() -> None:
- with tempfile.TemporaryDirectory() as tmp:
- registry_path = write_fixture(Path(tmp))
- payload = run_cli(registry_path)
- limited = run_cli(registry_path, limit=2)
- status_payload = run_status(registry_path)
- durable_receipts = project_evidence_log_read_receipts(
- load_rollout_events(
- Path(tmp)
- / "runtime"
- / "goals"
- / GOAL_ID
- / "rollout-event-log.jsonl"
- )
- )
- assert payload["ok"] is True
- assert payload["schema_version"] == "agent_scoped_evidence_log_v0"
- assert payload["goal_id"] == GOAL_ID
- assert payload["agent_id"] == AGENT_ID
- assert payload["todo_id"] == TODO_ID
- assert payload["read_receipt"]["goal_id"] == GOAL_ID
- assert payload["read_receipt"]["agent_id"] == AGENT_ID
- assert payload["read_receipt"]["todo_id"] == TODO_ID
- assert payload["read_receipt"]["read_window"]["limit"] == 10
- assert len(durable_receipts) == 2
- assert durable_receipts[0]["read_window"]["limit"] == 2
- assert durable_receipts[1]["event_id"] == payload["read_receipt"]["event_id"]
- status_item = next(
- item
- for item in status_payload["attention_queue"]["items"]
- if item.get("goal_id") == GOAL_ID
- )
- assert status_item["evidence_log_read_receipts"] == durable_receipts
- assert payload["rollout_event_count"] == 3
- assert payload["run_history_ref_count"] == 1
- assert payload["matched_count"] == 4
- assert payload["ledger_count"] == 4
- assert payload["truncated"] is False
- assert limited["matched_count"] == 4
- assert limited["ledger_count"] == 2
- assert limited["truncated"] is True
- rendered = json.dumps(payload, ensure_ascii=False)
- assert "authorization token label is safe as prose" in rendered
- assert "sk=should-not-surface" not in rendered
- assert "access_key=should-not-surface" not in rendered
- assert "other agent private stream should not expand" not in rendered
- assert "agent_b_old_frontier_should_not_surface" not in rendered
- assert "agent_e_frontier" not in rendered
- assert payload["other_agent_frontier"]["item_count"] == 3
- assert [item["agent_id"] for item in payload["other_agent_frontier"]["items"]] == [
- "agent-b",
- "agent-c",
- "agent-d",
- ]
- assert payload["boundary"]["other_agent_event_stream_expanded"] is False
- assert payload["boundary"]["credential_values_recorded"] is False
-
-
-if __name__ == "__main__":
- main()
diff --git a/examples/control_plane/quota-replan-decision-plane-smoke.py b/examples/control_plane/quota-replan-decision-plane-smoke.py
index 5e53790ab5..5e5f05cb34 100644
--- a/examples/control_plane/quota-replan-decision-plane-smoke.py
+++ b/examples/control_plane/quota-replan-decision-plane-smoke.py
@@ -25,9 +25,6 @@
from loopx.control_plane.todos.quota_summary import ( # noqa: E402
select_quota_todo_summary,
)
-from loopx.control_plane.runtime.agent_scoped_evidence_log import ( # noqa: E402
- build_agent_scoped_evidence_log_command,
-)
from loopx.control_plane.work_items.autonomous_replan_obligation import ( # noqa: E402
ensure_replan_novelty_policy,
)
@@ -54,7 +51,7 @@
"triggers": [{"kind": "periodic_review_due", "source": "fixture"}],
"replan_novelty_policy": {
"schema_version": "replan_novelty_policy_v0",
- "evidence_source": "agent_scoped_evidence_log",
+ "evidence_source": "compact_run_history",
"writeback": "repair_delta",
},
"stop_condition": "stop after one bounded replan slice writes back a concrete frontier delta",
@@ -270,10 +267,11 @@ def status_payload(
"agent_id": SIDE_AGENT,
"status": "completed",
"recorded_at": acked_at,
- "command": build_agent_scoped_evidence_log_command(
- goal_id=GOAL_ID,
- agent_id=SIDE_AGENT,
- required_read_id=obligation_id or None,
+ # Frozen persisted command from the retired interface, not executed.
+ "command": (
+ f"loopx --format json evidence-log --goal-id {GOAL_ID}"
+ f" --agent-id {SIDE_AGENT} --thin --limit 24"
+ + (f" --required-read-id {obligation_id}" if obligation_id else "")
),
"read_window": {"mode": "thin", "limit": 24},
**(
@@ -642,11 +640,11 @@ def assert_replan_beats_monitor_quiet_skip() -> None:
]
assert (
novelty_policy.get("evidence_source") or novelty_policy.get("evidence")
- ) == "agent_scoped_evidence_log", guard
+ ) == "compact_run_history", guard
assert novelty_policy["delivery"] == "host_projected", guard
assert novelty_policy["writeback"] == "typed_semantic_delta", guard
replan_context = guard["autonomous_replan_obligation"]["replan_context"]
- assert replan_context["evidence_source"] == "agent_scoped_evidence_log", guard
+ assert replan_context["evidence_source"] == "compact_run_history", guard
assert replan_context["delivery_receipt"]["status"] == "delivered", guard
action_packet = guard["replan_action_packet"]
assert action_packet["obligation_id"] == guard["autonomous_replan_obligation"]["obligation_id"], guard
@@ -1604,12 +1602,14 @@ def assert_unrelated_runs_do_not_promote_legacy_ack_into_semantic_closure() -> N
latest_runs=[
{
"classification": "state_refreshed",
+ "generated_at": "2026-07-04T00:01:00Z",
"agent_id": PRIMARY_AGENT,
"progress_scope": "agent_lane",
"recommended_action": "Primary lane refreshed unrelated state.",
},
{
"classification": "quota_monitor_poll",
+ "generated_at": "2026-07-04T00:02:00Z",
"agent_id": SIDE_AGENT,
"recommended_action": "Fixture monitor stayed unchanged.",
"monitor_target": {
@@ -1622,6 +1622,7 @@ def assert_unrelated_runs_do_not_promote_legacy_ack_into_semantic_closure() -> N
},
{
"classification": "monitor_poll_autonomous_replan_recorded_v0",
+ "generated_at": "2026-07-04T00:03:00Z",
"agent_id": SIDE_AGENT,
"progress_scope": "agent_lane",
"autonomous_replan_ack": {
@@ -1658,6 +1659,7 @@ def assert_non_frontier_replan_ack_does_not_clear_monitor_replan() -> None:
latest_runs=[
{
"classification": "monitor_poll_autonomous_replan_recorded_v0",
+ "generated_at": "2026-07-04T00:00:00Z",
"agent_id": SIDE_AGENT,
"progress_scope": "agent_lane",
"autonomous_replan_ack": {
diff --git a/examples/control_plane/review-packet-cli-smoke.py b/examples/control_plane/review-packet-cli-smoke.py
index 82f2a44327..0cabebd3a1 100644
--- a/examples/control_plane/review-packet-cli-smoke.py
+++ b/examples/control_plane/review-packet-cli-smoke.py
@@ -374,11 +374,8 @@ def assert_connected_delivery_surface_loop_requires_macro_evidence() -> None:
contract = payload["handoff_delivery_contract"]
assert payload["ok"] is True, payload
assert payload["connected_delivery_handoff"] is True, payload
- required_reads = payload["project_agent_required_reads"]
- assert required_reads[0]["kind"] == "agent_scoped_evidence_log", required_reads
- assert required_reads[0]["agent_id"] == "codex-side-bypass", required_reads
- assert "evidence-log" in required_reads[0]["command"], required_reads
- assert " --agent-id codex-side-bypass " in f" {required_reads[0]['command']} ", required_reads
+ assert "project_agent_required_reads" not in payload, payload
+ assert payload["replan_context"]["agent_id"] == "codex-side-bypass", payload
assert "quota should-run" in payload["project_agent_command"], payload
assert "--goal-id delivery-side-bypass" in payload["project_agent_command"], payload
assert contract["mode"] == "expand_after_surface_progress_loop", payload
@@ -387,8 +384,8 @@ def assert_connected_delivery_surface_loop_requires_macro_evidence() -> None:
assert "connected-delivery" in handoff, handoff
assert "真实 delivery" in handoff, handoff
assert "可改文件、验证、写回、spend" in handoff, handoff
- assert "必读流水账" in handoff, handoff
- assert "evidence-log" in handoff, handoff
+ assert "重规划上下文" in handoff, handoff
+ assert "evidence-log" not in handoff, handoff
assert "codex-side-bypass" in handoff, handoff
assert "surface-only 下游传播" in handoff, handoff
assert "只读或 dry-run 路径" not in handoff, handoff
@@ -406,7 +403,7 @@ def assert_connected_delivery_surface_loop_requires_macro_evidence() -> None:
handoff_only = review_packet_handoff_only_payload(payload)
assert_handoff_interface_budget(handoff_only, "connected-delivery handoff-only payload")
assert_handoff_only_top_level_budget(handoff_only, "connected-delivery handoff-only payload")
- assert handoff_only["project_agent_required_reads"] == required_reads, handoff_only
+ assert "project_agent_required_reads" not in handoff_only, handoff_only
agent_contract = handoff_only["handoff_delivery_contract"]
assert set(agent_contract) == {"mode", "instruction", "minimum_scale", "must_include", "if_blocked"}, handoff_only
assert agent_contract["mode"] == "expand_after_surface_progress_loop", handoff_only
diff --git a/examples/control_plane/review-packet-handoff-context-smoke.py b/examples/control_plane/review-packet-handoff-context-smoke.py
index 632e441446..30bdd13e89 100644
--- a/examples/control_plane/review-packet-handoff-context-smoke.py
+++ b/examples/control_plane/review-packet-handoff-context-smoke.py
@@ -15,7 +15,7 @@
agent_member_from_item,
agent_member_summary,
agent_todo_texts_for_handoff,
- project_agent_required_reads,
+ project_agent_replan_context,
project_asset_source,
project_asset_source_line,
todo_text_from_project_asset,
@@ -102,16 +102,13 @@ def assert_agent_member_contract(item: dict) -> None:
assert "worktree_policy=clean-worktree" in summary, summary
assert "claims=rp-context,canary" in summary, summary
- reads = project_agent_required_reads(GOAL_ID, item)
- assert len(reads) == 1, reads
- read = reads[0]
- assert read["kind"] == "agent_scoped_evidence_log", read
- assert read["goal_id"] == GOAL_ID, read
- assert read["agent_id"] == AGENT_ID, read
- assert read["other_agent_policy"] == "frontier_only", read
- assert "evidence-log" in read["command"], read
- assert f"--goal-id {GOAL_ID}" in read["command"], read
- assert f"--agent-id {AGENT_ID}" in read["command"], read
+ context = project_agent_replan_context(
+ GOAL_ID, item, {"run_history": {"goals": [{"id": GOAL_ID, "latest_runs": []}]}},
+ )
+ assert context["schema_version"] == "replan_context_v0", context
+ assert context["goal_id"] == GOAL_ID, context
+ assert context["agent_id"] == AGENT_ID, context
+ assert context["evidence"] == [], context
def main() -> None:
diff --git a/examples/control_plane/todo-readmodel-boundary-smoke.py b/examples/control_plane/todo-readmodel-boundary-smoke.py
index 54f6e0d281..9dea144054 100644
--- a/examples/control_plane/todo-readmodel-boundary-smoke.py
+++ b/examples/control_plane/todo-readmodel-boundary-smoke.py
@@ -23,7 +23,7 @@
from loopx.control_plane.goals import global_registry_shadow as global_registry_shadow_read_model # noqa: E402
from loopx.control_plane.goals import path_resolution as path_resolution_read_model # noqa: E402
from loopx.control_plane.agents import management_projection as management_projection_read_model # noqa: E402
-from loopx.control_plane.runtime import agent_scoped_evidence_log as evidence_log_read_model # noqa: E402
+from loopx.control_plane.runtime import agent_evidence_history as evidence_log_read_model # noqa: E402
from loopx.control_plane.runtime import run_compaction as run_compaction_read_model # noqa: E402
from loopx.control_plane.runtime import status_projection_cache as status_cache_read_model # noqa: E402
from loopx.control_plane.runtime import time as runtime_time_read_model # noqa: E402
diff --git a/examples/project/goal-vision-replan-contract-smoke.py b/examples/project/goal-vision-replan-contract-smoke.py
index 468e72126b..3061606b08 100644
--- a/examples/project/goal-vision-replan-contract-smoke.py
+++ b/examples/project/goal-vision-replan-contract-smoke.py
@@ -133,7 +133,7 @@ def main() -> int:
"`vision_gap_judge_v0` instruction packet",
"the agent is told to compare the active vision",
"projected required reads",
- "`loopx evidence-log",
+ "`replan_context` supplies",
"bounded public web research",
"`done=true` is only valid",
"explicit completion with authoritative evidence",
diff --git a/loopx/cli.py b/loopx/cli.py
index 4d0b71a3b1..92f1197ff4 100644
--- a/loopx/cli.py
+++ b/loopx/cli.py
@@ -103,7 +103,6 @@
handle_coordination_shadow_command,
handle_capability_command,
handle_dreaming_command,
- handle_evidence_log_command,
handle_explore_command,
handle_first_run_report_command,
handle_goal_channel_command,
@@ -141,7 +140,6 @@
register_capability_commands,
register_doctor_command,
register_dreaming_commands,
- register_evidence_log_command,
register_extension_commands,
register_explore_commands,
register_goal_channel_commands,
@@ -371,7 +369,6 @@ def build_parser() -> LoopXArgumentParser:
register_slash_commands_command(sub, add_subcommand_format)
register_workflow_skills_command(sub, add_subcommand_format)
register_dreaming_commands(sub, add_subcommand_format)
- register_evidence_log_command(sub, add_subcommand_format)
register_explore_commands(sub, add_subcommand_format)
register_todo_command(sub, add_subcommand_format)
register_coordination_shadow_command(sub, add_subcommand_format)
@@ -873,17 +870,6 @@ def main(argv: list[str] | None = None) -> int:
print_payload=print_payload,
)
- evidence_log_result = handle_evidence_log_command(
- args,
- registry_path=registry_path,
- runtime_root_arg=args.runtime_root,
- output_format=output_format,
- print_payload=print_payload,
- append_cli_rollout_event=append_cli_rollout_event,
- )
- if evidence_log_result is not None:
- return evidence_log_result
-
explore_result = handle_explore_command(
args,
registry_path=registry_path,
diff --git a/loopx/cli_commands/__init__.py b/loopx/cli_commands/__init__.py
index c21bfc6099..ec347f4467 100644
--- a/loopx/cli_commands/__init__.py
+++ b/loopx/cli_commands/__init__.py
@@ -58,7 +58,6 @@ def _load_exports() -> None:
register_first_run_report_command,
)
from .dreaming import handle_dreaming_command, register_dreaming_commands
- from .evidence_log import handle_evidence_log_command, register_evidence_log_command
from .explore import handle_explore_command, register_explore_commands
from .handoff_mode import handle_handoff_mode_command, register_handoff_mode_command
from .history import handle_history_command, register_history_command
@@ -197,7 +196,6 @@ def _load_exports() -> None:
"handle_doctor_command",
"handle_first_run_report_command",
"handle_dreaming_command",
- "handle_evidence_log_command",
"handle_explore_command",
"handle_handoff_mode_command",
"handle_history_command",
@@ -250,7 +248,6 @@ def _load_exports() -> None:
"register_doctor_command",
"register_first_run_report_command",
"register_dreaming_commands",
- "register_evidence_log_command",
"register_explore_commands",
"register_handoff_mode_command",
"register_history_command",
diff --git a/loopx/cli_commands/evidence_log.py b/loopx/cli_commands/evidence_log.py
deleted file mode 100644
index 57f6618af7..0000000000
--- a/loopx/cli_commands/evidence_log.py
+++ /dev/null
@@ -1,275 +0,0 @@
-from __future__ import annotations
-
-import argparse
-import shlex
-from collections.abc import Callable
-from pathlib import Path
-
-from ..control_plane.runtime.agent_scoped_evidence_log import (
- build_agent_scoped_evidence_log,
- build_agent_scoped_evidence_log_command,
- evidence_log_read_receipt,
- goal_history_runs,
-)
-from ..history import collect_history, load_registry
-from ..paths import resolve_runtime_root
-from ..rollout_event_log import load_rollout_events, rollout_event_log_path
-
-
-PrintPayload = Callable[
- [dict[str, object], str, Callable[[dict[str, object]], str]],
- None,
-]
-FormatSelector = Callable[..., str]
-RolloutEventAppender = Callable[..., dict[str, object]]
-
-
-def _evidence_log_receipt_command(args: argparse.Namespace) -> str:
- parts = shlex.split(
- build_agent_scoped_evidence_log_command(
- goal_id=args.goal_id,
- agent_id=args.agent_id,
- todo_id=args.todo_id,
- limit=max(0, int(args.limit)),
- required_read_id=args.required_read_id,
- )
- )
- if int(args.history_limit) != 80:
- parts.extend(["--history-limit", str(max(0, int(args.history_limit)))])
- if int(args.rollout_limit) != 400:
- parts.extend(["--rollout-limit", str(max(0, int(args.rollout_limit)))])
- if args.since:
- parts.extend(["--since", str(args.since)])
- for event_kind in args.event_kind:
- parts.extend(["--event-kind", str(event_kind)])
- return " ".join(shlex.quote(part) for part in parts)
-
-
-def _evidence_log_receipt_details(
- args: argparse.Namespace,
- *,
- command: str,
-) -> dict[str, object]:
- return {
- "command": command,
- "mode": "thin",
- "limit": max(0, int(args.limit)),
- "history_limit": max(0, int(args.history_limit)),
- "rollout_limit": max(0, int(args.rollout_limit)),
- "since": str(args.since or ""),
- "event_kinds": ",".join(
- str(kind).strip().lower().replace("-", "_")
- for kind in args.event_kind
- if str(kind).strip()
- ),
- "required_read_id": str(args.required_read_id or ""),
- }
-
-
-def register_evidence_log_command(
- subparsers: argparse._SubParsersAction,
- add_subcommand_format: Callable[[argparse.ArgumentParser], None],
-) -> None:
- parser = subparsers.add_parser(
- "evidence-log",
- help="Read a public-safe, agent-scoped evidence ledger for replan and handoff.",
- )
- add_subcommand_format(parser)
- parser.add_argument("--goal-id", required=True, help="Goal id whose evidence should be read.")
- parser.add_argument(
- "--agent-id",
- required=True,
- help="Registered agent id. The ledger expands only this agent's event stream.",
- )
- parser.add_argument("--todo-id", help="Optional todo id filter for the current replan/work slice.")
- parser.add_argument("--since", help="Optional ISO timestamp lower bound.")
- parser.add_argument(
- "--required-read-id",
- help="Opaque projected required-read identity recorded on the durable receipt.",
- )
- parser.add_argument(
- "--event-kind",
- action="append",
- default=[],
- help="Optional rollout event kind filter. Repeatable.",
- )
- parser.add_argument("--limit", type=int, default=24, help="Maximum merged ledger rows to return.")
- parser.add_argument(
- "--history-limit",
- type=int,
- default=80,
- help="Maximum compact run-history rows to scan before agent/todo filtering.",
- )
- parser.add_argument(
- "--rollout-limit",
- type=int,
- default=400,
- help="Maximum rollout-event rows to scan from the tail before filtering.",
- )
- parser.add_argument(
- "--thin",
- action="store_true",
- help="Return the thin public-safe shape. This is the only current mode and is accepted for clarity.",
- )
-
-
-def render_evidence_log_markdown(payload: dict[str, object]) -> str:
- if not payload.get("ok"):
- return "\n".join(
- [
- "# LoopX Evidence Log",
- "",
- f"- ok: `{payload.get('ok')}`",
- f"- error: `{payload.get('error')}`",
- "",
- ]
- )
- lines = [
- "# LoopX Evidence Log",
- "",
- f"- goal_id: `{payload.get('goal_id')}`",
- f"- agent_id: `{payload.get('agent_id')}`",
- f"- todo_id: `{payload.get('todo_id') or ''}`",
- f"- mode: `{payload.get('mode')}`",
- f"- rollout_events: `{payload.get('rollout_event_count')}`",
- f"- run_history_refs: `{payload.get('run_history_ref_count')}`",
- f"- ledger_count: `{payload.get('ledger_count')}`",
- f"- matched_count: `{payload.get('matched_count')}`",
- f"- truncated: `{payload.get('truncated')}`",
- "",
- "## Ledger",
- "",
- ]
- ledger = payload.get("ledger") if isinstance(payload.get("ledger"), list) else []
- if not ledger:
- lines.append("- No matching public-safe evidence rows.")
- for row in ledger:
- if not isinstance(row, dict):
- continue
- source = str(row.get("source") or "unknown")
- when = str(row.get("recorded_at") or "")
- if source == "rollout_event_log":
- title = str(row.get("event_kind") or "event")
- suffix = str(row.get("status") or row.get("classification") or "")
- else:
- title = str(row.get("classification") or "run")
- suffix = str(row.get("delivery_outcome") or row.get("progress_scope") or "")
- summary = str(row.get("summary") or row.get("recommended_action") or row.get("health_check") or "")
- line = f"- `{when}` `{source}` {title}"
- if suffix:
- line += f" ({suffix})"
- if summary:
- line += f": {summary}"
- lines.append(line)
- frontier = payload.get("other_agent_frontier") if isinstance(payload.get("other_agent_frontier"), dict) else {}
- items = frontier.get("items") if isinstance(frontier.get("items"), list) else []
- if items:
- lines.extend(["", "## Other Agent Frontier", ""])
- for item in items:
- if not isinstance(item, dict):
- continue
- agent = str(item.get("agent_id") or "")
- classification = str(item.get("classification") or "latest run")
- lines.append(f"- `{agent}`: {classification}")
- lines.append("")
- return "\n".join(lines)
-
-
-def handle_evidence_log_command(
- args: argparse.Namespace,
- *,
- registry_path: Path,
- runtime_root_arg: str | None,
- output_format: FormatSelector,
- print_payload: PrintPayload,
- append_cli_rollout_event: RolloutEventAppender,
-) -> int | None:
- if args.command != "evidence-log":
- return None
- try:
- registry = load_registry(registry_path)
- runtime_root = resolve_runtime_root(registry, runtime_root_arg)
- rollout_events = load_rollout_events(
- rollout_event_log_path(runtime_root, args.goal_id),
- limit=max(0, int(args.rollout_limit)),
- )
- history_payload = collect_history(
- registry_path=registry_path,
- runtime_root=runtime_root,
- goal_id=args.goal_id,
- limit=max(0, int(args.history_limit)),
- )
- payload = build_agent_scoped_evidence_log(
- goal_id=args.goal_id,
- agent_id=args.agent_id,
- todo_id=args.todo_id,
- since=args.since,
- event_kinds=args.event_kind,
- limit=max(0, int(args.limit)),
- rollout_events=rollout_events,
- history_runs=goal_history_runs(history_payload, args.goal_id),
- )
- command = _evidence_log_receipt_command(args)
- receipt_details = _evidence_log_receipt_details(args, command=command)
- append_cli_rollout_event(
- payload,
- registry_path=registry_path,
- runtime_root_arg=runtime_root_arg,
- event_kind="evidence_log_read",
- agent_id=args.agent_id,
- todo_id=args.todo_id,
- status="completed",
- summary="read the bounded agent-scoped evidence ledger",
- details=receipt_details,
- )
- event_view = payload.get("rollout_event")
- if isinstance(event_view, dict):
- receipt = evidence_log_read_receipt(
- {
- **event_view,
- "goal_id": args.goal_id,
- "agent_id": args.agent_id,
- "todo_id": args.todo_id,
- "details": receipt_details,
- }
- )
- if receipt:
- payload["read_receipt"] = receipt
- payload.pop("rollout_event", None)
- if "read_receipt" not in payload:
- payload["ok"] = False
- payload["error"] = (
- "evidence log was read but its durable read receipt could not be recorded"
- )
- except Exception as exc:
- try:
- registry = load_registry(registry_path)
- runtime_root = resolve_runtime_root(registry, runtime_root_arg)
- command = _evidence_log_receipt_command(args)
- details = _evidence_log_receipt_details(args, command=command)
- details["error"] = str(exc)
- append_cli_rollout_event(
- {"ok": False},
- registry_path=registry_path,
- runtime_root_arg=runtime_root_arg,
- event_kind="evidence_log_read",
- agent_id=args.agent_id,
- todo_id=args.todo_id,
- status="failed",
- summary="evidence log read failed",
- details=details,
- )
- except Exception:
- # The failure receipt is a best-effort escape hatch; never mask the
- # original read error.
- pass
- payload = {
- "ok": False,
- "schema_version": "agent_scoped_evidence_log_v0",
- "goal_id": getattr(args, "goal_id", None),
- "agent_id": getattr(args, "agent_id", None),
- "error": str(exc),
- }
- selected_format = output_format(args)
- print_payload(payload, selected_format, render_evidence_log_markdown)
- return 0 if payload.get("ok") else 1
diff --git a/loopx/cli_commands/history.py b/loopx/cli_commands/history.py
index d4b594549e..676b1559ab 100644
--- a/loopx/cli_commands/history.py
+++ b/loopx/cli_commands/history.py
@@ -52,6 +52,8 @@ def register_history_command(subparsers: argparse._SubParsersAction) -> None:
)
history_parser.add_argument("--goal-id", help="Only show one goal.")
history_parser.add_argument("--limit", type=int, default=10)
+ history_parser.add_argument("--agent-id", help="Scope compact history to one Agent.")
+ history_parser.add_argument("--evidence-ref", help="Read the exact public-safe evidence reference supplied by replan_context.")
history_parser.add_argument(
"--review-plan-json",
help=(
@@ -78,6 +80,32 @@ def handle_history_command(
runtime_root_arg: str | None,
print_payload: PrintPayload,
) -> int:
+ if args.agent_id and (not args.goal_id or args.history_action):
+ print_payload({"ok": False, "error": "--agent-id requires --goal-id without a history action"},
+ args.format, render_history_markdown)
+ return 1
+ if args.evidence_ref:
+ from ..control_plane.work_items.replan_context_codec import project_replan_context
+
+ try:
+ if not args.goal_id or not args.agent_id or args.history_action:
+ raise ValueError("--evidence-ref requires --goal-id and --agent-id without a history action")
+ registry = load_registry(registry_path)
+ runtime_root = resolve_runtime_root(registry, runtime_root_arg, registry_path=registry_path)
+ history = collect_history(
+ registry_path=registry_path, runtime_root=runtime_root, goal_id=args.goal_id,
+ agent_lane_id=args.agent_id, limit=max(0, args.limit),
+ )
+ payload = project_replan_context(
+ goal_id=args.goal_id, agent_id=args.agent_id,
+ runs=(), evidence_ref=args.evidence_ref,
+ source_status={"run_history": history, "registry": str(registry_path),
+ "runtime_root": str(runtime_root)},
+ )
+ except Exception as exc:
+ payload = {"ok": False, "error": str(exc)}
+ print_payload(payload, args.format, lambda value: json.dumps(value, ensure_ascii=False, indent=2))
+ return 0 if payload.get("ok") else 1
if args.history_action == "trajectory-hygiene":
try:
if not args.goal_id:
@@ -216,7 +244,27 @@ def handle_history_command(
runtime_root=runtime_root,
goal_id=args.goal_id,
limit=max(0, args.limit),
+ agent_lane_id=args.agent_id,
+ scoped_agent_id=args.agent_id,
)
+ if args.agent_id:
+ # Scope every row-bearing history surface, including retained
+ # semantic references. The global quota metadata stays goal-wide.
+ def scoped(rows):
+ return [row for row in rows if row.get("agent_id") == args.agent_id][:max(0, args.limit)]
+
+ payload["runs"] = scoped(payload["runs"])
+ for goal in payload["goals"]:
+ goal["latest_runs"] = scoped(goal["latest_runs"])
+ latest = goal.get("latest_status_run")
+ if latest and latest.get("agent_id") != args.agent_id:
+ goal["latest_status_run"] = None
+ semantic = goal["semantic_history"]
+ semantic["agents"] = scoped(semantic["agents"])
+ semantic["active_blocked_retry_runs"] = scoped(semantic["active_blocked_retry_runs"])
+ correction = semantic.get("latest_owner_correction_run")
+ if correction and correction.get("agent_id") != args.agent_id:
+ del semantic["latest_owner_correction_run"]
except Exception as exc:
payload = {
"ok": False,
diff --git a/loopx/cli_commands/status.py b/loopx/cli_commands/status.py
index 89d9d6e72f..b098f730de 100644
--- a/loopx/cli_commands/status.py
+++ b/loopx/cli_commands/status.py
@@ -128,8 +128,6 @@ def review_packet_handoff_only_payload(payload: dict[str, object]) -> dict[str,
"project_agent_command": payload.get("project_agent_command"),
"project_agent_handoff": handoff_text,
"handoff_text": handoff_text,
- "project_agent_required_reads": payload.get("project_agent_required_reads")
- or [],
"operator_gate_approved_handoff": payload.get(
"operator_gate_approved_handoff"
),
diff --git a/loopx/cli_commands/status_registration.py b/loopx/cli_commands/status_registration.py
index 04bdbdff24..42982a56d8 100644
--- a/loopx/cli_commands/status_registration.py
+++ b/loopx/cli_commands/status_registration.py
@@ -162,7 +162,7 @@ def register_status_commands(
"review-packet",
help=(
"Generate a CLI-visible Review Packet from the current status contract, "
- "including agent-scoped evidence-log read hints when available."
+ "including bounded replan context when available."
),
)
review_packet_parser.add_argument(
diff --git a/loopx/cli_commands/support_control_supervisor.py b/loopx/cli_commands/support_control_supervisor.py
index 2ff4714e62..d3156b9587 100644
--- a/loopx/cli_commands/support_control_supervisor.py
+++ b/loopx/cli_commands/support_control_supervisor.py
@@ -20,7 +20,7 @@
render_supervisor_event_markdown,
supervisor_event_log_path,
)
-from ..control_plane.runtime.agent_scoped_evidence_log import (
+from ..control_plane.runtime.agent_evidence_history import (
build_agent_scoped_evidence_log,
goal_history_runs,
)
diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts
index 2a4d441307..bb2772a003 100644
--- a/loopx/control_plane/effect_runtime_handlers.ts
+++ b/loopx/control_plane/effect_runtime_handlers.ts
@@ -34,6 +34,7 @@ import {
requireInteger,
} from "./runtime_decode.ts";
+
import type {TurnJournalInspectionRequest} from "./turn_driver/turn_journal.ts";
type EffectRuntimeHandler = (params: JsonObject) => unknown | Promise;
@@ -704,15 +705,19 @@ export function createEffectRuntimeHandlers(
settlementResultInput(params.result, "result"),
),
],
+
["turn.settlement.reduce", lazyHandler(() => import("./turn_driver/settlement.ts"), ({reduceTurnSettlementTransaction}) => reduceTurnSettlementTransaction)],
["turn.host_todo_completion.evaluate", lazyHandler(() => import("./turn_driver/host_todo_completion.ts"), ({evaluateHostTodoCompletion}) => evaluateHostTodoCompletion)],
["work_item.replan_settlement.project", lazyHandler(() => import("./work_items/replan_settlement.ts"), ({projectReplanSettlementContract}) => projectReplanSettlementContract)],
["work_item.replan_semantics.project", lazyHandler(() => import("./work_items/replan_semantics.ts"), ({projectReplanSemantics}) => projectReplanSemantics)],
+ ["work_item.replan_context.project", lazyHandler(() => import("./work_items/replan_context.ts"), ({projectReplanContext}) => projectReplanContext)],
+ ["work_item.replan_context.project_snapshot", lazyHandler(() => import("./work_items/replan_context.ts"), ({projectReplanContextSnapshot}) => projectReplanContextSnapshot)],
["explore.research.normalize", lazyHandler(() => import("./capabilities/explore_research.ts"), ({normalizeResearchObservation}) => normalizeResearchObservation)],
["explore.research.validate_attribution", lazyHandler(() => import("./capabilities/explore_research.ts"), ({validateResearchAttribution}) => validateResearchAttribution)],
["explore.research.frontier", lazyHandler(() => import("./capabilities/explore_research.ts"), ({projectResearchFrontier}) => projectResearchFrontier)],
["work_item.replan_history.project", lazyHandler(() => import("./work_items/replan_history.ts"), ({projectReplanHistory}) => projectReplanHistory)],
["work_item.replan_history.project_snapshot", lazyHandler(() => import("./work_items/replan_history_snapshot.ts"), ({projectReplanHistorySnapshot}) => projectReplanHistorySnapshot)],
+
[
"work_item.replan_settlement.reentry",
lazyHandler(() => import("./work_items/replan_settlement.ts"), ({projectTodoLifecycleSettlementReentry}) => projectTodoLifecycleSettlementReentry),
diff --git a/loopx/control_plane/goals/goal_frontier/__init__.py b/loopx/control_plane/goals/goal_frontier/__init__.py
index ea0b108b8e..0d1c29b83b 100644
--- a/loopx/control_plane/goals/goal_frontier/__init__.py
+++ b/loopx/control_plane/goals/goal_frontier/__init__.py
@@ -87,7 +87,6 @@
latest_autonomous_replan_ack_from_status_payload,
latest_missing_vision_checkpoint_from_status_payload,
latest_replan_ack_feedback_from_status_payload,
- latest_runs_for_goal,
)
from .terminal import (
GOAL_TERMINAL_SOURCE_COMPLETENESS_SCHEMA_VERSION, # noqa: F401
@@ -295,8 +294,11 @@ def autonomous_replan_scope_decision(
else:
selected_peer_agent = select_peer_for_work(
registered_agent_ids or [],
+ # Display evidence and prose are mutable projections, not work
+ # identity. Normalize before both selection and permission checks.
work_key=peer_work_key(
- replan_obligation,
+ {"obligation_id": replan_obligation.get("obligation_id")
+ or ensure_replan_novelty_policy(replan_obligation)["obligation_id"]},
fallback="autonomous_replan",
),
)
@@ -1308,7 +1310,7 @@ def derive_goal_frontier_replan_obligation_from_summaries(
}
],
guidance_actions=[
- "read_evidence_log",
+ "use_replan_context",
"run_bounded_public_research_if_local_evidence_is_missing",
"group_or_prune_todo_chain",
"update_agent_vision",
@@ -1741,10 +1743,8 @@ def build_goal_frontier_projection_context_from_status(
replan_obligation,
goal_id=goal_id,
agent_id=agent_id,
- newest_first_runs=latest_runs_for_goal(
- status_payload,
- goal_id=goal_id,
- ),
+ newest_first_runs=(), source_status=status_payload,
+ goal_acceptance_contract=(agent_todo_summary or {}).get("goal_acceptance_contract"),
)
goal_frontier_projection = build_goal_frontier_projection_from_summaries(
@@ -1792,7 +1792,7 @@ def compact_replan_obligation(replan_obligation: dict[str, Any]) -> dict[str, An
# Keep hot quota/status packets on the two authoritative seams. The
# detailed selection hint remains available on the full obligation.
compact["replan_novelty_policy"] = {
- "evidence_source": "agent_scoped_evidence_log",
+ "evidence_source": "compact_run_history",
"delivery": "host_projected",
"writeback": "typed_semantic_delta",
}
diff --git a/loopx/control_plane/goals/goal_frontier/semantic_history.py b/loopx/control_plane/goals/goal_frontier/semantic_history.py
index 0beef33c7f..cb2357e75b 100644
--- a/loopx/control_plane/goals/goal_frontier/semantic_history.py
+++ b/loopx/control_plane/goals/goal_frontier/semantic_history.py
@@ -51,16 +51,6 @@ def _latest_runs_for_goal(
)
-def latest_runs_for_goal(
- status_payload: dict[str, Any],
- *,
- goal_id: str,
-) -> list[dict[str, Any]]:
- """Return newest-first compact runs used by semantic reducers."""
-
- return _latest_runs_for_goal(status_payload, goal_id=goal_id)
-
-
def _semantic_agent_context_for_goal(
status_payload: dict[str, Any],
*,
diff --git a/loopx/control_plane/handoff/project_agent_context.py b/loopx/control_plane/handoff/project_agent_context.py
index 5c2d4968cb..639a1fcf8a 100644
--- a/loopx/control_plane/handoff/project_agent_context.py
+++ b/loopx/control_plane/handoff/project_agent_context.py
@@ -14,7 +14,7 @@
agent_member_from_item,
agent_member_summary,
agent_todo_texts_for_handoff,
- project_agent_required_reads,
+ project_agent_replan_context,
project_asset_source,
project_asset_source_line,
)
@@ -349,7 +349,7 @@ def project_agent_section(
agent_member_text: str | None = None,
handoff_followthrough_text: str | None = None,
handoff_delivery_contract_text: str | None = None,
- required_reads: list[dict[str, Any]] | None = None,
+ replan_context: dict[str, Any] | None = None,
approved_operator_gate: bool = False,
connected_delivery: bool = False,
) -> str:
@@ -382,20 +382,14 @@ def project_agent_section(
if handoff_delivery_contract_text
else None
)
- first_required_read = next(
- (
- item
- for item in (required_reads or [])
- if isinstance(item, dict) and item.get("command")
- ),
- None,
- )
+ evidence_lines = []
+ if replan_context is not None:
+ for row in replan_context["evidence"][:1]:
+ summary = row["summary"]
+ evidence_lines.append(f"{row['evidence_ref']}: {summary}")
required_read_line = (
- "必读流水账:replan/接力前运行 "
- f"`{compact_shell_command(str(first_required_read.get('command') or ''))}`;"
- "只展开本 agent,其他 agent 只看 frontier。"
- if first_required_read
- else None
+ "重规划上下文(replan_context):" + (";".join(evidence_lines) or "当前 Agent 没有匹配证据。")
+ if replan_context is not None else None
)
context_lines = [
goal_guard,
@@ -484,7 +478,7 @@ class ProjectAgentContext:
authority_summary: str | None
followthrough_summary: str | None
delivery_contract: dict[str, Any] | None
- required_reads: list[dict[str, Any]]
+ replan_context: dict[str, Any] | None
approved_handoff: bool
delivery_handoff: bool
@@ -510,7 +504,7 @@ def payload(self) -> dict[str, Any]:
handoff_delivery_contract_text=handoff_delivery_contract_summary(
self.delivery_contract
),
- required_reads=self.required_reads,
+ replan_context=self.replan_context,
approved_operator_gate=self.approved_handoff,
connected_delivery=self.delivery_handoff,
)
@@ -535,7 +529,7 @@ def payload(self) -> dict[str, Any]:
"authority_summary": self.authority_summary,
"handoff_followthrough_summary": self.followthrough_summary,
"handoff_delivery_contract": self.delivery_contract,
- "project_agent_required_reads": self.required_reads,
+ "replan_context": self.replan_context,
"handoff_interface_budget": build_handoff_interface_budget(text),
"project_asset_source": self.asset_source,
}
@@ -575,7 +569,9 @@ def assemble_project_agent_context(
authority_summary=authority_material_summary(goal),
followthrough_summary=handoff_followthrough_summary(item),
delivery_contract=handoff_delivery_contract(item),
- required_reads=project_agent_required_reads(goal_id, item),
+ replan_context=project_agent_replan_context(
+ goal_id, item, status_payload,
+ ),
approved_handoff=operator_gate_approved_handoff(item, goal),
delivery_handoff=connected_delivery_handoff(item, goal) and kind == "codex",
)
diff --git a/loopx/control_plane/handoff/review_packet_context.py b/loopx/control_plane/handoff/review_packet_context.py
index 42b9a86b36..b641b9896f 100644
--- a/loopx/control_plane/handoff/review_packet_context.py
+++ b/loopx/control_plane/handoff/review_packet_context.py
@@ -2,7 +2,6 @@
from typing import Any
-from ..runtime.agent_scoped_evidence_log import build_agent_scoped_required_read
from ..runtime.public_safety import compact_text
@@ -178,23 +177,25 @@ def agent_member_summary(item: dict[str, Any] | None) -> str | None:
return compact_packet_text(" ".join(str(part) for part in parts if part))
-def project_agent_required_reads(
- goal_id: str,
- item: dict[str, Any] | None,
-) -> list[dict[str, Any]]:
+def project_agent_replan_context(
+ goal_id: str, item: dict[str, Any] | None, status_payload: dict[str, Any],
+) -> dict[str, Any] | None:
+ from ..work_items.replan_context_codec import project_replan_context
+
member = agent_member_from_item(item)
if not member:
- return []
+ return None
agent_id = str(member.get("agent_id") or "").strip()
- read = build_agent_scoped_required_read(
- goal_id=goal_id,
- agent_id=agent_id,
- reason=(
- "read this target agent's thin evidence ledger before replan or "
- "handoff continuation; other agents stay frontier-only"
- ),
- )
- return [read] if read else []
+ if not agent_id:
+ return None
+ asset = (item or {}).get("project_asset") or {}
+ acceptance = next((candidate for candidate in (
+ (item or {}).get("goal_acceptance_contract"), asset.get("goal_acceptance_contract"),
+ ((item or {}).get("agent_todos") or {}).get("goal_acceptance_contract"),
+ (asset.get("agent_todos") or {}).get("goal_acceptance_contract"),
+ ) if isinstance(candidate, dict) and candidate.get("enabled") is True), None)
+ return project_replan_context(goal_id=goal_id, agent_id=agent_id, runs=(), source_status=status_payload,
+ goal_acceptance_contract=acceptance)
def project_asset_source(item: dict[str, Any] | None) -> str:
diff --git a/loopx/control_plane/quota/should_run_prepare.py b/loopx/control_plane/quota/should_run_prepare.py
index 22a1a96779..c9bcf3a8b3 100644
--- a/loopx/control_plane/quota/should_run_prepare.py
+++ b/loopx/control_plane/quota/should_run_prepare.py
@@ -183,6 +183,7 @@ def _preserve_receipt_bound_replan_obligation(
replan_obligation: Mapping[str, Any] | None,
receipt_bound_replan_obligation_id: str | None,
*, guard_scoped: bool = False,
+ agent_id: str | None = None,
replay_phase: ReceiptBoundReplayPhase | None = None,
transition_candidates: list[dict[str, Any]] | None = None,
) -> dict[str, Any] | None:
@@ -196,6 +197,7 @@ def _preserve_receipt_bound_replan_obligation(
"operation": "receipt_bound_obligation",
"current_obligation": dict(replan_obligation) if replan_obligation is not None else None,
"selected_obligation_id": preserved_replan_id, "guard_scoped": guard_scoped,
+ "agent_id": agent_id,
"replay_phase": replay_phase.value if replay_phase is not None else None,
"transition_candidates": transition_candidates or [],
})
@@ -838,6 +840,7 @@ def _prepare_quota_should_run_item(
goal_frontier_context.get("replan_obligation"),
receipt_bound_replan_obligation_id,
guard_scoped=receipt_bound_replan_guard_scoped,
+ agent_id=agent_frontier_id,
replay_phase=receipt_bound_replay_phase,
transition_candidates=goal_frontier_context.get("replan_transition_candidates"),
)
diff --git a/loopx/control_plane/runtime/agent_scoped_evidence_log.py b/loopx/control_plane/runtime/agent_evidence_history.py
similarity index 77%
rename from loopx/control_plane/runtime/agent_scoped_evidence_log.py
rename to loopx/control_plane/runtime/agent_evidence_history.py
index bdd15b5ae3..3b9095fa6b 100644
--- a/loopx/control_plane/runtime/agent_scoped_evidence_log.py
+++ b/loopx/control_plane/runtime/agent_evidence_history.py
@@ -1,6 +1,5 @@
from __future__ import annotations
-import shlex
from collections.abc import Iterable, Mapping
from datetime import datetime
from typing import Any
@@ -11,7 +10,6 @@
SCHEMA_VERSION = "agent_scoped_evidence_log_v0"
-REQUIRED_READ_SCHEMA_VERSION = "loopx_agent_required_read_v0"
READ_RECEIPT_SCHEMA_VERSION = "evidence_log_read_receipt_v0"
MAX_PROJECTED_READ_RECEIPTS = 12
@@ -221,100 +219,25 @@ def goal_history_runs(
) -> list[dict[str, Any]]:
"""Select compact history rows for one goal from either supported history shape."""
- goals = history_payload.get("goals")
- for goal in goals if isinstance(goals, list) else []:
- if not isinstance(goal, Mapping) or str(goal.get("id") or "") != goal_id:
- continue
- latest_runs = goal.get("latest_runs")
- if not isinstance(latest_runs, list):
+ if history_payload.get("ok") is False:
+ raise ValueError("history source failed")
+ if "goals" in history_payload:
+ goals = history_payload["goals"]
+ if not isinstance(goals, list) or any(not isinstance(goal, Mapping) for goal in goals):
+ raise ValueError("history.goals must be an array of objects")
+ for goal in goals:
+ if str(goal.get("id") or "") == goal_id:
+ runs = goal.get("latest_runs")
+ break
+ else:
return []
- return [dict(row) for row in latest_runs if isinstance(row, Mapping)]
- runs = history_payload.get("runs")
- return [
- dict(row)
- for row in (runs if isinstance(runs, list) else [])
- if isinstance(row, Mapping) and str(row.get("goal_id") or "") == goal_id
- ]
-
-
-def build_agent_scoped_evidence_log_command(
- *,
- goal_id: str,
- agent_id: str,
- todo_id: str | None = None,
- cli_bin: str = "loopx",
- output_format: str = "json",
- limit: int = 24,
- required_read_id: str | None = None,
-) -> str:
- safe_goal_id = _compact_text(goal_id, limit=180)
- safe_agent_id = _compact_text(agent_id, limit=180)
- if not safe_goal_id:
- raise ValueError("goal_id is required")
- if not safe_agent_id:
- raise ValueError("agent_id is required")
- safe_todo_id = _compact_text(todo_id, limit=180) if todo_id else None
- parts = [
- cli_bin or "loopx",
- "--format",
- output_format or "json",
- "evidence-log",
- "--goal-id",
- safe_goal_id,
- "--agent-id",
- safe_agent_id,
- "--thin",
- "--limit",
- str(max(0, int(limit))),
- ]
- if safe_todo_id:
- parts.extend(["--todo-id", safe_todo_id])
- safe_required_read_id = (
- _compact_text(required_read_id, limit=180) if required_read_id else None
- )
- if safe_required_read_id:
- parts.extend(["--required-read-id", safe_required_read_id])
- return " ".join(shlex.quote(part) for part in parts)
-
-
-def build_agent_scoped_required_read(
- *,
- goal_id: str,
- agent_id: str | None,
- todo_id: str | None = None,
- reason: str = "read this agent's thin public-safe evidence ledger before replan",
- cli_bin: str = "loopx",
- limit: int = 24,
- required_read_id: str | None = None,
-) -> dict[str, Any] | None:
- safe_agent_id = _compact_text(agent_id, limit=180) if agent_id else None
- if not safe_agent_id:
- return None
- command = build_agent_scoped_evidence_log_command(
- goal_id=goal_id,
- agent_id=safe_agent_id,
- todo_id=todo_id,
- cli_bin=cli_bin,
- limit=limit,
- required_read_id=required_read_id,
- )
- required_read = {
- "schema_version": REQUIRED_READ_SCHEMA_VERSION,
- "kind": "agent_scoped_evidence_log",
- "goal_id": _compact_text(goal_id, limit=180),
- "agent_id": safe_agent_id,
- "todo_id": _compact_text(todo_id, limit=180) if todo_id else None,
- "mode": "thin",
- "command": command,
- "reason": _compact_text(reason, limit=180),
- "other_agent_policy": "frontier_only",
- }
- if required_read_id:
- required_read["required_read_id"] = _compact_text(
- required_read_id,
- limit=180,
- )
- return required_read
+ elif "runs" in history_payload:
+ runs = history_payload["runs"]
+ else:
+ raise ValueError("history source is missing goals/runs")
+ if not isinstance(runs, list) or any(not isinstance(row, Mapping) for row in runs):
+ raise ValueError("history runs must be an array of objects")
+ return [dict(row) for row in runs if row.get("goal_id", goal_id) == goal_id]
def _other_agent_frontier(
diff --git a/loopx/control_plane/testing/cli_output_budget.py b/loopx/control_plane/testing/cli_output_budget.py
index bef3e8a80d..7fbf59082c 100644
--- a/loopx/control_plane/testing/cli_output_budget.py
+++ b/loopx/control_plane/testing/cli_output_budget.py
@@ -140,16 +140,21 @@ class CliOutputCommandClassification:
markdown_anchor="# LoopX Quota Should Run",
max_chars={
"small": {"json": 20_000, "markdown": 6_700},
- "crowded": {"json": 30_000, "markdown": 7_800},
+ "crowded": {"json": 34_000, "markdown": 7_800},
"multi_agent": {"json": 23_000, "markdown": 7_000},
},
max_lines={
"small": {"json": 520, "markdown": 72},
- "crowded": {"json": 750, "markdown": 78},
+ "crowded": {"json": 830, "markdown": 78},
"multi_agent": {"json": 650, "markdown": 75},
},
scale_axis="todo_count",
max_json_growth_chars_per_unit=300,
+ # Required replan carries dense decision evidence from the full index.
+ # The unchanged fixture grows 26,844 -> 33,523 chars / 718 -> 806 lines;
+ # ordinary and multi-agent non-replan guards remain byte-identical.
+ # This fixed decision packet must not relax per-Todo growth or other routes.
+ max_json_fixed_semantic_growth_chars=6_000,
),
CliOutputBudgetSpec(
surface_id="loopx_turn_plan",
@@ -230,16 +235,19 @@ class CliOutputCommandClassification:
markdown_anchor="# LoopX Diagnosis Packet",
max_chars={
"small": {"json": 21_000, "markdown": 4_300},
- "crowded": {"json": 34_000, "markdown": 4_500},
+ "crowded": {"json": 44_000, "markdown": 4_500},
"multi_agent": {"json": 21_000, "markdown": 4_300},
},
max_lines={
"small": {"json": 470, "markdown": 72},
- "crowded": {"json": 720, "markdown": 72},
+ "crowded": {"json": 850, "markdown": 72},
"multi_agent": {"json": 480, "markdown": 72},
},
scale_axis="todo_count",
max_json_growth_chars_per_unit=520,
+ # The existing selected + Goal-array diagnostic contract includes the
+ # required-replan context twice. Keep that cold-path caller contract.
+ max_json_fixed_semantic_growth_chars=7_000,
),
CliOutputBudgetSpec(
surface_id="review_packet_handoff_only",
@@ -343,28 +351,7 @@ class CliOutputCommandClassification:
scale_axis="returned_run_count",
max_json_growth_chars_per_unit=1_600,
),
- CliOutputBudgetSpec(
- surface_id="evidence_log_thin",
- command="evidence-log --thin --limit 5",
- owner="agent-scoped evidence ledger",
- consumer_action="read bounded public-safe evidence for replan",
- qualification_policy="explicit_limit_cold_path",
- cold_path="referenced run-history and rollout-event artifacts",
- semantic_json_keys=("ledger", "truncated", "other_agent_frontier"),
- markdown_anchor="# LoopX Evidence Log",
- max_chars={
- "small": {"json": 2_900, "markdown": 800},
- "crowded": {"json": 3_500, "markdown": 1_100},
- "multi_agent": {"json": 4_300, "markdown": 1_200},
- },
- max_lines={
- "small": {"json": 100, "markdown": 26},
- "crowded": {"json": 120, "markdown": 30},
- "multi_agent": {"json": 140, "markdown": 34},
- },
- scale_axis="returned_evidence_count",
- max_json_growth_chars_per_unit=800,
- ),
+
)
@@ -717,12 +704,7 @@ class CliOutputCommandClassification:
surface_id=None,
rationale="explicit operator preview, stop, and resume command family",
),
- CliOutputCommandClassification(
- command_id="evidence-log",
- qualification="qualified_default",
- surface_id="evidence_log_thin",
- rationale="bounded evidence read before replan or handoff",
- ),
+
CliOutputCommandClassification(
command_id="agent-capabilities",
qualification="explicit_cold_path_exception",
diff --git a/loopx/control_plane/testing/cli_output_differential.py b/loopx/control_plane/testing/cli_output_differential.py
index 183a56cd93..036f70f4c0 100644
--- a/loopx/control_plane/testing/cli_output_differential.py
+++ b/loopx/control_plane/testing/cli_output_differential.py
@@ -362,6 +362,30 @@ def _removed(base: dict[str, Any], candidate: dict[str, Any], field: str) -> lis
return sorted(base_values - candidate_values)
+def _has_decision_evidence(row: dict[str, Any]) -> bool:
+ paths = set(row.get("json_shape_paths") or [])
+ return any(path.endswith(".replan_context.core_goal") and
+ path.removesuffix("core_goal") + "evidence" in paths and
+ path.removesuffix("core_goal") + "coverage_ledger" in paths for path in paths)
+
+
+def _decision_evidence_migration_allowance(base: dict[str, Any], candidate: dict[str, Any], metric: Metric) -> int:
+ # One-time, measured transition from coverage-only to dense decision context.
+ # Future dense->dense edits and ordinary guard rows keep their original budget.
+ if candidate.get("format") != "json" or _has_decision_evidence(base) or not _has_decision_evidence(candidate):
+ return 0
+ row_id = str(candidate["row_id"])
+ if row_id == "surface/quota_should_run/crowded/json":
+ limits = (7_000, 7_000, 96, 6_000)
+ elif row_id == "surface/diagnose/crowded/json":
+ limits = (15_000, 15_000, 190, 12_000)
+ elif row_id == "variant/review_packet_full/small/json":
+ limits = (800, 800, 24, 650)
+ else:
+ return 0
+ return dict(zip(("chars", "utf8_bytes", "lines", "compact_payload_chars"), limits))[metric]
+
+
def _action_signature_migration(
base: dict[str, Any], candidate: dict[str, Any]
) -> str | None:
@@ -726,6 +750,7 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
metric,
),
_schema_migration_growth_allowance(migration, metric),
+ _decision_evidence_migration_allowance(base, candidate, metric),
projection_allowance.get(metric, 0),
)
# Thin installed prompts contain bilingual lifecycle instructions. A
@@ -882,14 +907,22 @@ def compare_cli_output_receipts(
}
)
elif candidate is None:
+ parts = row_id.split("/")
+ replacement = next((row for key, row in candidate_rows.items()
+ if key.startswith(("variant/review_packet_full/", "surface/quota_should_run/"))
+ and _has_decision_evidence(row)), None)
+ retired_evidence_command = bool(
+ len(parts) == 4 and parts[:2] == ["surface", "evidence_log_thin"]
+ and replacement and _has_decision_evidence(replacement)
+ )
results.append(
{
"row_id": row_id,
- "status": "failed",
+ "status": "passed" if retired_evidence_command else "failed",
"deltas": {},
"allowances": {},
- "failures": ["qualified base row is missing from candidate"],
- "review_signals": [],
+ "failures": [] if retired_evidence_command else ["qualified base row is missing from candidate"],
+ "review_signals": ["evidence-log retired; scoped decision evidence is delivered by replan_context"] if retired_evidence_command else [],
}
)
else:
diff --git a/loopx/control_plane/testing/quota_fixtures.py b/loopx/control_plane/testing/quota_fixtures.py
index dd937c25d6..f733bfd648 100644
--- a/loopx/control_plane/testing/quota_fixtures.py
+++ b/loopx/control_plane/testing/quota_fixtures.py
@@ -207,8 +207,7 @@ def quota_status_payload(
}
if coordination:
goal["coordination"] = coordination
- if latest_runs is not None:
- goal["latest_runs"] = latest_runs
+ goal["latest_runs"] = latest_runs if latest_runs is not None else []
if goal_extra:
goal.update(goal_extra)
diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py
index fd41b9ca64..8f740de088 100644
--- a/loopx/control_plane/testing/replan_semantic_action_behavior.py
+++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py
@@ -550,7 +550,7 @@ def _quota_behavior_observation(packet: Mapping[str, Any]) -> dict[str, Any]:
or replan_action.get("decision") != "replan_required"
or replan_action.get("obligation_id") != obligation.get("obligation_id")
or context.get("delivery") != "host_projected"
- or context.get("evidence_source") != "agent_scoped_evidence_log"
+ or context.get("evidence_source") != "compact_run_history"
or dict(context.get("delivery_receipt") or {}).get("status") != "delivered"
or packet.get("required_reads") not in (None, [])
):
@@ -1137,7 +1137,7 @@ def _dispatch_behavior_command(
"semantic_replan_writeback",
True,
)
- if "evidence-log" in command:
+ if "history" in shlex.split(command):
raise ValueError("manual_evidence_read_is_not_replan")
raise ValueError("unexpected_command")
diff --git a/loopx/control_plane/work_items/autonomous_replan_obligation.py b/loopx/control_plane/work_items/autonomous_replan_obligation.py
index 0e24a471bb..01f3a20982 100644
--- a/loopx/control_plane/work_items/autonomous_replan_obligation.py
+++ b/loopx/control_plane/work_items/autonomous_replan_obligation.py
@@ -40,7 +40,7 @@ def build_replan_novelty_policy() -> dict[str, str]:
return {
"schema_version": REPLAN_NOVELTY_POLICY_SCHEMA_VERSION,
- "evidence_source": "agent_scoped_evidence_log",
+ "evidence_source": "compact_run_history",
"delivery": "host_projected",
"writeback": "typed_semantic_delta",
}
diff --git a/loopx/control_plane/work_items/progress_observation.py b/loopx/control_plane/work_items/progress_observation.py
index a507a3f5e5..31c5169c14 100644
--- a/loopx/control_plane/work_items/progress_observation.py
+++ b/loopx/control_plane/work_items/progress_observation.py
@@ -19,13 +19,10 @@
normalize_progress_identifier,
)
-REPLAN_CONTEXT_SCHEMA_VERSION = "replan_context_v0"
-REPLAN_CONTEXT_RECEIPT_SCHEMA_VERSION = "replan_context_delivery_receipt_v0"
REPLAN_ACTION_PACKET_SCHEMA_VERSION = "replan_action_packet_v0"
PROGRESS_REPEAT_TRIGGER_KIND = "typed_progress_repeat"
PROGRESS_REPEAT_THRESHOLD = 2
MAX_PROGRESS_EVIDENCE_IDS = 12
-MAX_COVERAGE_LEDGER_ITEMS = 6
SEMANTIC_DIMENSIONS = (
"surface_id",
@@ -406,6 +403,7 @@ def semantic_delta_from_writeback(
obligation: Mapping[str, Any],
progress_observation: Mapping[str, Any] | None,
agent_vision: Mapping[str, Any] | None = None,
+ history_runs: Iterable[Mapping[str, Any]] | None = None,
) -> dict[str, Any]:
"""Qualify concrete typed writeback evidence against one obligation.
@@ -437,10 +435,27 @@ def semantic_delta_from_writeback(
None,
)
context = obligation.get("replan_context")
- coverage = context.get("coverage_ledger") if isinstance(context, Mapping) else None
+ if context is not None:
+ from .replan_context_codec import validate_replan_context
+ context = validate_replan_context(context)
+ if context["obligation_id"] != obligation.get("obligation_id"):
+ raise ValueError("replan context obligation identity mismatch")
+ coverage = context["coverage_ledger"] if context is not None else None
claimed = list(window) if isinstance(window, list) else []
if isinstance(coverage, list):
claimed.extend(coverage)
+ if history_runs is not None and context is not None:
+ from .replan_context_codec import replan_evidence_rows
+
+ # The readable context has a display budget; admission does not forget
+ # an older claim merely because it was omitted from that display.
+ claimed.extend(
+ row["progress_observation"]
+ for row in replan_evidence_rows(history_runs, goal_id=context["goal_id"], agent_id=context["agent_id"])
+ if row["progress_observation"] is not None
+ )
+ elif context is not None and context.get("coverage_truncated"):
+ raise ValueError("replan writeback requires the complete evidence history when coverage is truncated")
observation_delta = semantic_progress_delta(
progress_observation,
baseline=baseline if isinstance(baseline, Mapping) else None,
@@ -464,90 +479,19 @@ def semantic_delta_from_writeback(
}
-def _latest_typed_observations(
- newest_first_runs: Iterable[Mapping[str, Any]],
- *,
- agent_id: str | None,
-) -> list[dict[str, Any]]:
- normalized_agent_id = str(agent_id or "").strip()
- observations: list[dict[str, Any]] = []
- for run in newest_first_runs:
- run_agent_id = str(run.get("agent_id") or "").strip()
- if normalized_agent_id and run_agent_id not in {"", normalized_agent_id}:
- continue
- observation = progress_observation_from_run(run)
- if observation is None:
- continue
- observations.append(
- {
- **observation,
- "generated_at": str(run.get("generated_at") or ""),
- }
- )
- if len(observations) >= MAX_COVERAGE_LEDGER_ITEMS:
- break
- return observations
-
-
def build_replan_context(
- obligation: Mapping[str, Any],
- *,
- goal_id: str,
- agent_id: str | None,
+ obligation: Mapping[str, Any], *, goal_id: str, agent_id: str | None,
newest_first_runs: Iterable[Mapping[str, Any]],
+ source_status: Mapping[str, Any] | None = None,
+ goal_acceptance_contract: Mapping[str, Any] | None = None,
) -> dict[str, Any]:
- """Project compact evidence context into the action packet on the host."""
+ from .replan_context_codec import project_replan_context
- obligation_id = _stable_id(
- obligation.get("obligation_id"),
- field="obligation_id",
- required=True,
- )
- coverage_ledger = _latest_typed_observations(
- newest_first_runs,
- agent_id=agent_id,
+ return project_replan_context(
+ goal_id=goal_id, agent_id=agent_id or obligation.get("agent_id"),
+ runs=newest_first_runs, obligation=obligation, source_status=source_status,
+ goal_acceptance_contract=goal_acceptance_contract,
)
- baseline = obligation.get("progress_baseline")
- if not isinstance(baseline, Mapping):
- baseline = next(
- (
- trigger.get("progress_baseline")
- for trigger in obligation.get("triggers") or []
- if isinstance(trigger, Mapping)
- and isinstance(trigger.get("progress_baseline"), Mapping)
- ),
- None,
- )
- uncovered_frontier = {
- "baseline": dict(baseline) if isinstance(baseline, Mapping) else None,
- "required_any_of": required_semantic_outcomes(obligation),
- }
- context_identity = {
- "goal_id": goal_id,
- "agent_id": str(agent_id or "").strip() or None,
- "obligation_id": obligation_id,
- "coverage_fingerprints": [
- item["fingerprint"] for item in coverage_ledger
- ],
- "uncovered_frontier": uncovered_frontier,
- }
- context_id = "replan-context-" + _digest(context_identity)
- return {
- "schema_version": REPLAN_CONTEXT_SCHEMA_VERSION,
- "context_id": context_id,
- "obligation_id": obligation_id,
- "evidence_source": "agent_scoped_evidence_log",
- "delivery": "host_projected",
- "coverage_ledger": coverage_ledger,
- "uncovered_frontier": uncovered_frontier,
- "delivery_receipt": {
- "schema_version": REPLAN_CONTEXT_RECEIPT_SCHEMA_VERSION,
- "context_id": context_id,
- "obligation_id": obligation_id,
- "status": "delivered",
- "delivered_by": "quota_host_projection",
- },
- }
def build_replan_action_packet(
@@ -561,8 +505,10 @@ def build_replan_action_packet(
if isinstance(settlement_packet, Mapping):
return dict(settlement_packet)
context = obligation.get("replan_context")
- if not isinstance(context, Mapping):
- raise TypeError("replan obligation is missing host-projected context")
+ from .replan_context_codec import validate_replan_context
+ context = validate_replan_context(context)
+ if context["obligation_id"] != obligation.get("obligation_id"):
+ raise ValueError("replan context obligation identity mismatch")
todo_actions = obligation.get("todo_actions")
todo_action = next(
(
@@ -646,7 +592,7 @@ def build_replan_action_packet(
"schema_version": REPLAN_ACTION_PACKET_SCHEMA_VERSION,
"decision": "replan_required",
"obligation_id": obligation.get("obligation_id"),
- "uncovered_frontier": context.get("uncovered_frontier"),
+ "uncovered_frontier": context["uncovered_frontier"],
"required_outcome": "semantic_delta",
"planning_guidance": requirements["planning_guidance"],
"writeback_contract": writeback_contract,
diff --git a/loopx/control_plane/work_items/replan_context.ts b/loopx/control_plane/work_items/replan_context.ts
new file mode 100644
index 0000000000..02a5b2d793
--- /dev/null
+++ b/loopx/control_plane/work_items/replan_context.ts
@@ -0,0 +1,196 @@
+/** Bounded evidence for the existing replan boundary; no settlement authority. */
+import {createHash} from "node:crypto";
+import type {JsonObject} from "../effect_program.ts";
+import {EffectRuntimeRequestError} from "../effect_runtime_errors.ts";
+import {requireJsonObject, requireNonEmptyString} from "../runtime_decode.ts";
+import {requiredSemanticOutcomes} from "./replan_semantics.ts";
+import {readReplanSnapshot} from "./replan_history_snapshot.ts";
+
+const CONTEXT_SCHEMA = "replan_context_v0";
+// Replan is an infrequent decision, not the ordinary guard's display window.
+// Bound distinct observations after repetition reduction, not raw newest runs.
+const MAX_EVIDENCE_ROWS = 24;
+const ID = /^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/;
+
+function identifier(value: unknown, label: string): string {
+ const result = requireNonEmptyString(value, label);
+ if (!ID.test(result)) throw new EffectRuntimeRequestError(`${label} must be a public-safe identifier`);
+ return result;
+}
+
+function objects(value: unknown, label: string): JsonObject[] {
+ if (!Array.isArray(value)) throw new EffectRuntimeRequestError(`${label} must be an array`);
+ return value.map(item => requireJsonObject(item, label));
+}
+
+function digest(value: unknown): string {
+ return createHash("sha256").update(JSON.stringify(value)).digest("hex").slice(0, 24);
+}
+
+function coreGoal(value: unknown): JsonObject {
+ const facts = value == null ? {} : requireJsonObject(value, "goal_facts");
+ const active = facts.active_state_objective;
+ const registered = facts.registry_objective;
+ for (const objective of [active, registered]) {
+ if (objective != null && typeof objective !== "string") throw new EffectRuntimeRequestError("Goal objective must be text");
+ }
+ const objective = active || registered || null;
+ const acceptance = facts.acceptance_contract == null ? null
+ : requireJsonObject(facts.acceptance_contract, "goal_acceptance_contract");
+ return {objective, objective_source: active ? "active_state" : registered ? "registry" : null,
+ objective_missing: objective === null, acceptance_contract: acceptance};
+}
+
+export function validateReplanContext(value: unknown): JsonObject {
+ const context = requireJsonObject(value, "replan_context");
+ if (context.schema_version !== CONTEXT_SCHEMA) {
+ throw new EffectRuntimeRequestError("unsupported replan_context schema_version");
+ }
+ identifier(context.goal_id, "replan_context.goal_id");
+ if (context.agent_id !== null) identifier(context.agent_id, "replan_context.agent_id");
+ identifier(context.context_id, "replan_context.context_id");
+ const core = requireJsonObject(context.core_goal, "replan_context.core_goal");
+ if ((core.objective !== null && typeof core.objective !== "string") ||
+ core.objective_missing !== (core.objective === null)) {
+ throw new EffectRuntimeRequestError("replan_context core_goal requires objective and explicit missing state");
+ }
+ if (core.acceptance_contract !== null) requireJsonObject(core.acceptance_contract, "core_goal.acceptance_contract");
+ objects(context.coverage_ledger, "replan_context.coverage_ledger");
+ for (const row of objects(context.evidence, "replan_context.evidence")) {
+ identifier(row.evidence_ref, "evidence.evidence_ref");
+ requireNonEmptyString(row.generated_at, "evidence.generated_at");
+ }
+ const frontier = requireJsonObject(context.uncovered_frontier, "replan_context.uncovered_frontier");
+ if (!Array.isArray(frontier.required_any_of)) {
+ throw new EffectRuntimeRequestError("replan_context uncovered_frontier requires required_any_of");
+ }
+ return context;
+}
+
+export async function projectReplanContextSnapshot(value: unknown): Promise {
+ return projectReplanContext(await readReplanSnapshot(value));
+}
+
+export function projectReplanContext(value: unknown): JsonObject {
+ const request = requireJsonObject(value, "replan context request");
+ if (request.operation === "validate") return validateReplanContext(request.context);
+ if (request.operation !== "project" && request.operation !== "resolve") {
+ throw new EffectRuntimeRequestError("unsupported replan context operation");
+ }
+ const goal = identifier(request.goal_id, "goal_id");
+ const agent = request.agent_id === null ? null : identifier(request.agent_id, "agent_id");
+ const obligation = request.obligation === null ? null : requireJsonObject(request.obligation, "obligation");
+ const obligationId = obligation ? identifier(obligation.obligation_id, "obligation_id") : null;
+ const input = objects(request.rows, "rows");
+ for (const row of input) {
+ identifier(row.goal_id, "evidence.goal_id");
+ if (row.agent_id !== null) identifier(row.agent_id, "evidence.agent_id");
+ }
+ const rows = input.filter(row => row.goal_id === goal && (agent === null || row.agent_id === agent));
+ for (const row of rows) {
+ requireNonEmptyString(row.generated_at, "evidence.generated_at");
+ if (typeof row.observed_at !== "number" || !Number.isFinite(row.observed_at)) {
+ throw new EffectRuntimeRequestError("evidence.observed_at must be a valid timestamp");
+ }
+ if (row.progress_observation !== null) {
+ const progress = requireJsonObject(row.progress_observation, "progress_observation");
+ if (progress.schema_version !== "typed_progress_observation_v0") {
+ throw new EffectRuntimeRequestError("unsupported progress_observation schema_version");
+ }
+ identifier(progress.fingerprint, "progress_observation.fingerprint");
+ }
+ }
+ rows.sort((a, b) => Number(b.observed_at) - Number(a.observed_at) ||
+ JSON.stringify(a).localeCompare(JSON.stringify(b)));
+ const unique = new Map();
+ for (const row of rows) {
+ const {observed_at: _time, ...publicRow} = row;
+ const ref = "replan-evidence-" + digest(publicRow);
+ if (!unique.has(ref)) unique.set(ref, {...publicRow, evidence_ref: ref});
+ }
+ const evidence = [...unique.values()];
+ if (request.operation === "resolve") {
+ if (!agent) throw new EffectRuntimeRequestError("evidence resolution requires agent_id");
+ const ref = identifier(request.evidence_ref, "evidence_ref");
+ const match = unique.get(ref);
+ if (!match) throw new EffectRuntimeRequestError(
+ "evidence reference is unavailable in this Goal and Agent history window; refresh replan_context",
+ "replan_evidence_unavailable");
+ return {ok: true, goal_id: goal, agent_id: agent, evidence: match};
+ }
+ const coverage = new Map();
+ for (const row of evidence) {
+ if (row.progress_observation === null) continue;
+ const progress = requireJsonObject(row.progress_observation, "progress_observation");
+ const fingerprint = String(progress.fingerprint);
+ if (!coverage.has(fingerprint)) coverage.set(fingerprint, {...progress, generated_at: row.generated_at});
+ }
+ const baseline = obligation?.progress_baseline ?? (
+ Array.isArray(obligation?.triggers)
+ ? (obligation.triggers as JsonObject[]).find(trigger => trigger.progress_baseline)?.progress_baseline
+ : null
+ ) ?? null;
+ const frontier = {baseline, required_any_of: obligation ? requiredSemanticOutcomes(obligation) : []};
+ const prefix = request.read_prefix === undefined ? "loopx" : requireNonEmptyString(request.read_prefix, "read_prefix");
+ const groups = new Map();
+ for (const row of evidence) {
+ const {generated_at: _time, evidence_ref: _ref, ...facts} = row;
+ const key = digest(facts);
+ const group = groups.get(key);
+ if (group) { group.count++; group.first = row.generated_at; }
+ else groups.set(key, {row, count: 1, first: row.generated_at});
+ }
+ const candidates = [...groups.values()];
+ const selected = new Set();
+ // Preserve observed result diversity, then distinct routes, then recency.
+ // No prose classifier infers that an unchanged score proves a hypothesis false.
+ for (const dimensions of [["result_class"], ["surface_id", "hypothesis_id", "probe_kind"]]) {
+ const seen = new Set();
+ for (const candidate of candidates) {
+ const progress = candidate.row.progress_observation as JsonObject | null;
+ if (!progress) continue;
+ const key = JSON.stringify(dimensions.map(field => progress[field] ?? null));
+ if (!seen.has(key) && selected.size < MAX_EVIDENCE_ROWS) selected.add(candidate);
+ seen.add(key);
+ }
+ }
+ for (const candidate of candidates) {
+ if (selected.size < MAX_EVIDENCE_ROWS) selected.add(candidate);
+ }
+ const chosen = candidates.filter(candidate => selected.has(candidate));
+ const omitted = candidates.filter(candidate => !selected.has(candidate));
+ const shown = chosen.map(({row, count, first}) => ({
+ generated_at: row.generated_at,
+ evidence_ref: row.evidence_ref,
+ summary: [...new Set([row.health_check, row.recommended_action].filter(Boolean))].join(" ") ||
+ row.classification || "Recorded observation",
+ ...(row.delivery_outcome ? {delivery_outcome: row.delivery_outcome} : {}),
+ ...(row.progress_observation ? {coverage_ref: (row.progress_observation as JsonObject).fingerprint} : {}),
+ ...(count > 1 ? {occurrences: count, first_observed_at: first} : {}),
+ ...(agent ? {read_action: `${prefix} --format json history --goal-id ${goal} --agent-id ${agent} --evidence-ref ${row.evidence_ref}`} : {}),
+ }));
+ const selectedCoverage = new Set(chosen.flatMap(({row}) => row.progress_observation
+ ? [String((row.progress_observation as JsonObject).fingerprint)] : []));
+ const ledger = [...coverage].filter(([key]) => selectedCoverage.has(key)).map(([, value]) => value);
+ const core = coreGoal(request.goal_facts);
+ const contextId = "replan-context-" + digest({goal, agent, obligationId, core, ledger, shown, frontier});
+ const context: JsonObject = {
+ schema_version: CONTEXT_SCHEMA, context_id: contextId, goal_id: goal, agent_id: agent,
+ obligation_id: obligationId, evidence_source: "compact_run_history", delivery: "host_projected",
+ from_full_index: request.from_full_index === true,
+ core_goal: core,
+ evidence: shown, evidence_count: evidence.length, distinct_observation_count: candidates.length,
+ evidence_truncated: omitted.length > 0,
+ coverage_ledger: ledger, coverage_count: coverage.size, coverage_truncated: coverage.size > ledger.length,
+ uncovered_frontier: frontier,
+ };
+ if (omitted.length) context.omitted_evidence = {
+ distinct_observation_count: omitted.length,
+ newest_at: omitted[0].row.generated_at, oldest_at: omitted.at(-1)!.first,
+ ...(agent ? {read_action: `${prefix} --format json history --goal-id ${goal} --agent-id ${agent} --limit ${input.length}`} : {}),
+ };
+ if (obligation) context.delivery_receipt = {schema_version: "replan_context_delivery_receipt_v0",
+ context_id: contextId, obligation_id: obligationId, status: "delivered",
+ delivered_by: "quota_host_projection"};
+ return context;
+}
diff --git a/loopx/control_plane/work_items/replan_context_codec.py b/loopx/control_plane/work_items/replan_context_codec.py
new file mode 100644
index 0000000000..b538bd61f2
--- /dev/null
+++ b/loopx/control_plane/work_items/replan_context_codec.py
@@ -0,0 +1,136 @@
+"""Public-safe history codec for the TypeScript-owned replan evidence view."""
+from __future__ import annotations
+
+from collections.abc import Iterable, Mapping
+from pathlib import Path
+import shlex
+from typing import Any
+
+from ..effect_runtime import EffectRuntimeRejected, effect_runtime_result
+from ..runtime.public_safety import public_safe_compact_text
+from ..runtime.time import parse_timestamp
+from ..runtime.agent_evidence_history import goal_history_runs
+
+
+def replan_history_from_status(status: Mapping[str, Any], goal_id: str) -> list[dict[str, Any]]:
+ history = status.get("run_history")
+ if not isinstance(history, Mapping):
+ raise ValueError("replan context requires the run_history source")
+ snapshot = goal_history_runs(history, goal_id)
+ # A real host route supplies both authorities. Pure snapshot callers stay
+ # pure; live replan/handoff reads are independent of the display limit.
+ if status.get("registry") and status.get("runtime_root"):
+ from ...history import load_index_snapshot, validate_goal_id_path_segment
+
+ goal = validate_goal_id_path_segment(goal_id)
+ path = Path(str(status["runtime_root"])) / "goals" / goal / "runs" / "index.jsonl"
+ return load_index_snapshot(path, include_artifact_status=False).records
+ return snapshot
+
+
+def replan_evidence_rows(runs: Iterable[Mapping[str, Any]], *, goal_id: str, agent_id: str | None) -> list[dict[str, Any]]:
+ from .progress_observation import normalize_progress_observation
+
+ rows = []
+ for run in runs:
+ if not isinstance(run, Mapping):
+ raise ValueError("replan history row must be an object")
+ # Legacy compact records may omit the enclosing Goal id. Agent attribution
+ # is never guessed: another or unattributed lane cannot supply scoped proof.
+ run_goal = run.get("goal_id", goal_id)
+ run_agent = run.get("agent_id")
+ if run_goal != goal_id or (agent_id and run_agent != agent_id):
+ continue
+ timestamp = parse_timestamp(run.get("generated_at"))
+ if timestamp is None:
+ raise ValueError("replan evidence has an invalid generated_at timestamp")
+ progress = run.get("progress_observation")
+ if "progress_observation" in run:
+ if not isinstance(progress, Mapping):
+ raise ValueError("replan evidence progress_observation must be an object")
+ if progress.get("schema_version") != "typed_progress_observation_v0":
+ raise ValueError("unsupported replan evidence progress_observation schema_version")
+ progress = normalize_progress_observation(progress)
+ row: dict[str, Any] = {
+ "goal_id": run_goal,
+ "agent_id": run_agent,
+ "generated_at": str(run["generated_at"]),
+ "observed_at": timestamp.timestamp(),
+ "progress_observation": progress,
+ }
+ for key in ("classification", "delivery_outcome", "health_check", "recommended_action", "todo_id"):
+ text = public_safe_compact_text(run.get(key), limit=240)
+ if text:
+ row[key] = text
+ rows.append(row)
+ return rows
+
+
+def _goal_facts(status: Mapping[str, Any] | None, goal_id: str, acceptance: Mapping[str, Any] | None) -> dict[str, Any]:
+ """Read existing Goal sources; source precedence belongs to TypeScript."""
+ def goal_text(value: Any) -> str | None:
+ if value is not None and not isinstance(value, str):
+ raise ValueError("Goal objective must be text")
+ # Preserve the complete core objective, including trailing constraints.
+ return public_safe_compact_text(value, limit=max(1, len(value or "")))
+
+ facts: dict[str, Any] = {}
+ if status and status.get("registry"):
+ from ...agent_registry import load_goal_from_registry
+ from ...materials import goal_state_path
+ from ..goals.active_state_metadata import active_state_section_text, split_state_frontmatter
+
+ goal = load_goal_from_registry(Path(str(status["registry"])), goal_id)
+ if goal:
+ facts["registry_objective"] = goal_text(goal.get("objective"))
+ path = goal_state_path(goal)
+ if path and path.exists():
+ state = path.read_text(encoding="utf-8")
+ metadata, _ = split_state_frontmatter(state)
+ objective = active_state_section_text(state, "Objective") or metadata.get("objective")
+ facts["active_state_objective"] = goal_text(objective)
+ if acceptance and acceptance.get("enabled") is True:
+ # This is the acceptance owner's public projection, including its scope.
+ # It must never be mistaken for the whole Goal's original objective.
+ facts["acceptance_contract"] = {key: acceptance[key] for key in (
+ "objective", "non_goals", "criteria", "scope", "revision", "digest", "status",
+ ) if key in acceptance}
+ return facts
+
+
+def project_replan_context(
+ *, goal_id: str, agent_id: str | None, runs: Iterable[Mapping[str, Any]],
+ obligation: Mapping[str, Any] | None = None, evidence_ref: str | None = None,
+ source_status: Mapping[str, Any] | None = None,
+ goal_acceptance_contract: Mapping[str, Any] | None = None,
+) -> dict[str, Any]:
+ from .replan_history_codec import project_replan_request
+
+ prefix = ["loopx"]
+ if source_status is not None:
+ runs = replan_history_from_status(source_status, goal_id)
+ for field, option in (("registry", "--registry"), ("runtime_root", "--runtime-root")):
+ if source_status.get(field):
+ prefix.extend([option, str(source_status[field])])
+ try:
+ return project_replan_request({
+ "operation": "resolve" if evidence_ref is not None else "project",
+ "goal_id": goal_id, "agent_id": agent_id,
+ "obligation": dict(obligation) if obligation is not None else None,
+ "rows": replan_evidence_rows(runs, goal_id=goal_id, agent_id=agent_id),
+ "evidence_ref": evidence_ref,
+ "read_prefix": shlex.join(prefix),
+ "goal_facts": _goal_facts(source_status, goal_id, goal_acceptance_contract) if evidence_ref is None else {},
+ "from_full_index": bool(source_status and source_status.get("registry") and source_status.get("runtime_root")),
+ }, method="work_item.replan_context")
+ except EffectRuntimeRejected as exc:
+ raise ValueError(str(exc)) from None
+
+
+def validate_replan_context(context: Any) -> dict[str, Any]:
+ try:
+ return effect_runtime_result("work_item.replan_context.project", {
+ "operation": "validate", "context": context,
+ })
+ except EffectRuntimeRejected as exc:
+ raise ValueError(str(exc)) from None
diff --git a/loopx/control_plane/work_items/replan_history_codec.py b/loopx/control_plane/work_items/replan_history_codec.py
index 7443a00508..b013153622 100644
--- a/loopx/control_plane/work_items/replan_history_codec.py
+++ b/loopx/control_plane/work_items/replan_history_codec.py
@@ -142,7 +142,7 @@ def project_replan_history(
"todos": _todo_facts(agent_todos, include_resume=needs_resume),
}
try:
- result = _project(params)
+ result = project_replan_request(params)
except EffectRuntimeRejected as exc:
raise ValueError(str(exc)) from None
if not isinstance(result, dict) or result.get("schema_version") != "replan_history_result_v0":
@@ -153,12 +153,12 @@ def project_replan_history(
return trigger
-def _project(params: dict[str, Any]) -> Any:
+def project_replan_request(params: dict[str, Any], *, method: str = "work_item.replan_history") -> Any:
# Same serialization as the bridge. Reserve envelope overhead; the limit is
# a wire budget, not permission to drop older evidence or duplicate TS policy.
encoded = json.dumps(params, separators=(",", ":")).encode()
if len(encoded) <= MAX_REQUEST_BYTES // 2:
- return effect_runtime_result("work_item.replan_history.project", params)
+ return effect_runtime_result(f"{method}.project", params)
# Local same-UID runtime only. The private directory survives runtime retries
# and is removed on success/rejection. This snapshot is never durable state.
with TemporaryDirectory(prefix="loopx-replan-history-") as directory:
@@ -166,7 +166,7 @@ def _project(params: dict[str, Any]) -> Any:
with path.open("xb") as handle:
path.chmod(0o600)
handle.write(encoded)
- return effect_runtime_result("work_item.replan_history.project_snapshot", {
+ return effect_runtime_result(f"{method}.project_snapshot", {
"schema_version": "replan_history_snapshot_v0",
"path": str(path), "byte_count": len(encoded),
"sha256": hashlib.sha256(encoded).hexdigest(),
diff --git a/loopx/control_plane/work_items/replan_history_snapshot.ts b/loopx/control_plane/work_items/replan_history_snapshot.ts
index e17d58900d..7515174b5c 100644
--- a/loopx/control_plane/work_items/replan_history_snapshot.ts
+++ b/loopx/control_plane/work_items/replan_history_snapshot.ts
@@ -12,6 +12,11 @@ import { projectReplanHistory } from "./replan_history.ts";
import { BARE_SHA256_PATTERN } from "../content_digest.ts";
export async function projectReplanHistorySnapshot(value: unknown): Promise {
+ return projectReplanHistory(await readReplanSnapshot(value));
+}
+
+/** Shared local transport; the requesting typed owner still interprets the facts. */
+export async function readReplanSnapshot(value: unknown): Promise {
const request = requireJsonObject(value, "replan history snapshot");
requireStringLiteral(request.schema_version, ["replan_history_snapshot_v0"], "snapshot schema");
const path = requireNonEmptyString(request.path, "snapshot path");
@@ -46,5 +51,5 @@ export async function projectReplanHistorySnapshot(value: unknown): Promise None:
},
{
"command": "loopx review-packet --goal-id ",
- "purpose": "Render a handoff or review packet with any required evidence-log reads.",
+ "purpose": "Render a handoff or review packet with host-projected replan context.",
},
{
"command": "loopx goal-channel --help",
@@ -110,10 +110,6 @@ def register_command_reference(subparsers: argparse._SubParsersAction) -> None:
"command": "loopx goal-lifecycle --help",
"purpose": "Preview, stop, or resume a Goal without deleting its history, todos, or evidence.",
},
- {
- "command": "loopx evidence-log --goal-id --agent-id --thin",
- "purpose": "Read the current agent's thin public-safe ledger before replan or handoff.",
- },
{
"command": "loopx agent-capabilities --help",
"purpose": "Inspect or correct a registered Agent's observed runtime capabilities.",
@@ -484,8 +480,6 @@ def render_concise_help(program: str = "loopx") -> str:
"Daily operator commands:",
" loopx status Show current goals, gates, and next action.",
" loopx diagnose --goal-id ID Build a compact evidence packet.",
- " loopx evidence-log --goal-id ID --agent-id AGENT --thin",
- " Read this agent's thin ledger before replan.",
" loopx todo --help Add, claim, complete, update, or archive todos.",
" loopx task-lease --help Manage a hard per-todo lease.",
" loopx coordination-shadow --help",
diff --git a/loopx/history.py b/loopx/history.py
index 925329b283..0adc982f59 100644
--- a/loopx/history.py
+++ b/loopx/history.py
@@ -330,6 +330,7 @@ def collect_history(
include_runtime_goals: bool = True,
activation_state_filter: GoalActivationState | str | None = None,
agent_lane_id: str | None = None,
+ scoped_agent_id: str | None = None,
registry: dict[str, Any] | None = None,
) -> dict[str, Any]:
from .capabilities.machine_configuration.builtins import (
@@ -395,6 +396,12 @@ def collect_history(
]
for run in runs:
run["goal_id"] = str(run.get("goal_id") or current_goal_id)
+ # Explicit history drill-down scopes the complete source before its
+ # requested cap. Status/quota retain their separate Goal-wide source
+ # and bounded lane decision window; quota accounting is always Goal-wide.
+ goal_runs = runs
+ if scoped_agent_id:
+ runs = [run for run in goal_runs if run.get("agent_id") == scoped_agent_id]
run_count += len(runs)
recent_runs = list(
islice(
@@ -409,7 +416,7 @@ def collect_history(
)
adapter = meta.get("adapter") if isinstance(meta.get("adapter"), dict) else {}
- quota = goal_quota_with_spend_ledger(meta, runs) if registry_member else None
+ quota = goal_quota_with_spend_ledger(meta, goal_runs) if registry_member else None
goal_record = {
"id": current_goal_id,
"activation_state": activation_state.value,
@@ -442,7 +449,7 @@ def collect_history(
"latest_runs": latest_runs_with_agent_context(
runs,
limit=limit,
- agent_lane_id=agent_lane_id,
+ agent_lane_id=None if scoped_agent_id else agent_lane_id,
),
"semantic_history": goal_semantic_history_from_runs(runs),
}
diff --git a/loopx/kunluncode_goal_mode/control_plane.py b/loopx/kunluncode_goal_mode/control_plane.py
index 6bdf21f560..a61cae4d82 100644
--- a/loopx/kunluncode_goal_mode/control_plane.py
+++ b/loopx/kunluncode_goal_mode/control_plane.py
@@ -139,22 +139,26 @@ def claim(self, todo_id: str) -> dict[str, Any]:
)
def evidence_since(self, since: str, *, todo_id: str) -> dict[str, Any]:
- return self._run_json(
- [
- "evidence-log",
- "--goal-id",
- self.goal_id,
- "--agent-id",
- self.agent_id,
- "--todo-id",
- todo_id,
- "--thin",
- "--since",
- since,
- "--limit",
- "128",
- ]
- )
+ # Recovery reads its owning persisted event source directly. An agent-facing
+ # diagnostic command is not a dependency of native writeback reconciliation.
+ from loopx.history import load_registry
+ from loopx.paths import resolve_runtime_root
+ from loopx.rollout_event_log import load_rollout_events, rollout_event_log_path
+ from loopx.control_plane.runtime.agent_evidence_history import build_agent_scoped_evidence_log
+
+ try:
+ registry_path = Path(self.registry).expanduser()
+ if not registry_path.is_absolute():
+ registry_path = self.project / registry_path
+ registry = load_registry(registry_path)
+ runtime_root = resolve_runtime_root(registry, None, registry_path=registry_path)
+ events = load_rollout_events(rollout_event_log_path(runtime_root, self.goal_id), limit=400)
+ return build_agent_scoped_evidence_log(
+ goal_id=self.goal_id, agent_id=self.agent_id, todo_id=todo_id,
+ since=since, rollout_events=events, history_runs=[], limit=128,
+ )
+ except (OSError, ValueError, TypeError) as exc:
+ raise KunlunNativeGoalRuntimeError(f"writeback evidence unavailable: {exc}") from exc
def record_verified_delivery(self, *, mode: str, todo_id: str) -> dict[str, Any]:
workspace_arguments = (
diff --git a/loopx/kunluncode_goal_mode/runtime.py b/loopx/kunluncode_goal_mode/runtime.py
index 6033124c99..7254d1812b 100644
--- a/loopx/kunluncode_goal_mode/runtime.py
+++ b/loopx/kunluncode_goal_mode/runtime.py
@@ -325,7 +325,9 @@ def _reconcile_writeback_evidence(
todo_id = str(state.get("todo_id") or "")
goal_admission.require_current()
payload = control.evidence_since(since, todo_id=todo_id)
- ledger = payload.get("ledger") if isinstance(payload.get("ledger"), list) else []
+ if payload.get("schema_version") != "agent_scoped_evidence_log_v0" or not isinstance(payload.get("ledger"), list):
+ raise KunlunNativeGoalRuntimeError("invalid internal writeback evidence schema")
+ ledger = payload["ledger"]
writeback = state["writeback"]
classification = VERIFIED_CLASSIFICATIONS[str(native.get("mode") or "goal-pro")]
for event in ledger:
diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json
index 27d95ee802..31e2b3351d 100644
--- a/loopx/semantics/project_registry_io_manifest_v1.json
+++ b/loopx/semantics/project_registry_io_manifest_v1.json
@@ -495,7 +495,7 @@
},
{
"site": "loopx/cli.py::.main::codec_read:load_project_registry#1",
- "line": 820,
+ "line": 817,
"column": 17,
"kind": "codec_read",
"api": "load_project_registry",
@@ -621,22 +621,6 @@
"api": "load_registry",
"classification": "codec_api"
},
- {
- "site": "loopx/cli_commands/evidence_log.py::.handle_evidence_log_command::codec_read:load_registry#1",
- "line": 190,
- "column": 20,
- "kind": "codec_read",
- "api": "load_registry",
- "classification": "codec_api"
- },
- {
- "site": "loopx/cli_commands/evidence_log.py::.handle_evidence_log_command::codec_read:load_registry#2",
- "line": 246,
- "column": 24,
- "kind": "codec_read",
- "api": "load_registry",
- "classification": "codec_api"
- },
{
"site": "loopx/cli_commands/explore.py::.handle_explore_command::codec_read:load_registry#1",
"line": 466,
@@ -671,7 +655,7 @@
},
{
"site": "loopx/cli_commands/history.py::.handle_history_command::codec_read:load_registry#1",
- "line": 85,
+ "line": 93,
"column": 24,
"kind": "codec_read",
"api": "load_registry",
@@ -679,7 +663,7 @@
},
{
"site": "loopx/cli_commands/history.py::.handle_history_command::codec_read:load_registry#2",
- "line": 120,
+ "line": 113,
"column": 24,
"kind": "codec_read",
"api": "load_registry",
@@ -687,7 +671,7 @@
},
{
"site": "loopx/cli_commands/history.py::.handle_history_command::codec_read:load_registry#3",
- "line": 146,
+ "line": 148,
"column": 24,
"kind": "codec_read",
"api": "load_registry",
@@ -695,7 +679,7 @@
},
{
"site": "loopx/cli_commands/history.py::.handle_history_command::codec_read:load_registry#4",
- "line": 190,
+ "line": 174,
"column": 24,
"kind": "codec_read",
"api": "load_registry",
@@ -703,7 +687,15 @@
},
{
"site": "loopx/cli_commands/history.py::.handle_history_command::codec_read:load_registry#5",
- "line": 208,
+ "line": 218,
+ "column": 24,
+ "kind": "codec_read",
+ "api": "load_registry",
+ "classification": "codec_api"
+ },
+ {
+ "site": "loopx/cli_commands/history.py::.handle_history_command::codec_read:load_registry#6",
+ "line": 236,
"column": 20,
"kind": "codec_read",
"api": "load_registry",
@@ -1919,7 +1911,7 @@
},
{
"site": "loopx/history.py::.collect_history::codec_read:load_registry#1",
- "line": 342,
+ "line": 343,
"column": 20,
"kind": "codec_read",
"api": "load_registry",
@@ -1927,7 +1919,7 @@
},
{
"site": "loopx/history.py::.inspect_index_duplicates::codec_read:load_registry#1",
- "line": 598,
+ "line": 605,
"column": 16,
"kind": "codec_read",
"api": "load_registry",
@@ -1935,7 +1927,7 @@
},
{
"site": "loopx/history.py::.rebuild_index_artifact_collisions::codec_read:load_registry#1",
- "line": 812,
+ "line": 819,
"column": 16,
"kind": "codec_read",
"api": "load_registry",
@@ -1943,7 +1935,7 @@
},
{
"site": "loopx/history.py::.repair_index_duplicates::codec_read:load_registry#1",
- "line": 702,
+ "line": 709,
"column": 16,
"kind": "codec_read",
"api": "load_registry",
@@ -1965,6 +1957,14 @@
"api": "load_project_registry",
"classification": "codec_api"
},
+ {
+ "site": "loopx/kunluncode_goal_mode/control_plane.py::.LoopXControlPlane.evidence_since::codec_read:load_registry#1",
+ "line": 153,
+ "column": 24,
+ "kind": "codec_read",
+ "api": "load_registry",
+ "classification": "codec_api"
+ },
{
"site": "loopx/operator_gate.py::.record_operator_gate::codec_read:load_registry#1",
"line": 316,
diff --git a/loopx/status.py b/loopx/status.py
index ea6912b0d4..8eda1dce40 100644
--- a/loopx/status.py
+++ b/loopx/status.py
@@ -207,7 +207,7 @@
from .control_plane.agents.management_projection import (
build_agent_management_projection as _build_agent_management_projection_read_model,
)
-from .control_plane.runtime.agent_scoped_evidence_log import (
+from .control_plane.runtime.agent_evidence_history import (
MAX_PROJECTED_READ_RECEIPTS,
project_evidence_log_read_receipts,
)
diff --git a/man/loopx.1 b/man/loopx.1
index daab1ab594..c5e7f6bfa0 100644
--- a/man/loopx.1
+++ b/man/loopx.1
@@ -73,7 +73,7 @@ Open Goal Studio, ask the read\-only Agent for a bounded Todo, and approve its p
Build a compact evidence packet when behavior is surprising.
.TP
\fBloopx review\-packet \-\-goal\-id \fR
-Render a handoff or review packet with any required evidence\-log reads.
+Render a handoff or review packet with host\-projected replan context.
.TP
\fBloopx goal\-channel \-\-help\fR
Set up, inspect, sync, or notify the provider channel bound to one goal.
@@ -87,9 +87,6 @@ Read source\-bound Manager context and record the receiving Agent's decision.
\fBloopx goal\-lifecycle \-\-help\fR
Preview, stop, or resume a Goal without deleting its history, todos, or evidence.
.TP
-\fBloopx evidence\-log \-\-goal\-id \-\-agent\-id \-\-thin\fR
-Read the current agent's thin public\-safe ledger before replan or handoff.
-.TP
\fBloopx agent\-capabilities \-\-help\fR
Inspect or correct a registered Agent's observed runtime capabilities.
.TP
diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md
index 70a9deb29c..e13d27ca46 100644
--- a/skills/loopx-self-repair/references/repair-patterns.md
+++ b/skills/loopx-self-repair/references/repair-patterns.md
@@ -103,7 +103,7 @@ teaches a reusable control-plane lesson.
| `nonblocking_user_action_execution_suppression` | Quota selects a required non-delivery obligation such as a ready-successor replan while an unrelated `user_action` is open, but the final interaction contract becomes `notify,wait` and scheduler backoff blocks the Goal. | Selected Todo, execution obligation `must_attempt_work` and `delivery_allowed`, user Todo task class, final interaction channels and CLI actions, and scheduler disposition. | Empty-agent-lane detection looked only at ordinary executable Todos. It treated a selected control-plane obligation with `delivery_allowed=false` as absence of agent work, promoted the notice to a blocking user action, and discarded the selected command. | Let an open `user_action` own the empty lane only when no selected execution obligation must be attempted. Preserve the obligation mode, selected lifecycle command, no-delivery policy, and active scheduler cadence while projecting the user item as a non-blocking notice. Keep explicit `user_gate` precedence and cover successor replan plus other non-delivery obligations end to end. |
| `repeat_vision_wait_ack_gap` | Two exact blocked-successor stalls correctly trigger replan, but a maintenance-only continuation appears to close the obligation while a `repeat_until_closed` vision still has no evidence-linked path outcome; later heartbeats repeat the same wait cycle. | active agent vision advancement policy, blocked-successor frontier identity, current obligation id, host-projected coverage ledger, typed progress observation, current runnable successor ids, and semantic receipt. | Replan validity considered a caller-supplied delta label instead of the obligation's typed outcomes and current state. | Require an accepted semantic delta bound to the exact current obligation. A vision-derived duty accepts a fresh evidence-linked vision path, new evidence-backed blocker, or coverage-backed terminal; a successor id is accepted only when current state proves it runnable, and maintenance-only continuation never closes the vision duty. Cover repeat/as-needed routing and stale-obligation identities. |
| `future_monitor_heartbeat_liveness_overreach` | A valid future-scheduled monitor repeatedly enters `frontier_exhausted_monitor_lane` or `dead_monitor_repeat` even though it is not due and no real monitor execution occurred. | generated heartbeat guard command, trigger `current_time_iso`, quota `heartbeat_receipt`, monitor Todo `next_due_at`, run-history `quota_monitor_poll` rows, concrete Todo/target identity, and monitor mode. | Future wait state was collapsed into empty-frontier exhaustion, while distinct heartbeat turn ids were treated as monitor executions and lowered the dead-monitor threshold. | Add typed `future_monitor_wait` and `due_monitor_execution` frontier rules. Keep future monitors quiet/no-spend until due; count `dead_monitor_repeat` only from unchanged due/external executions carrying a concrete Todo or target identity. Heartbeat turn ids remain idempotency/liveness evidence and never reduce the real execution threshold. |
-| `replan_context_delivery_gap` | A replan obligation is visible, but a weak protocol-following model ignores or mis-executes a manual evidence-log read ritual, then spends turns debugging ACK/read order instead of choosing a new direction. | current obligation id, host-projected `replan_context_v0`, compact coverage ledger, delivery receipt, typed action packet, selected model tool action, semantic writeback receipt, and successor turn boundary. | Evidence access was modeled as an agent CLI obligation even though the control plane already owned the ledger; read execution and semantic use were conflated. A separate Todo-add then refresh ACK also let the model execute the successor in the replan turn. | Have the host prefetch and project bounded evidence coverage into quota. Treat the delivery receipt as context delivery only; require the model's real next action to produce an accepted typed semantic delta bound to the current obligation. When the delta is a runnable successor, bind the exact obligation on the Todo write itself and return `end_current_heartbeat`; do not require a second ACK or run the successor in that turn. Keep evidence-log as the durable chronology and optional drill-down, not as a closeout gate. Cover real quota→model action→Todo/refresh CLI behavior plus missing/wrong obligation identity. |
+| `replan_context_delivery_gap` | A replan obligation is visible, but a weak protocol-following model ignores or mis-executes a manual evidence-log read ritual, then spends turns debugging ACK/read order instead of choosing a new direction. | current obligation id, host-projected `replan_context_v0`, compact coverage ledger, delivery receipt, typed action packet, selected model tool action, semantic writeback receipt, and successor turn boundary. | Evidence access was modeled as an agent CLI obligation even though the control plane already owned the ledger; read execution and semantic use were conflated. A separate Todo-add then refresh ACK also let the model execute the successor in the replan turn. | Have the host prefetch and project bounded evidence coverage into quota. Treat the delivery receipt as context delivery only; require the model's real next action to produce an accepted typed semantic delta bound to the current obligation. When the delta is a runnable successor, bind the exact obligation on the Todo write itself and return `end_current_heartbeat`; do not require a second ACK or run the successor in that turn. Remove the standalone evidence command and generated read rituals. Reuse the typed replan context for the core Goal, dense scoped decision evidence, repetition reduction and exact history-reference reads. Keep replan selection independent of the routine display window, and validate novelty against complete available history even when the readable context is truncated; reject malformed schemas and unavailable references instead of treating them as empty evidence. Preserve the durable history and historical receipt decoder, not a compatibility command. Cover real quota→model action→Todo/refresh CLI behavior plus missing/wrong obligation identity. |
| `replan_settlement_binding_projection_gap` | A replan packet and selected Todo are both visible, so an agent combines `--todo-id` with `--replan-obligation-id`, alternates between them after rejection, or performs the Todo-bound refresh but omits the matching spend. | Original quota receipt identity, selected Todo, semantic obligation id, interaction settlement plan, projected refresh/spend commands, and accepted semantic ACK. | The semantic obligation target and causal settlement identity were adjacent but unlabeled, while one boolean conflated direct obligation binding with whether a full settlement chain was required. | Keep exactly one causal settlement binding. For a turn-scoped executable plan, project a domain-local typed replan settlement contract that labels the selected Todo or Todo-less obligation as the binding and keeps the replan id as a separate semantic target when necessary. Todo-bound and directly bound replan writes both project material refresh followed by spend; never combine their CLI flags. Keep unscoped diagnostic reads compact and non-executable. Cover TS projection, hot-path envelope, real Todo-bound/Todo-less CLI chains, output budget, and in-flight delivery continuity. |
| `native_goal_todoless_replan_reentry_gap` | A native visible Goal completes all advancement Todos, semantic closeout derives a Todo-less autonomous replan obligation, and the next unbound quota read projects a bare refresh with no Turn identity or spend chain. | Native Goal runtime profile, selected Todo absence, replan obligation id, interaction CLI actions before and after Turn creation, and settlement identity. | Native-Goal Turn re-entry recognized only selected Todos as actionable settlement bindings and ignored a valid Todo-less replan obligation. | Treat either a selected Todo or a valid replan obligation as a native-Goal settlement binding. When no Turn or settlement plan exists, project only the exact `quota should-run ... --begin-turn` re-entry. After execution, require the obligation-bound refresh and spend chain; keep generic unscoped diagnostic behavior unchanged. |
| `replan_meta_successor_churn` | An autonomous replan repeatedly adds and completes obligation-bound advancement Todos whose only action is to replan, re-arm, wait for the host, or report no new direction; Todo count grows while executable targets and evidence do not. | current obligation and acceptance gaps, generated replan action packet, bound Todo `action_kind`/`target_key`/Explore refs, semantic ACK, Todo lifecycle, and compact progress fingerprints. | A planning instruction was compiled into a runnable successor command, while Todo authoring and ACK validation treated obligation identity alone as proof of executable semantic change. | Generate a successor command only from a concrete bounded frontier with a typed action and stable target. Require the same semantic binding at Todo write and ACK time; generic vision/frontier replans must resolve through a typed progress or coverage-backed terminal outcome instead of creating a meta Todo. Cover packet generation, write rejection, ACK rejection, and a real quota→model→CLI successor re-entry. |
diff --git a/tests/control_plane/test_agent_scoped_evidence_log.py b/tests/control_plane/test_agent_evidence_history.py
similarity index 98%
rename from tests/control_plane/test_agent_scoped_evidence_log.py
rename to tests/control_plane/test_agent_evidence_history.py
index 1007b4dbc4..56c6bfccf4 100644
--- a/tests/control_plane/test_agent_scoped_evidence_log.py
+++ b/tests/control_plane/test_agent_evidence_history.py
@@ -4,7 +4,7 @@
import pytest
-from loopx.control_plane.runtime.agent_scoped_evidence_log import (
+from loopx.control_plane.runtime.agent_evidence_history import (
build_agent_scoped_evidence_log,
project_evidence_log_read_receipts,
)
diff --git a/tests/control_plane/test_cli_output_budget.py b/tests/control_plane/test_cli_output_budget.py
index a2f69761d7..513c9fc6c8 100644
--- a/tests/control_plane/test_cli_output_budget.py
+++ b/tests/control_plane/test_cli_output_budget.py
@@ -466,21 +466,6 @@ def _surface_commands(
+ ["todo", "list", "--goal-id", GOAL_ID, "--agent-id", AGENT_IDS[0]],
"history_limited": common
+ ["history", "--goal-id", GOAL_ID, "--limit", "5"],
- "evidence_log_thin": common
- + [
- "evidence-log",
- "--goal-id",
- GOAL_ID,
- "--agent-id",
- AGENT_IDS[0],
- "--limit",
- "5",
- "--history-limit",
- "10",
- "--rollout-limit",
- "20",
- "--thin",
- ],
}
@@ -775,7 +760,6 @@ def test_manifest_covers_the_declared_agent_facing_surface_set() -> None:
"heartbeat_prompt_thin",
"todo_list",
"history_limited",
- "evidence_log_thin",
}
manifest = public_manifest()
assert set(CLI_OUTPUT_BUDGET_BY_ID) == expected
@@ -1444,7 +1428,17 @@ def _assert_collection_growth_and_bootstrap_duplication(
- small[spec.surface_id]["json"]["chars"]
)
fixed_semantic_growth = spec.max_json_fixed_semantic_growth_chars
- if fixed_semantic_growth:
+ if fixed_semantic_growth and spec.surface_id == "quota_should_run":
+ context = crowded[spec.surface_id]["json"]["payload"]["autonomous_replan_obligation"]["replan_context"]
+ assert 0 < len(context["evidence"]) <= 24
+ assert len(context["coverage_ledger"]) <= 24
+ assert context["from_full_index"] is True
+ assert len(context["evidence"]) == SCENARIOS[1].run_count
+ assert all("--evidence-ref" in row["read_action"] for row in context["evidence"])
+ elif fixed_semantic_growth and spec.surface_id == "diagnose":
+ context = crowded[spec.surface_id]["json"]["payload"]["selected"]["projection_warnings"]["autonomous_replan_obligation"]["replan_context"]
+ assert 0 < len(context["evidence"]) <= 24
+ elif fixed_semantic_growth:
assert spec.surface_id == "loopx_turn_plan"
small_packet = small[spec.surface_id]["json"]["payload"][
"turn_envelope"
diff --git a/tests/control_plane/test_cli_output_differential.py b/tests/control_plane/test_cli_output_differential.py
index 1174830433..d97b11bbce 100644
--- a/tests/control_plane/test_cli_output_differential.py
+++ b/tests/control_plane/test_cli_output_differential.py
@@ -1151,3 +1151,29 @@ def test_malformed_command_never_grants_route_growth(command, render) -> None:
def test_json_escaped_paths_and_duplicate_arguments_are_counted_once() -> None:
command = """loopx --format json --registry '/tmp/a \"quoted\" path' --registry /tmp/final --runtime-root '/tmp/root path' turn plan"""
assert command_route_counts(json.dumps({"command": command})) == {"registry": 1, "runtime_root": 1}
+
+
+def test_dense_replan_growth_is_one_time_and_keeps_semantic_checks():
+ paths = ['$.autonomous_replan_obligation.replan_context.' + key
+ for key in ('core_goal', 'evidence', 'coverage_ledger')]
+ base = _row(row_id='surface/quota_should_run/crowded/json', surface_id='quota_should_run', scenario='crowded')
+ candidate = {**base, 'json_shape_paths': [*base['json_shape_paths'], *paths],
+ 'chars': base['chars'] + 6600, 'lines': base['lines'] + 88}
+ assert compare_cli_output_receipts(_receipt(base), _receipt(candidate))['ok']
+ assert not compare_cli_output_receipts(_receipt(candidate), _receipt({**candidate, 'chars': candidate['chars'] + 6600}))['ok']
+ assert not compare_cli_output_receipts(_receipt(base), _receipt({**candidate, 'chars': base['chars'] + 7001}))['ok']
+ assert not compare_cli_output_receipts(_receipt(base), _receipt({**candidate, 'action_signature_sha256': 'changed'}))['ok']
+ ordinary_base = {**base, 'row_id': 'surface/quota_should_run/small/json'}
+ ordinary_head = {**candidate, 'row_id': ordinary_base['row_id']}
+ assert not compare_cli_output_receipts(_receipt(ordinary_base), _receipt(ordinary_head))['ok']
+
+
+def test_only_retired_evidence_command_with_real_replacement_is_allowed():
+ base = _row(row_id='surface/evidence_log_thin/small/json', surface_id='evidence_log_thin')
+ replacement = _row(row_id='variant/review_packet_full/small/json', json_shape_paths=[
+ '$.replan_context.core_goal', '$.replan_context.evidence', '$.replan_context.coverage_ledger'])
+ result = compare_cli_output_receipts(_receipt(base), _receipt(replacement))
+ assert result['ok'] and result['review_required']
+ assert not compare_cli_output_receipts(_receipt(base), _receipt())['ok']
+ assert not compare_cli_output_receipts(_receipt(base), _receipt({**replacement, 'json_shape_paths': []}))['ok']
+ assert not compare_cli_output_receipts(_receipt({**base, 'row_id': 'surface/status/small/json'}), _receipt(replacement))['ok']
diff --git a/tests/control_plane/test_goal_scoped_log_paths.py b/tests/control_plane/test_goal_scoped_log_paths.py
index 059437bc7c..493db5cd2a 100644
--- a/tests/control_plane/test_goal_scoped_log_paths.py
+++ b/tests/control_plane/test_goal_scoped_log_paths.py
@@ -63,7 +63,7 @@ def test_supervisor_event_builder_rejects_unsafe_goal_id() -> None:
)
-def test_evidence_log_cli_rejects_unsafe_goal_without_writing_outside_root(
+def test_history_evidence_cli_rejects_unsafe_goal_without_writing_outside_root(
tmp_path: Path,
capsys: pytest.CaptureFixture[str],
) -> None:
@@ -86,7 +86,9 @@ def test_evidence_log_cli_rejects_unsafe_goal_without_writing_outside_root(
str(runtime_root),
"--format",
"json",
- "evidence-log",
+ "history",
+ "--evidence-ref",
+ "replan-evidence-missing",
"--goal-id",
"../escape",
"--agent-id",
diff --git a/tests/control_plane/test_goal_vision_blocked_successor.py b/tests/control_plane/test_goal_vision_blocked_successor.py
index d4cd204a9b..7b9f2ba3f5 100644
--- a/tests/control_plane/test_goal_vision_blocked_successor.py
+++ b/tests/control_plane/test_goal_vision_blocked_successor.py
@@ -14,9 +14,6 @@
build_goal_frontier_projection_context_from_status,
select_autonomous_replan_obligation,
)
-from loopx.control_plane.runtime.agent_scoped_evidence_log import (
- build_agent_scoped_evidence_log_command,
-)
from loopx.control_plane.quota.monitor_poll import build_quota_monitor_poll_event
from loopx.control_plane.scheduler.execution_context import (
GENERIC_CLI_OUTER_CONTROLLER_SCHEDULER_CONTEXT,
@@ -224,10 +221,11 @@ def _quota(payload: dict) -> dict:
"agent_id": AGENT_ID,
"status": "completed",
"recorded_at": acked_at,
- "command": build_agent_scoped_evidence_log_command(
- goal_id=GOAL_ID,
- agent_id=AGENT_ID,
- required_read_id=obligation_id,
+ # Frozen persisted command from the retired interface, not executed.
+ "command": (
+ f"loopx --format json evidence-log --goal-id {GOAL_ID}"
+ f" --agent-id {AGENT_ID} --thin --limit 24"
+ + (f" --required-read-id {obligation_id}" if obligation_id else "")
),
"read_window": {"mode": "thin", "limit": 24},
**(
diff --git a/tests/control_plane/test_progress_observation.py b/tests/control_plane/test_progress_observation.py
index f5e7e65db1..b0a2c19121 100644
--- a/tests/control_plane/test_progress_observation.py
+++ b/tests/control_plane/test_progress_observation.py
@@ -478,7 +478,7 @@ def test_host_projects_evidence_context_and_minimal_action_packet() -> None:
enriched = {**obligation, "replan_context": context}
packet = build_replan_action_packet(enriched)
- assert context["evidence_source"] == "agent_scoped_evidence_log"
+ assert context["evidence_source"] == "compact_run_history"
assert context["delivery"] == "host_projected"
assert context["delivery_receipt"]["status"] == "delivered"
assert context["coverage_ledger"][0]["fingerprint"] == trigger[
diff --git a/tests/control_plane/test_replan_context_evidence.py b/tests/control_plane/test_replan_context_evidence.py
new file mode 100644
index 0000000000..91d5911596
--- /dev/null
+++ b/tests/control_plane/test_replan_context_evidence.py
@@ -0,0 +1,238 @@
+from __future__ import annotations
+
+import json
+import shlex
+import subprocess
+import sys
+from datetime import datetime, timedelta, timezone
+from pathlib import Path
+
+import pytest
+
+from loopx.control_plane.runtime.agent_evidence_history import goal_history_runs
+from loopx.control_plane.work_items.replan_context_codec import project_replan_context, replan_history_from_status
+
+
+def _run(**changes):
+ return {
+ "goal_id": "evidence-goal", "agent_id": "agent-a",
+ "generated_at": "2026-08-18T01:00:00Z",
+ "classification": "bounded_probe",
+ "health_check": "Negative result changes the next probe.",
+ **changes,
+ }
+
+
+def test_history_contract_errors_are_not_empty_results():
+ assert goal_history_runs({"goals": []}, "evidence-goal") == []
+ assert goal_history_runs({"goals": [{"id": "evidence-goal", "latest_runs": []}]}, "evidence-goal") == []
+ for bad in [{}, {"ok": False, "goals": []}, {"goals": None},
+ {"goals": [{"id": "evidence-goal"}]}, {"runs": [None]}]:
+ with pytest.raises(ValueError):
+ goal_history_runs(bad, "evidence-goal")
+ for status in [{}, {"run_history": None}, {"run_history": {"goals": [{"id": "evidence-goal"}]}}]:
+ with pytest.raises(ValueError):
+ replan_history_from_status(status, "evidence-goal")
+
+
+def test_projection_rejects_invalid_typed_records_and_redacts_prose():
+ for invalid in [None, {}, {"schema_version": "future"}]:
+ with pytest.raises(ValueError):
+ project_replan_context(goal_id="evidence-goal", agent_id="agent-a",
+ runs=[_run(progress_observation=invalid)])
+ context = project_replan_context(
+ goal_id="evidence-goal", agent_id="agent-a",
+ runs=[_run(health_check="access_key=private-value"),
+ _run(agent_id="agent-b", health_check="other Agent detail")],
+ )
+ assert len(context["evidence"]) == 1
+ assert "private-value" not in json.dumps(context)
+ assert "other Agent detail" not in json.dumps(context)
+
+
+def test_real_history_reads_the_projected_reference_and_rejects_stale_scope(tmp_path: Path):
+ runtime = tmp_path / "runtime"
+ registry = tmp_path / "registry.json"
+ registry.write_text(json.dumps({
+ "schema_version": 1, "common_runtime_root": str(runtime),
+ "goals": [{"id": "evidence-goal", "status": "active-read-only", "domain": "fixture"}],
+ }))
+ index = runtime / "goals/evidence-goal/runs/index.jsonl"
+ index.parent.mkdir(parents=True)
+ index.write_text(json.dumps(_run()) + "\n")
+ context = project_replan_context(goal_id="evidence-goal", agent_id="agent-a", runs=(),
+ source_status={"registry": str(registry), "runtime_root": str(runtime),
+ "run_history": {"goals": []}})
+ evidence = context["evidence"][0]
+ prefix = [sys.executable, "-m", "loopx.cli", "--registry", str(registry), "--format", "json"]
+ args = shlex.split(evidence["read_action"])[1:]
+ assert "--registry" in args and "--runtime-root" in args
+ result = subprocess.run([*prefix, *args], capture_output=True, text=True, check=False)
+ assert result.returncode == 0, result.stdout + result.stderr
+ assert json.loads(result.stdout)["evidence"]["health_check"] == _run()["health_check"]
+ for agent in ("agent-b", "missing-agent"):
+ rejected = subprocess.run([*prefix, *[agent if arg == "agent-a" else arg for arg in args]],
+ capture_output=True, text=True, check=False)
+ assert rejected.returncode == 1
+ assert "unavailable" in json.loads(rejected.stdout)["error"]
+ index.write_text("")
+ stale = subprocess.run([*prefix, *args], capture_output=True, text=True, check=False)
+ assert stale.returncode == 1
+ assert "unavailable" in json.loads(stale.stdout)["error"]
+ removed = subprocess.run([*prefix, "evidence-log"], capture_output=True, text=True, check=False)
+ assert removed.returncode == 2
+ assert "invalid choice" in removed.stderr
+ assert not (runtime / "goals/evidence-goal/rollout-event-log.jsonl").exists()
+
+
+def test_old_evidence_survives_display_limit_and_core_goal_uses_registered_state(tmp_path: Path):
+ runtime, project = tmp_path / "runtime", tmp_path / "project"
+ project.mkdir()
+ (project / "ACTIVE_GOAL_STATE.md").write_text("# Goal\n\n## Objective\n> Deliver the accepted outcome across the whole workload.\n\n## Agent Todo\n- Run another probe.\n")
+ registry = tmp_path / "registry.json"
+ registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{
+ "id": "evidence-goal", "repo": str(project), "state_file": "ACTIVE_GOAL_STATE.md",
+ "objective": "Original registered objective", "status": "active-read-only", "domain": "fixture",
+ }]}))
+ index = runtime / "goals/evidence-goal/runs/index.jsonl"
+ index.parent.mkdir(parents=True)
+ old = _run(progress_observation={"schema_version": "typed_progress_observation_v0",
+ "result_class": "advanced", "work_item_id": "old-work", "evidence_ids": ["old-best"]})
+ newer = [_run(generated_at=f"2026-08-19T01:{i//60:02}:{i%60:02}Z", progress_observation={
+ "schema_version": "typed_progress_observation_v0", "result_class": "unchanged",
+ "work_item_id": "new-work", "evidence_ids": ["repeat-proof"]}) for i in range(180)]
+ index.write_text("".join(json.dumps(row) + "\n" for row in [old, *newer]))
+ context = project_replan_context(goal_id="evidence-goal", agent_id="agent-a", runs=(),
+ source_status={"registry": str(registry), "runtime_root": str(runtime), "run_history": {
+ "goals": [{"id": "evidence-goal", "latest_runs": newer[-3:]}]}},
+ goal_acceptance_contract={"enabled": True, "scope": {"kind": "selected_work", "todo_ids": ["new-work"]},
+ "objective": "Validate only the new route"})
+ assert context["from_full_index"] is True
+ assert context["evidence_count"] == 181
+ assert len(context["evidence"]) == 2
+ assert context["core_goal"]["objective"] == "Deliver the accepted outcome across the whole workload."
+ assert context["core_goal"]["acceptance_contract"]["scope"]["kind"] == "selected_work"
+ command = context["evidence"][-1]["read_action"]
+ result = subprocess.run([sys.executable, "-m", "loopx.cli", *shlex.split(command)[1:]], capture_output=True, text=True)
+ assert result.returncode == 0, result.stdout + result.stderr
+ assert json.loads(result.stdout)["evidence"]["progress_observation"]["evidence_ids"] == ["old-best"]
+
+
+@pytest.mark.parametrize("peer_count", [0, 1, 3])
+def test_generated_omitted_read_recovers_full_agent_history_before_limit(tmp_path, peer_count):
+ runtime, registry = tmp_path / "runtime", tmp_path / "registry.json"
+ registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{
+ "id": "evidence-goal", "status": "active-read-only", "domain": "fixture",
+ }]}))
+ index = runtime / "goals/evidence-goal/runs/index.jsonl"
+ index.parent.mkdir(parents=True)
+ start = datetime(2026, 8, 18, tzinfo=timezone.utc)
+ own = []
+ for i in range(221):
+ own.append(_run(generated_at=(start + timedelta(seconds=i)).isoformat(),
+ progress_observation={"schema_version": "typed_progress_observation_v0",
+ "result_class": "advanced" if i == 0 else "unchanged",
+ "surface_id": f"route-{i}" if i < 41 else "repeated-route",
+ "evidence_ids": ["public-proof"]}))
+ peers = [_run(agent_id=f"peer-{peer}",
+ generated_at=(start + timedelta(days=peer + 1, seconds=i)).isoformat())
+ for peer in range(peer_count) for i in range(240)]
+ # Index presentation order is irrelevant; chronology and attribution own the read.
+ before = "".join(json.dumps(row) + "\n" for row in [*reversed(peers), *own]).encode()
+ index.write_bytes(before)
+ context = project_replan_context(goal_id="evidence-goal", agent_id="agent-a", runs=(),
+ source_status={"registry": str(registry), "runtime_root": str(runtime),
+ "run_history": {"goals": []}})
+ assert context["evidence_count"] == 221
+ assert context["omitted_evidence"]["distinct_observation_count"] == 18
+ args = shlex.split(context["omitted_evidence"]["read_action"])[1:]
+ prefix = [sys.executable, "-m", "loopx.cli"]
+ result = subprocess.run([*prefix, *args], capture_output=True, text=True)
+ assert result.returncode == 0, result.stdout + result.stderr
+ payload = json.loads(result.stdout)
+ goal = payload["goals"][0]
+ expected = {row["progress_observation"]["surface_id"] for row in own}
+ for rows in (payload["runs"], goal["latest_runs"]):
+ assert len(rows) == 221
+ assert all(row["agent_id"] == "agent-a" for row in rows)
+ assert {row["progress_observation"]["surface_id"] for row in rows} == expected
+ assert goal["latest_status_run"]["agent_id"] == "agent-a"
+ short_args = ["10" if arg == "221" else arg for arg in args]
+ short = subprocess.run([*prefix, *short_args], capture_output=True, text=True)
+ assert short.returncode == 0, short.stdout + short.stderr
+ short_payload = json.loads(short.stdout)
+ assert len(short_payload["runs"]) == len(short_payload["goals"][0]["latest_runs"]) == 10
+ assert short_payload["goals"][0]["latest_status_run"]["agent_id"] == "agent-a"
+ assert index.read_bytes() == before
+ assert not (runtime / "goals/evidence-goal/rollout-event-log.jsonl").exists()
+
+
+def test_truncated_display_cannot_make_an_old_blocker_new():
+ from loopx.control_plane.work_items.progress_observation import semantic_delta_from_writeback
+
+ rows = [_run(generated_at=f"2026-08-19T01:{i:02}:00Z", progress_observation={
+ "schema_version": "typed_progress_observation_v0", "result_class": "blocked",
+ "blocker_id": f"blocker-{i}", "evidence_ids": [f"proof-{i}"],
+ }) for i in range(40)]
+ obligation = {"obligation_id": "replan-a", "triggers": [{"kind": "typed_progress_repeat"}],
+ "progress_baseline": rows[-1]["progress_observation"]}
+ obligation["replan_context"] = project_replan_context(goal_id="evidence-goal", agent_id="agent-a", runs=rows, obligation=obligation)
+ assert obligation["replan_context"]["coverage_truncated"] is True
+ assert not any(row.get("blocker_id") == "blocker-0" for row in obligation["replan_context"]["coverage_ledger"])
+ with pytest.raises(ValueError, match="complete evidence history"):
+ semantic_delta_from_writeback(obligation=obligation, progress_observation=rows[0]["progress_observation"])
+ result = semantic_delta_from_writeback(obligation=obligation, progress_observation=rows[0]["progress_observation"], history_runs=rows)
+ assert result["accepted"] is False
+ assert result["reason_code"] == "progress_observation_replayed"
+
+
+def test_dense_context_uses_shared_private_snapshot_transport(monkeypatch):
+ from loopx.control_plane.work_items import replan_history_codec
+
+ expected = project_replan_context(goal_id="evidence-goal", agent_id="agent-a", runs=[_run()])
+ monkeypatch.setattr(replan_history_codec, "MAX_REQUEST_BYTES", 1)
+ assert project_replan_context(goal_id="evidence-goal", agent_id="agent-a", runs=[_run()]) == expected
+
+
+def test_unscoped_replan_assignment_ignores_evidence_and_presentation_changes():
+ from loopx.control_plane.goals.goal_frontier import autonomous_replan_scope_decision
+
+ obligation = {"schema_version": "autonomous_replan_obligation_v0", "required": True,
+ "triggers": [{"kind": "periodic_review_due"}]}
+ selected = []
+ for extra in [{}, {"recommended_action": "different presentation"},
+ {"replan_context": {"evidence": [{"summary": "new observation"}]}},
+ {"replan_novelty_policy": {"evidence_source": "legacy-source"}}]:
+ decisions = [
+ autonomous_replan_scope_decision({**obligation, **extra}, agent_id=agent,
+ registered_agent_ids=["agent-a", "agent-b"])
+ for agent in ["agent-a", "agent-b"]
+ ]
+ assert sum(decision["applies"] for decision in decisions) == 1
+ selected.extend(decision["selected_peer_agent"] for decision in decisions)
+ assert len(set(selected)) == 1
+
+
+def test_native_recovery_reads_its_event_source_without_the_removed_cli(tmp_path: Path):
+ from loopx.kunluncode_goal_mode.control_plane import LoopXControlPlane
+
+ runtime = tmp_path / "runtime"
+ registry = tmp_path / "registry.json"
+ registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": []}))
+ log = runtime / "goals/evidence-goal/rollout-event-log.jsonl"
+ log.parent.mkdir(parents=True)
+ event = {"schema_version": "loopx_rollout_event_v0", "goal_id": "evidence-goal",
+ "agent_id": "agent-a", "todo_id": "todo-a", "event_id": "event-a",
+ "event_kind": "refresh_state", "classification": "kunluncode_native_goal_verified",
+ "recorded_at": "2026-08-18T01:00:00Z"}
+ log.write_text("\n".join(json.dumps(item) for item in [
+ event, {**event, "agent_id": "agent-b"}, {**event, "todo_id": "todo-b"},
+ {**event, "recorded_at": "2026-08-17T01:00:00Z"},
+ ]) + "\n")
+ before = log.read_bytes()
+ control = LoopXControlPlane(tmp_path, {"registry": "registry.json",
+ "goal_id": "evidence-goal", "agent_id": "agent-a"})
+ result = control.evidence_since("2026-08-18T00:00:00Z", todo_id="todo-a")
+ assert result["ledger_count"] == 1
+ assert result["ledger"][0]["event_id"] == "event-a"
+ assert log.read_bytes() == before
diff --git a/tests/control_plane/test_replan_host_context_projection.py b/tests/control_plane/test_replan_host_context_projection.py
index de03deb4c4..bb02103282 100644
--- a/tests/control_plane/test_replan_host_context_projection.py
+++ b/tests/control_plane/test_replan_host_context_projection.py
@@ -121,7 +121,7 @@ def test_quota_delivers_coverage_context_and_minimal_replan_action() -> None:
obligation = payload["autonomous_replan_obligation"]
context = obligation["replan_context"]
action = payload["replan_action_packet"]
- assert context["evidence_source"] == "agent_scoped_evidence_log"
+ assert context["evidence_source"] == "compact_run_history"
assert context["delivery"] == "host_projected"
assert context["delivery_receipt"] == {
"schema_version": "replan_context_delivery_receipt_v0",
diff --git a/tests/control_plane/test_replan_novelty_policy.py b/tests/control_plane/test_replan_novelty_policy.py
index c2a708c7f8..a33dd391c0 100644
--- a/tests/control_plane/test_replan_novelty_policy.py
+++ b/tests/control_plane/test_replan_novelty_policy.py
@@ -45,7 +45,7 @@ def test_generic_stall_obligation_carries_novelty_guidance() -> None:
policy = obligation["replan_novelty_policy"]
assert policy["schema_version"] == "replan_evidence_delivery_policy_v0"
- assert policy["evidence_source"] == "agent_scoped_evidence_log"
+ assert policy["evidence_source"] == "compact_run_history"
assert policy["delivery"] == "host_projected"
assert policy["writeback"] == "typed_semantic_delta"
@@ -68,7 +68,7 @@ def test_dead_monitor_obligation_reuses_the_same_repair_delta_contract() -> None
)
policy = obligation["replan_novelty_policy"]
- assert policy["evidence_source"] == "agent_scoped_evidence_log"
+ assert policy["evidence_source"] == "compact_run_history"
assert policy["writeback"] == "typed_semantic_delta"
action = obligation["recommended_action"]
assert "resolve a dead monitor loop" in action
@@ -87,7 +87,7 @@ def test_periodic_review_obligation_reuses_the_same_preflight_contract() -> None
)
policy = obligation["replan_novelty_policy"]
- assert policy["evidence_source"] == "agent_scoped_evidence_log"
+ assert policy["evidence_source"] == "compact_run_history"
action = obligation["recommended_action"]
assert "bounded autonomous periodic review" in action
assert "host-projected coverage ledger" in action
@@ -134,7 +134,7 @@ def test_compact_replan_obligation_keeps_only_authoritative_seam_refs() -> None:
)
compact = compact_replan_obligation(obligation)
assert compact["replan_novelty_policy"] == {
- "evidence_source": "agent_scoped_evidence_log",
+ "evidence_source": "compact_run_history",
"delivery": "host_projected",
"writeback": "typed_semantic_delta",
}
@@ -154,7 +154,7 @@ def test_payload_builder_defaults_to_novelty_guidance_and_policy() -> None:
assert "host-projected coverage ledger" in payload["recommended_action"]
policy = payload["replan_novelty_policy"]
- assert policy["evidence_source"] == "agent_scoped_evidence_log"
+ assert policy["evidence_source"] == "compact_run_history"
assert policy["delivery"] == "host_projected"
assert policy["writeback"] == "typed_semantic_delta"
@@ -199,7 +199,7 @@ def test_payload_builder_replaces_conflicting_policy_extra_fields() -> None:
assert payload["replan_novelty_policy"] == {
"schema_version": "replan_evidence_delivery_policy_v0",
- "evidence_source": "agent_scoped_evidence_log",
+ "evidence_source": "compact_run_history",
"delivery": "host_projected",
"writeback": "typed_semantic_delta",
}
@@ -218,7 +218,7 @@ def test_novelty_policy_materializes_host_context_and_action_packet() -> None:
enriched = {**obligation, "replan_context": context}
action = build_replan_action_packet(enriched)
- assert context["evidence_source"] == "agent_scoped_evidence_log"
+ assert context["evidence_source"] == "compact_run_history"
assert context["delivery"] == "host_projected"
assert action["obligation_id"] == obligation["obligation_id"]
assert action["required_outcome"] == "semantic_delta"
diff --git a/tests/control_plane/test_replan_semantic_action_behavior.py b/tests/control_plane/test_replan_semantic_action_behavior.py
index d6d2191373..d0c9b3d01e 100644
--- a/tests/control_plane/test_replan_semantic_action_behavior.py
+++ b/tests/control_plane/test_replan_semantic_action_behavior.py
@@ -539,9 +539,9 @@ def test_manual_evidence_read_is_context_not_replan_completion(
ScriptedExecToolAction(command=fixture.quota_guard_command),
ScriptedExecToolAction(
command=(
- "loopx --format json evidence-log "
+ "loopx --format json history "
"--goal-id replan-semantic-action-fixture "
- "--agent-id codex-replan-semantic-action --thin"
+ "--agent-id codex-replan-semantic-action"
)
),
]
diff --git a/tests/control_plane/test_replan_successor_guard_reentry.py b/tests/control_plane/test_replan_successor_guard_reentry.py
index 6413296ceb..cd1092b638 100644
--- a/tests/control_plane/test_replan_successor_guard_reentry.py
+++ b/tests/control_plane/test_replan_successor_guard_reentry.py
@@ -27,7 +27,7 @@ def _fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str):
project, runtime = tmp_path / "project", tmp_path / "runtime"
project.mkdir()
state = project / "ACTIVE_GOAL_STATE.md"
- state.write_text("---\nstatus: active\n---\n\n# Synthetic Goal\n\n## Agent Todo\n")
+ state.write_text("---\nstatus: active\n---\n\n# Synthetic Goal\n\n## Objective\nDeliver the independently accepted source outcome.\n\n## Agent Todo\n")
index = runtime / "goals" / GOAL / "runs" / "index.jsonl"
index.parent.mkdir(parents=True)
evidence = index.parent / "synthetic-artifact.json"
@@ -85,6 +85,9 @@ def test_successor_guard_returns_original_settlement_not_repeated_planning(
) -> None:
call, runtime, index = _fixture(tmp_path, monkeypatch, provider)
original = _guard(call)
+ core_goal = original["autonomous_replan_obligation"]["replan_context"]["core_goal"]
+ assert core_goal["objective"] == "Deliver the independently accepted source outcome."
+ assert core_goal["objective_source"] == "active_state"
identity = original["heartbeat_receipt"]["settlement_identity"]
assert identity["binding_kind"] == "autonomous_replan"
obligation_id = identity["replan_obligation_id"]
diff --git a/tests/control_plane_ts/replan_context.test.ts b/tests/control_plane_ts/replan_context.test.ts
new file mode 100644
index 0000000000..9690415d2d
--- /dev/null
+++ b/tests/control_plane_ts/replan_context.test.ts
@@ -0,0 +1,94 @@
+import assert from "node:assert/strict";
+import test from "node:test";
+import {projectReplanContext, validateReplanContext} from "../../loopx/control_plane/work_items/replan_context.ts";
+import type {JsonObject} from "../../loopx/control_plane/effect_program.ts";
+
+const progress = {schema_version: "typed_progress_observation_v0", result_class: "unchanged",
+ surface_id: "surface-a", work_item_id: "todo-a", fingerprint: "same-work"};
+function row(time: number, agent = "agent-a"): JsonObject {
+ return {goal_id: "goal-a", agent_id: agent, generated_at: new Date(time * 1000).toISOString(),
+ observed_at: time, progress_observation: progress, health_check: "Probe did not improve the result."};
+}
+const request = {operation: "project", goal_id: "goal-a", agent_id: "agent-a",
+ obligation: {obligation_id: "replan-a", triggers: []}, rows: [] as JsonObject[]};
+
+test("scope, chronology, replay deduplication, and coverage are owned by code", () => {
+ const context = projectReplanContext({...request,
+ rows: [row(2), row(3, "agent-b"), row(1), row(2), {...row(4), goal_id: "goal-b"}]});
+ const evidence = context.evidence as JsonObject[];
+ assert.equal(evidence.length, 1);
+ assert.deepEqual(evidence.map(item => item.generated_at), [row(2).generated_at]);
+ assert.equal((context.coverage_ledger as JsonObject[]).length, 1);
+ assert.equal(context.evidence_count, 2);
+ assert.equal(evidence[0].occurrences, 2);
+ assert.equal(evidence[0].first_observed_at, row(1).generated_at);
+ assert.equal(context.coverage_count, 1);
+ assert.match(String(evidence[0].read_action), /history --goal-id goal-a --agent-id agent-a --evidence-ref replan-evidence-/);
+});
+
+test("context is bounded with explicit omitted coverage, not a completeness claim", () => {
+ const rows = Array.from({length: 30}, (_, i) => ({...row(i),
+ progress_observation: {...progress, fingerprint: "work-" + i}}));
+ const context = projectReplanContext({...request, rows});
+ assert.equal((context.evidence as unknown[]).length, 24);
+ assert.equal((context.coverage_ledger as unknown[]).length, 24);
+ assert.equal(context.evidence_truncated, true);
+ assert.equal(context.coverage_truncated, true);
+ assert.equal(context.coverage_count, 30);
+ assert.equal(validateReplanContext(context), context);
+});
+
+test("older different results survive many recent attempts, while repeated records retain their span", () => {
+ const old = {...row(1), progress_observation: {...progress, result_class: "advanced", fingerprint: "old-best"}};
+ const repeated = Array.from({length: 100}, (_, i) => row(i + 100));
+ const context = projectReplanContext({...request, rows: [old, ...repeated]});
+ const evidence = context.evidence as JsonObject[];
+ assert.equal(evidence.length, 2);
+ assert.equal(evidence[0].occurrences, 100);
+ assert.equal(evidence[0].first_observed_at, row(100).generated_at);
+ assert.equal(evidence[1].coverage_ref, "old-best");
+ assert.equal(context.evidence_truncated, false);
+ const manyRoutes = Array.from({length: 40}, (_, i) => ({...row(i + 10),
+ progress_observation: {...progress, probe_kind: "probe-" + i, fingerprint: "work-" + i}}));
+ const bounded = projectReplanContext({...request, rows: [old, ...manyRoutes]});
+ assert.equal((bounded.evidence as JsonObject[]).length, 24);
+ assert.ok((bounded.coverage_ledger as JsonObject[]).some(item => item.fingerprint === "old-best"));
+ assert.equal((bounded.omitted_evidence as JsonObject).distinct_observation_count, 17);
+});
+
+test("core Goal is independent of work-scoped acceptance and recent evidence", () => {
+ const acceptance = {scope: {kind: "selected_work", todo_ids: ["todo-a"]}, objective: "Check one route"};
+ const context = projectReplanContext({...request, goal_facts: {registry_objective: "Registered objective",
+ active_state_objective: "Deliver the complete accepted outcome", acceptance_contract: acceptance}, rows: [row(1)]});
+ assert.deepEqual(context.core_goal, {objective: "Deliver the complete accepted outcome",
+ objective_source: "active_state", objective_missing: false, acceptance_contract: acceptance});
+ const absent = projectReplanContext({...request, goal_facts: {acceptance_contract: acceptance}, rows: [row(1)]});
+ assert.equal((absent.core_goal as JsonObject).objective, null);
+ assert.equal((absent.core_goal as JsonObject).objective_missing, true);
+ assert.throws(() => validateReplanContext({...context, core_goal: null}));
+});
+
+test("an empty source is legal; a missing or unsupported contract is not empty", () => {
+ const empty = projectReplanContext(request);
+ assert.deepEqual(empty.evidence, []);
+ assert.deepEqual(empty.coverage_ledger, []);
+ for (const bad of [undefined, {}, {...empty, schema_version: "future"},
+ {...empty, coverage_ledger: null}, {...empty, evidence: undefined},
+ {...empty, uncovered_frontier: null}]) {
+ assert.throws(() => validateReplanContext(bad));
+ }
+ assert.throws(() => projectReplanContext({...request, rows: undefined}), /rows/);
+ assert.throws(() => projectReplanContext({...request, rows: [{...row(1),
+ progress_observation: {...progress, schema_version: "future"}}]}), /schema_version/);
+});
+
+test("references resolve only the exact immutable row within the same Goal and Agent", () => {
+ const context = projectReplanContext({...request, rows: [row(1)]});
+ const ref = (context.evidence as JsonObject[])[0].evidence_ref;
+ const resolve = {...request, operation: "resolve", evidence_ref: ref, rows: [row(1)]};
+ assert.equal((projectReplanContext(resolve).evidence as JsonObject).health_check, row(1).health_check);
+ for (const invalid of [{agent_id: "agent-b"}, {goal_id: "goal-b"}, {rows: []},
+ {rows: [{...row(1), health_check: "Replaced observation"}]}, {evidence_ref: "missing"}]) {
+ assert.throws(() => projectReplanContext({...resolve, ...invalid}), /unavailable/);
+ }
+});
diff --git a/tests/test_kunluncode_goal_mode.py b/tests/test_kunluncode_goal_mode.py
index 093aab60b4..fec72b0ccb 100644
--- a/tests/test_kunluncode_goal_mode.py
+++ b/tests/test_kunluncode_goal_mode.py
@@ -923,13 +923,12 @@ def capture(
runner=capture,
)
- control.evidence_since("2026-08-18T00:00:00Z", todo_id="todo-native-1")
control.record_verified_delivery(
mode="goal-pro", todo_id="todo-native-1"
)
control.spend(todo_id="todo-native-1")
- assert len(commands) == 3
+ assert len(commands) == 2
for command in commands:
todo_index = command.index("--todo-id")
assert command[todo_index + 1] == "todo-native-1"
@@ -955,7 +954,7 @@ def evidence_since(self, _since: str, *, todo_id: str) -> dict[str, object]:
self.calls.append("evidence_since")
if self.evidence_error:
raise KunlunNativeGoalRuntimeError("evidence ledger unavailable")
- return {"ok": True, "ledger": list(self.ledger)}
+ return {"ok": True, "schema_version": "agent_scoped_evidence_log_v0", "ledger": list(self.ledger)}
def record_verified_delivery(
self, *, mode: str, todo_id: str