From 4f5e36ec38c7c5a8f3b680de252a00d61784bfc5 Mon Sep 17 00:00:00 2001 From: 222twotwotwo Date: Sat, 26 Sep 2026 19:57:53 +0800 Subject: [PATCH 1/2] feat(memory): enforce capacity budgets for long-lived Memories --- benchmark/README.md | 1 + benchmark/memory_capacity/README.md | 70 ++ benchmark/memory_capacity/__init__.py | 15 + benchmark/memory_capacity/__main__.py | 211 ++++++ docs/en/development/memory-layer.md | 62 ++ docs/en/rfcs/1718_memory_capacity_contract.md | 629 ++++++++++++++++++ docs/zh/development/memory-layer.md | 56 ++ docs/zh/rfcs/1718_memory_capacity_contract.md | 565 ++++++++++++++++ .../plugins/powercontext/hooks/bind_tools.py | 1 + .../dsh/plugins/powercontext/lib/index.js | 11 + .../powercontext/src/operations.generated.ts | 1 + .../powercontext/src/operations.generated.ts | 1 + .../powercontext/src/operations.generated.ts | 1 + openapi/powercontext.yaml | 93 ++- .../builtin/artifacts/memory/__init__.py | 12 + .../builtin/artifacts/memory/errors.py | 8 + .../builtin/artifacts/memory/models.py | 52 +- .../builtin/artifacts/memory/protocols.py | 5 + .../builtin/artifacts/memory/service.py | 150 ++++- .../builtin/persistence/artifacts.py | 7 +- .../builtin/persistence/family_management.py | 7 + .../builtin/persistence/memory.py | 32 +- .../builtin/runtime/application.py | 10 + .../builtin/runtime/composition.py | 22 + src/powercontext/builtin/runtime/config.py | 13 + .../builtin/runtime/relational.py | 20 + src/powercontext/client/client.py | 8 + src/powercontext/http/__init__.py | 8 + src/powercontext/http/_generated/models.py | 41 ++ .../http/_generated/operations.py | 29 + src/powercontext/http/_generated/schema.py | 99 ++- src/powercontext/server/app.py | 39 +- .../builtin/artifacts/memory/test_capacity.py | 466 +++++++++++++ tests/e2e/test_memory_capacity.py | 122 ++++ tests/test_api_contract.py | 14 + 35 files changed, 2868 insertions(+), 13 deletions(-) create mode 100644 benchmark/memory_capacity/README.md create mode 100644 benchmark/memory_capacity/__init__.py create mode 100644 benchmark/memory_capacity/__main__.py create mode 100644 docs/en/rfcs/1718_memory_capacity_contract.md create mode 100644 docs/zh/rfcs/1718_memory_capacity_contract.md create mode 100644 tests/builtin/artifacts/memory/test_capacity.py create mode 100644 tests/e2e/test_memory_capacity.py diff --git a/benchmark/README.md b/benchmark/README.md index f343fcb753..c846288590 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -7,3 +7,4 @@ long time or incur inference cost. - [`locomo/`](locomo/README.md): conversation-memory retrieval and end-to-end question-answer accuracy. - [`locomo_plus/`](locomo_plus/README.md): pinned LoCoMo-Plus factual and cognitive memory evaluation, with a four-case smoke profile, a bundled ten-case dataset with complete histories, and an explicit full profile. +- [`memory_capacity/`](memory_capacity/README.md): manifest capacity, append cost, and tombstone compaction without model calls. diff --git a/benchmark/memory_capacity/README.md b/benchmark/memory_capacity/README.md new file mode 100644 index 0000000000..76d2f67909 --- /dev/null +++ b/benchmark/memory_capacity/README.md @@ -0,0 +1,70 @@ +# Memory capacity benchmark + +This provider-free benchmark appends one entry per Revision at 200, 1,000 and 5,000 entries, then retires 80% of the +entries, advances the tombstone recovery window, previews and commits compaction, and appends once more. It measures +canonical bytes, database bytes, mean append latency, the final 100 appends, affected projection rows, and preservation +of FTS hit identities across compaction. It retains historical manifests and entry bodies. + +```bash +uv run python -m benchmark.memory_capacity --output .artifacts/memory-capacity/run-01/sqlite.json +uv run python -m benchmark.memory_capacity --backend oceanbase --output .artifacts/memory-capacity/run-01/oceanbase.json +``` + +Choose a new run directory for each measurement. Raw JSON, logs, and databases are local acceptance artifacts under the +Git-ignored `.artifacts/` directory. Attach reviewed results to the relevant PR or CI run and keep reusable methodology +and concise measurement summaries in this README. + +SQLite creates a fresh database beside the output JSON, retained for inspection. Samples checkpoint and truncate the +WAL; the compaction report also records file bytes after `VACUUM`. +The full run requires several GB of disk because every historical manifest remains authoritative. + +OceanBase requires `POWERCONTEXT_TEST_OCEANBASE_URL` pointing at a disposable test database. It creates a unique Scope; +reported database bytes cover the entire database, so use an otherwise idle database. The default observation delay is +30 seconds, adjustable with `--reclamation-delay`. No storage-engine compaction is forced. An unavailable database is +recorded as `not_run`, never represented as a measured result. No model calls are made on either backend. + +OceanBase database bytes sum `OCCUPY_SIZE` from `oceanbase.DBA_OB_TABLE_SPACE_USAGE` for the selected database, including +its index tables. This measures reported SSTable occupancy, excluding memtables, transaction logs, and preallocated +cluster files; early samples can be zero before a flush. `information_schema.tables` statistics can remain zero even +after SSTables occupy space and are not used for this measurement. + +`reclaimed_bytes` compares full canonical contents, including the new compaction audit records. The follow-up Revision +shows the continuing cost of the reduced manifest. Projection rows are counted using the database cursor's affected +row count; zero-row deletes do not count. FTS and active-head rows are included; immutable entry bodies are separate. + +The latency sample is a local calibration, not an SLA. Default budgets remain 5,000 active entries, 10,000 manifest +entries, and 4 MiB pending deployment-specific latency requirements. + + +## Recorded local SQLite run + +Windows, Python 3.11.9, one append per Revision; other repository validation ran concurrently, so these latencies are +indicative and must not be used as an isolated performance baseline. Byte counts are exact. + +| Entries | Canonical bytes | Checkpointed database bytes | Mean append (ms) | Final 100 mean (ms) | Projection rows/append | +| --- | --- | --- | --- | --- | --- | +| 200 | 44,866 | 5,607,424 | 14.42 | 15.85 | 2 | +| 1,000 | 223,266 | 114,847,744 | 29.29 | 46.09 | 2 | +| 5,000 | 1,115,266 | 2,804,195,328 | 116.19 | 217.74 | 2 | + +Compaction removed 4,000 tombstones with zero projection writes and preserved the same sentinel search hit. Complete +canonical content fell from 1,123,307 to 939,091 bytes in the compaction Revision, then to 223,489 bytes on the next +append. The audit delta accounts for the difference. Database bytes after that append were 2,815,942,656 and after +`VACUUM` were 2,815,492,096: old manifests still occupy live pages. + +## OceanBase validation status + +On 2026-09-26, all 13 OceanBase capacity behavior cases passed against OceanBase CE 4.3.5.6 without skips. Another +67 SQLite, HTTP/API contract, and projection regression checks passed. The complete 5,000-entry scale run and +compaction cycle also passed locally: 4,000 tombstones were removed, zero projection rows were written by compaction, +and the sentinel FTS hit identity was preserved. + +| Entries | Canonical bytes | Observed database bytes | Mean append (ms) | Final 100 mean (ms) | Projection rows/append | +| --- | ---: | ---: | ---: | ---: | ---: | +| 200 | 44,866 | 73,706 | 24.69 | 25.94 | 1 | +| 1,000 | 223,266 | 73,706 | 36.14 | 47.70 | 1 | +| 5,000 | 1,115,266 | 1,679,697,682 | 99.85 | 180.02 | 1 | + +The compaction revision reduced manifest entries from 5,000 to 1,000 and canonical bytes from 1,123,307 to 939,091; +the follow-up append produced 1,001 entries and 223,489 bytes. Database bytes stayed at 1,770,500,275 during the +compaction observation window, as historical revisions remain retained. diff --git a/benchmark/memory_capacity/__init__.py b/benchmark/memory_capacity/__init__.py new file mode 100644 index 0000000000..dd40eb90f4 --- /dev/null +++ b/benchmark/memory_capacity/__init__.py @@ -0,0 +1,15 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Capacity envelope benchmark without model calls.""" diff --git a/benchmark/memory_capacity/__main__.py b/benchmark/memory_capacity/__main__.py new file mode 100644 index 0000000000..d1e56e16f4 --- /dev/null +++ b/benchmark/memory_capacity/__main__.py @@ -0,0 +1,211 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import platform +from pathlib import Path +from statistics import mean +from time import perf_counter +from uuid import uuid4 + +from pydantic import SecretStr +from sqlalchemy import event, text + +from powercontext.builtin.artifacts.memory import MemoryEntryInput +from powercontext.builtin.artifacts.memory.canonical import memory_content_bytes +from powercontext.builtin.persistence.oceanbase import OceanBaseConfig +from powercontext.builtin.persistence.sqlite import SQLiteConfig +from powercontext.builtin.runtime import BuiltinConfig, open_builtin_contexts +from powercontext.builtin.runtime.config import RuntimeConfig + + +async def run(args): # noqa: C901 + if args.backend == "oceanbase": + url = os.environ.get("POWERCONTEXT_TEST_OCEANBASE_URL") + if not url: + return { + "backend": "oceanbase", + "status": "not_run", + "reason": "POWERCONTEXT_TEST_OCEANBASE_URL is not configured", + } + database = OceanBaseConfig(url=SecretStr(url)) + database_path = None + else: + database_path = args.output.parent / ("memory-capacity-" + uuid4().hex + ".db") + database = SQLiteConfig(url=f"sqlite+aiosqlite:///{database_path.resolve().as_posix()}") + config = BuiltinConfig(database=database, runtime=RuntimeConfig(memory_compaction_enabled=True)) + output = { + "backend": args.backend, + "status": "running", + "python": platform.python_version(), + "platform": platform.platform(), + "counts": args.counts, + "final_window": args.final_window, + "measurements": [], + "reclamation": "WAL checkpoint before samples; VACUUM measured separately" + if database_path + else f"observed table bytes after {args.reclamation_delay}s; no forced engine compaction", + } + async with open_builtin_contexts(config) as contexts: + scope_id = "capacity-benchmark-" + uuid4().hex + service = (await contexts.get(scope_id)).artifacts.memory + projection_rows = 0 + + def record(_connection, cursor, statement, _parameters, _context, _executemany): + nonlocal projection_rows + sql = statement.lower().lstrip() + if sql.startswith(("insert", "update", "delete")) and any( + name in sql for name in ("pc_memory_entry_heads", "pc_memory_entry_fts") + ): + projection_rows += max(cursor.rowcount, 0) + + event.listen(contexts.database.engine.sync_engine, "after_cursor_execute", record) + + async def database_bytes(): + if database_path is not None: + async with contexts.database.engine.connect() as connection: + await connection.exec_driver_sql("PRAGMA wal_checkpoint(TRUNCATE)") + return database_path.stat().st_size + async with contexts.database.connection() as connection: + # General table statistics can remain zero while OceanBase already occupies SSTable space. + return int( + await connection.scalar( + text( + "SELECT COALESCE(SUM(OCCUPY_SIZE), 0) " + "FROM oceanbase.DBA_OB_TABLE_SPACE_USAGE WHERE DATABASE_NAME = DATABASE()" + ) + ) + or 0 + ) + + async def snapshot(memory): + capacity = await service.capacity(memory) + return { + "revision": memory.revision, + "active_entries": capacity.active_entry_count, + "manifest_entries": capacity.manifest_entry_count, + "manifest_bytes": capacity.manifest_bytes, + "database_bytes": await database_bytes(), + } + + memory = None + latencies = [] + for number in range(1, max(args.counts) + 1): + started = perf_counter() + memory = await service.remember( + memory=memory, + entries=( + MemoryEntryInput( + kind="fact", + text=f"Capacity benchmark record {number}; project token item{number}.", + ), + ), + mode="append", + ) + latencies.append((perf_counter() - started) * 1000) + if number in args.counts: + sample = await snapshot(memory) + sample.update({ + "mean_append_ms": mean(latencies), + "mean_final_window_append_ms": mean(latencies[-args.final_window :]), + "projection_rows_per_append": projection_rows / number, + }) + output["measurements"].append(sample) + print(json.dumps(sample), flush=True) + args.output.write_text(json.dumps(output, indent=2) + "\n", encoding="utf-8") + assert memory is not None # noqa: S101 + entries = await service.entries(memory) + retired_entries = entries[: int(len(entries) * 0.8)] + retained = entries[-1] + memory = await service.forget(memory, entries=retired_entries) + for number in range(config.runtime.memory_compaction_min_tombstone_revisions): + memory = await service.remember( + memory=memory, + entries=( + MemoryEntryInput( + kind="fact", + text=f"Capacity sentinel retained record generation {number}.", + entry=retained, + ), + ), + mode="append", + ) + assert memory is not None # noqa: S101 + retained = next(entry for entry in await service.entries(memory) if entry.entry_id == retained.entry_id) + assert memory is not None # noqa: S101 + before = await snapshot(memory) + hits_before = await service.search("Capacity sentinel retained", memories=(memory,), mode="fts") + preview = await service.compact(memory, dry_run=True) + writes_before = projection_rows + compacted = await service.compact(memory) + compaction_writes = projection_rows - writes_before + hits_after = await service.search("Capacity sentinel retained", memories=(compacted.memory,), mode="fts") + identities_before = [(hit.entry_id, hit.entry_version_id) for hit in hits_before.hits] + identities_after = [(hit.entry_id, hit.entry_version_id) for hit in hits_after.hits] + output["compaction"] = { + "before": before, + "after": await snapshot(compacted.memory), + "removed_entries": len(compacted.entry_ids), + "reclaimed_bytes": compacted.reclaimed_bytes, + "preview_matches": preview.entry_ids == compacted.entry_ids, + "projection_rows_written": compaction_writes, + "search_hit_count_before": len(identities_before), + "search_hit_count_after": len(identities_after), + "search_identities_preserved": identities_before == identities_after, + } + if not identities_before or identities_before != identities_after or compaction_writes: + raise RuntimeError("capacity benchmark invariants failed") # noqa: TRY003 + followup = await service.remember( + memory=compacted.memory, + entries=(MemoryEntryInput(kind="fact", text="New entry after compaction."),), + mode="append", + ) + assert followup is not None # noqa: S101 + output["compaction"]["followup"] = await snapshot(followup) + if database_path is not None: + async with contexts.database.engine.connect() as connection: + await connection.exec_driver_sql("VACUUM") + output["compaction"]["post_vacuum_database_bytes"] = await database_bytes() + else: + await asyncio.sleep(args.reclamation_delay) + output["compaction"]["delayed_database_bytes"] = await database_bytes() + output["compaction"]["followup_canonical_bytes"] = len(memory_content_bytes(followup.content)) + event.remove(contexts.database.engine.sync_engine, "after_cursor_execute", record) + output["status"] = "completed" + return output + + +def main(): + parser = argparse.ArgumentParser(description="Measure the Memory capacity envelope without inference.") + parser.add_argument("--backend", choices=("sqlite", "oceanbase"), default="sqlite") + parser.add_argument("--counts", nargs="+", type=int, default=[200, 1000, 5000]) + parser.add_argument("--final-window", type=int, default=100) + parser.add_argument("--reclamation-delay", type=float, default=30) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if min(args.counts) < 5 or max(args.counts) > 5000 or args.final_window < 1 or args.reclamation_delay < 0: + parser.error("counts must be 5..5000, final-window positive, and reclamation-delay nonnegative") + args.output.parent.mkdir(parents=True, exist_ok=True) + result = asyncio.run(run(args)) + args.output.write_text(json.dumps(result, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") + print(args.output) + + +if __name__ == "__main__": + main() diff --git a/docs/en/development/memory-layer.md b/docs/en/development/memory-layer.md index 956a5ac0b8..55a2b7d1c4 100644 --- a/docs/en/development/memory-layer.md +++ b/docs/en/development/memory-layer.md @@ -71,6 +71,68 @@ revised = await memory.revise( `retire()` marks an entry inactive without deleting immutable content. `changes()` returns compact revision changes. Expected revisions and citations preserve optimistic concurrency without requiring callers to rebuild references. +## Capacity and tombstone compaction + +`await runtime.memory.for_scope(scope_id).capacity()` reports the current head's active entries, total manifest +entries, exact canonical content bytes, eligible tombstones, budget, and exceeded dimensions. The direct service +method `await service.capacity(memory)` measures the exact Revision supplied. Remote callers use +`POST /v1/memory/capacity` with `{"scope_id": "project-alpha"}`, or +`PowerContextClient.get_memory_capacity(GetMemoryCapacityRequest(scope_id="project-alpha"))`. +A Scope without a Memory returns 404; reading capacity does not create one. + +`RuntimeConfig` supplies deployment-wide defaults: + +| Setting | Default | +| --- | --- | +| `memory_max_active_entries` | 5,000 | +| `memory_max_manifest_entries` | 10,000 | +| `memory_max_manifest_bytes` | 4,194,304 | +| `memory_compaction_enabled` | `False` | +| `memory_compaction_min_tombstone_revisions` | 10 | +| `memory_max_history_revisions` | 100 | + +The capacity defaults bound growth of each Revision; they do not guarantee append latency or cap total database size. +Retained historical manifests keep accumulating. Tune deployment budgets against representative backend measurements. + +Active-entry limits cannot exceed manifest-entry limits. Explicit writes, extraction, and generic Artifact management +share the budget. A write is refused only when it exceeds a limit and increases that dimension relative to the base. +The deterministic priority is bytes, manifest entries, then active entries. HTTP returns +`409 memory_capacity_exceeded` with `dimension`, `limit`, and `observed`; the rejected write persists no content. +`manifest_bytes` includes the complete canonical Revision content, including its changes and reasons. + +`forget()` and `organize()` remain available over budget. `reactivate()` checks active-entry growth only. +Compaction removes aged, untagged inactive entries from the current manifest. Enable it explicitly on `RuntimeConfig`, +or construct a `MemoryService` with `MemoryCompactionPolicy(enabled=True)`, then preview the operation: + +```python +preview = await service.compact(memory, dry_run=True, limit=100) +result = await service.compact(memory, limit=100) +memory = result.memory +``` + +A preview works while compaction is disabled and writes no Revision. Eligibility counts completed Revision advances: +an entry deactivated at Revision 2 qualifies at Revision 12 with the default age of 10. Reactivation and a subsequent +deactivation restart the window. Only this recent window is read. No-op maintenance does not advance the Revision; +if every tombstone is too recent, explicitly configure `memory_compaction_min_tombstone_revisions=0` (or +`MemoryCompactionPolicy(enabled=True, min_tombstone_revisions=0)`) to preview and compact immediately. Zero bypasses +only the recovery window: active entries and tagged tombstones remain protected. Keep the default window unless +immediate recovery is needed, because a compacted entry cannot be reactivated. +Tags protect inactive entries; a tag added during compaction aborts the transaction with +`CapabilityNotSupportedError("compaction-tag-conflict")` so the caller can preview again. + +Compaction preserves all entry bodies, prior Revisions, and exact citations. A removed entry cannot be reactivated or +listed in the current manifest. Each removal records the additive `compact` change operation; update consumers that +exhaustively enumerate change operations before enabling compaction. Compaction is available in process only. +`reclaimed_bytes` is the signed difference between complete canonical contents. New audit records or a long reason +can outweigh a small directory reduction; subsequent revisions no longer carry those compaction records. + +`MemoryService.revisions()` refuses histories longer than `memory_max_history_revisions` with +`CapabilityNotSupportedError("history-window")` before loading them. Stored history and exact Revision reads remain +available; results are never silently truncated. The default 100 Revisions can already contain about 400 MiB of +canonical content near the 4 MiB budget, before object overhead. This is a read fan-out bound, not a hard memory limit; +lowered budgets and relief operations can leave Revisions above the byte budget. Increase the configurable history +limit only when the caller can afford the complete snapshots. This bound does not paginate `entries()` or `changes()`. + ## Search, expand, and cite SQLite and OceanBase both initialize a full-text index, so either database can search without an embedding model: diff --git a/docs/en/rfcs/1718_memory_capacity_contract.md b/docs/en/rfcs/1718_memory_capacity_contract.md new file mode 100644 index 0000000000..31efa66b59 --- /dev/null +++ b/docs/en/rfcs/1718_memory_capacity_contract.md @@ -0,0 +1,629 @@ +- Proposal Name: `memory_capacity_contract` +- Start Date: 2026-09-23 +- RFC PR: [oceanbase/powercontext#1718](https://github.com/oceanbase/powercontext/pull/1718) +- Tracking Issue: [oceanbase/powercontext#1718](https://github.com/oceanbase/powercontext/issues/1718) +- Related RFCs: [RFC 0014](0014_memory_layer_design.md) and [RFC 1652](1652_memory_quality_and_lifecycle.md) +- Related work: [#1321](https://github.com/oceanbase/powercontext/issues/1321), + [#1709](https://github.com/oceanbase/powercontext/pull/1709), [#1656](https://github.com/oceanbase/powercontext/issues/1656), + [#1657](https://github.com/oceanbase/powercontext/issues/1657), and + [#1425](https://github.com/oceanbase/powercontext/issues/1425) + +# Summary + +This RFC defines the capacity contract for a long-lived Memory Artifact: what is measured, where the ceiling is, what +happens at the ceiling, and how an operator recovers headroom. + +A Memory gains three measured dimensions (active entries, manifest entries, manifest bytes), a configured budget over +them, and a deterministic refusal when a write would cross the budget. Recovery is explicit: `forget()` retires entries +from the active surface, and a new opt-in `compact()` operation drops qualifying inactive tombstones from the *current* +manifest without deleting any entry body, any prior Revision, or any exact citation. + +Automatic Memory splitting and routing are **not** part of this RFC. They require the routing manifest that RFC 0014 +lists as a future possibility, and a second identity layer that RFC 0014 declares a non-goal. This RFC defines stable, +observable failure at the ceiling instead, and specifies the seam a later routing design plugs into. + +# Motivation + +PR #1709 removed the projection write amplification reported in #1321: an append now rewrites only the rows of the +entry it changed. It deliberately did not define a capacity boundary, and #1321 was closed with that gap open. + +What remains is that a Memory has no ceiling. Every Revision stores a complete `flat-v1` manifest, so appending entry +*N* writes a manifest of *N* items; inactive entries stay in that manifest forever as tombstones; and nothing in the +public API tells a caller how close a Memory is to a practical limit, or refuses when it passes one. RFC 0014 records +the first half of this as a drawback: + +> `flat-v1` duplicates the directory and accumulates inactive tombstones indefinitely, so manifest cost grows linearly +> with the number of entries. + +and defers the second half to empirical work: + +> Empirical results will determine thresholds for Memory splitting, inactive-tombstone compaction, and a public routing +> manifest. + +RFC 1652 then supplies the *logical* lifecycle (importance, retention tiers, reversible automated deactivation) and +explicitly declines the physical half, pointing at a later design: + +> Physical tombstone/manifest compaction, legal retention, external erasure, and cross-artifact cleanup are out of +> scope. + +This RFC is that design, narrowed to one question: the capacity envelope of a single Memory Artifact. It is the step +RFC 1652 anticipates as "propose a separate RFC only if inventory and evaluation demonstrate a storage problem" — #1321 +is that demonstration. + +The outcome we expect: an operator can read a Memory's capacity through a public API, a runaway writer fails with a +specific actionable error instead of degrading silently, and an operator can recover capacity through supported +lifecycle operations and policy settings without losing a citation. + +## Non-goals + +- **No automatic split or routing.** Memory identity is the Artifact ID (RFC 0014), and a Scope resolves exactly one + Memory Artifact ID. Routing to a second Memory requires a routing manifest and a mapping identity that RFC 0014 + excludes. This RFC specifies the refusal, plus the interface a routing design would later satisfy. +- **No deletion of Revisions or entry bodies.** Old Revisions and old entry versions are never rewritten *or removed*. + Physical erasure, legal retention, and cross-artifact cleanup belong to #1425. +- **No change to `flat-v1` storage growth.** Per-Revision manifest duplication is inherent to the format. This RFC + bounds and observes that growth; a delta manifest format is sketched under future possibilities. +- **No cursor pagination.** #1656 owns bounded entry listing and #1657 owns history pagination. This RFC bounds the + *internal* read fan-out those issues depend on and defines no cursor of its own. +- **No quality or importance scoring.** Which entry deserves to survive is RFC 1652's question. This RFC only counts. + +# Guide-level explanation + +## A Memory now has a readable capacity + +Every Memory reports where it stands: + +```python +capacity = await memory_service.capacity(memory) + +capacity.active_entry_count # 412 entries on the active retrieval surface +capacity.manifest_entry_count # 468 active + inactive items in the current manifest +capacity.manifest_bytes # 104_568 canonical bytes committed by this Revision +capacity.compactable_entry_count # 31 tombstones currently eligible for compact() +capacity.budget # the configured ceiling +capacity.exceeded # () — no dimension is over budget +``` + +`manifest_bytes` is not an estimate. It is `len(memory_content_bytes(content))`, the exact byte string that the Revision +content hash already commits to, so the number an operator reads is the number the storage layer writes. + +## Crossing the ceiling is a specific, actionable failure + +A write that would push a Memory past its budget is refused before anything is persisted: + +``` +409 Conflict +{ + "code": "memory_capacity_exceeded", + "message": "The Memory has reached its capacity budget.", + "details": {"dimension": "manifest_bytes", "limit": 4194304, "observed": 4194527} +} +``` + +The refusal names the dimension that bound, the configured limit, and what the rejected write would have produced. The +same write retried unchanged fails identically — it is a state conflict, not a transient error. + +## Recovering capacity at the budget + +A capacity ceiling that blocked its own remedy would brick a Memory. So relief always runs, even over budget: + +- `forget()` retires entries from the active surface. Always permitted. +- `organize(mode="dedupe")` retires exact duplicates. Always permitted. +- `compact()` drops qualifying tombstones from the current manifest. Never blocked by capacity; explicit enablement + and eligibility rules still apply. + +Only operations that *grow* the dimension that is over budget are refused: appending or revising an entry, and +reactivating an entry when the active surface is already full. + +If a full Memory has only recent tombstones, an operator can set the minimum age to zero and preview compaction +without creating artificial Revisions. Tagged entries remain protected: retaining those tags may require increasing +the budget instead of compacting their entries. The capacity contract does not override retention decisions. + +## Compaction removes tombstones, not history + +`forget()` leaves an inactive item in the manifest so the entry can be reactivated and so its history stays readable. +That is the right default, and it is also why manifests only grow. `compact()` is the explicit way to reclaim that +space: + +```python +plan = await memory_service.compact(memory, dry_run=True) +plan.entry_ids # the tombstones that qualify +plan.reclaimed_bytes # signed decrease in complete canonical content bytes + +result = await memory_service.compact(memory) # requires compaction to be enabled +memory = result.memory +``` + +What compaction does **not** touch is the part that makes Memory citable. Entry bodies stay in +`pc_memory_entry_versions`. Every prior Revision keeps its own manifest. A Handoff citation resolves +`ArtifactRef + entry_id + entry_version_id` against the **exact Revision it names**, so a citation written before +compaction still validates afterward, byte for byte. + +What it does change, and what an operator must accept before enabling it: a compacted entry is gone from the *current* +manifest, so `reactivate()` can no longer restore it and `list(include_inactive=True)` no longer shows it. Entries with +tags are excluded so their tags continue to resolve. Compaction is **opt-in, dry-run first, and irreversible for the +active surface**. The default age of 10 completed Revision advances keeps an accidental `forget()` recoverable; +explicitly setting the age to zero permits immediate compaction. Previews work while compaction is disabled. + +## Defaults + +The default budget is 5,000 active entries, 10,000 manifest entries, and 4 MiB of complete canonical content. These +limits bound growth of each Revision; they do not guarantee latency or cap total database size. Compaction is disabled +by default because removing an entry from the current manifest prevents reactivation. The default history read limit +is 100 Revisions, with an explicit error rather than silent truncation. + +# Reference-level explanation + +## Design invariants + +1. **Authoritative content is never destroyed.** No entry body row is deleted, no prior Revision is deleted or + rewritten, no content hash changes. Compaction only decides which items the *next* manifest carries. +2. **Citations are stable.** Exact-Revision citation validation is unaffected by any operation in this RFC, because it + resolves against the Revision it names rather than the current head. +3. **Budgets never block relief.** Deactivation, deduplication, and enabled compaction run regardless of budget state. +4. **Refusal is side-effect free and deterministic.** The budget is evaluated on the fully prepared next manifest, + before the transaction opens. The same input produces the same decision. +5. **#1709's guarantees hold.** Enforcement adds no per-entry I/O to the append path, and compaction writes no + projection rows at all. +6. **Backend-neutral semantics.** Counts, bytes, decisions, and errors are identical on SQLite and OceanBase. Only + physical space reclamation differs. + +## Measured dimensions + +All three derive from the manifest of one exact Revision. None is persisted redundantly, matching RFC 0014's rule that +entry counts are derived from the manifest. + +| Dimension | Definition | Grows on | Shrinks on | +| --- | --- | --- | --- | +| `active_entry_count` | manifest items with `state == "active"` | add, reactivate | deactivate | +| `manifest_entry_count` | all manifest items | add | compact | +| `manifest_bytes` | `len(memory_content_bytes(content))` | content-dependent, including audit changes | content-dependent | + +`manifest_bytes` is the load-bearing dimension: it is what each Revision physically writes, and it is the only one that +accounts for identifier and hash width rather than assuming a fixed per-entry cost. The two counts exist because they +are what an operator reasons about, and because an entry-count ceiling catches a pathological writer earlier than a byte +ceiling does. + +`deactivate` lowers the active count but keeps the manifest count unchanged. Every operation replaces the Revision's +audit changes and reasons, so complete canonical bytes can grow or shrink independently of either count. Compaction +reduces the directory, but its new audit records can make that Revision larger; later Revisions do not repeat those +records. This is why `reclaimed_bytes` is signed and why `compact()` is required for manifest-entry relief. + +## Budget value + +New in `src/powercontext/builtin/artifacts/memory/models.py`: + +```python +class MemoryCapacityBudget(BaseModel): + """The capacity ceiling applied to one Memory Artifact.""" + + max_active_entries: int = Field(default=5_000, ge=1) + max_manifest_entries: int = Field(default=10_000, ge=1) + max_manifest_bytes: int = Field(default=4_194_304, ge=1_024) + + @model_validator(mode="after") + def validate_entry_ceiling_order(self): + if self.max_active_entries > self.max_manifest_entries: + raise ValueError("max_active_entries cannot exceed max_manifest_entries") + return self + + +class MemoryCapacity(BaseModel): + """Observed capacity of one exact Memory Revision against its budget.""" + + memory_ref: ArtifactRef + active_entry_count: int = Field(ge=0) + manifest_entry_count: int = Field(ge=0) + manifest_bytes: int = Field(ge=0) + compactable_entry_count: int = Field(ge=0) + budget: MemoryCapacityBudget + exceeded: tuple[MemoryCapacityDimension, ...] = () +``` + +with `MemoryCapacityDimension: TypeAlias = Literal["active_entries", "manifest_entries", "manifest_bytes"]`. + +### Default budgets and calibration + +The defaults are growth ceilings. Deployment-specific latency targets require representative backend measurements. + +A canonical manifest item measures **223 bytes** at the identifier widths the service generates today: `mem_ent_` plus +a 32-character hex UUID, the same again for `mem_ver_`, a 64-character hex content hash, a state, and JSON framing. +Measured against the canonical encoder, 5,000 items is 1,115,092 bytes and 10,000 items is 2,230,092 bytes. + +The byte ceiling is therefore 4 MiB, deliberately above the entry ceilings' own footprint. Setting it at 1 MiB would +make both entry ceilings unreachable — the byte ceiling would always bind first, at roughly 4,700 items — leaving two +documented dimensions that never fire. At 4 MiB and current identifier widths, `manifest_entries` binds first at about +2.13 MiB for the directory alone. `manifest_bytes` also includes audit records and reasons, so a large batch can hit +the byte ceiling even at the default identifier widths. + +The 5,000 / 10,000 / 4 MiB defaults remain configurable growth limits. Calibrate deployment budgets with isolated, +representative measurements on the selected backend, including retained-history cost. Benchmark methodology and +measurement summaries belong in `benchmark/memory_capacity/README.md`; raw run results accompany acceptance evidence. + +## Configuration + +`RuntimeConfig` in `src/powercontext/builtin/runtime/config.py`, beside the existing memory settings: + +```python +memory_max_active_entries: int = Field(default=5_000, ge=1, le=100_000) +memory_max_manifest_entries: int = Field(default=10_000, ge=1, le=200_000) +memory_max_manifest_bytes: int = Field(default=4_194_304, ge=1_024, le=67_108_864) +memory_compaction_enabled: bool = False +memory_compaction_min_tombstone_revisions: int = Field(default=10, ge=0) +memory_max_history_revisions: int = Field(default=100, ge=1) +``` + +These thread to the service exactly as `memory_rerank_candidate_limit` does today: read in +`builtin/runtime/composition.py`, carried as fields on the relational runtime in `builtin/runtime/relational.py`, and +passed to the `MemoryService` constructor. `MemoryService.__init__` gains `capacity_budget: MemoryCapacityBudget | None` +and `compaction: MemoryCompactionPolicy | None`, plus `max_history_revisions: int = 100`. `None` selects the default +budget and disabled compaction. Generic Artifact writes in `family_management.py` receive the configured capacity +budget too, so they cannot bypass deployment limits. + +## Error + +New in `src/powercontext/builtin/artifacts/memory/errors.py`: + +```python +class MemoryCapacityExceededError(MemoryLayerError, RuntimeError): + def __init__(self, dimension: str, limit: int, observed: int) -> None: + self.dimension = dimension + self.limit = limit + self.observed = observed + super().__init__(f"memory capacity budget is exceeded: {dimension} {observed} > {limit}") +``` + +It subclasses `MemoryLayerError`. The generic Artifact write path preserves this specific exception instead of +converting it to `InvalidBaseAccessRequestError`, so those writes return the same capacity conflict. Export it from +`src/powercontext/builtin/artifacts/memory/__init__.py`. + +`_map_domain_error` in `src/powercontext/server/app.py` maps it ahead of the broad `InvalidMemoryCandidateError` branch: + +```python +if isinstance(error, MemoryCapacityExceededError): + return ( + status.HTTP_409_CONFLICT, + "memory_capacity_exceeded", + "The Memory has reached its capacity budget.", + {"dimension": error.dimension, "limit": error.limit, "observed": error.observed}, + ) +``` + +409 rather than 422 or 503: the request is well-formed and the Server is healthy, but the target resource's state +rejects it and an identical retry fails identically — the same reasoning that already maps `RevisionConflictError` and +`MemoryEntryInactiveError` to 409. + +## Enforcement point + +Enforcement happens once, on the prepared manifest, in `src/powercontext/builtin/artifacts/memory/service.py`: + +- `_prepare_commit` builds `sorted_manifest` and `content` before constructing the `Memory`. The check goes directly + after `content` is built and before the `MemoryCommit` is returned. +- `_commit_existing_transition` does the same for `forget`, `reactivate`, `organize`, and `compact`. + +Both paths call one helper: + +```python +def _require_capacity( + self, base: Memory | None, content: MemoryContent, *, growth: frozenset[str], content_bytes: bytes +) -> None: + """Refuse a prepared Revision that grows a dimension past its budget.""" +``` + +`growth` names the dimensions this operation may increase; a dimension absent from `growth` is never checked, which is +how invariant 3 is enforced structurally rather than by convention: + +| Operation | `growth` | +| --- | --- | +| append / revise (`_prepare_commit`) | `active_entries`, `manifest_entries`, `manifest_bytes` | +| `reactivate` | `active_entries` | +| `forget`, `organize`, `compact` | `frozenset()` | + +A dimension is refused only when the prepared value both exceeds the limit **and** exceeds the base Revision's value for +that dimension. A Memory already over budget — after a config change that lowered a ceiling, say — therefore still +accepts a write that does not make the breach worse, and still accepts every relief operation. Dimensions are evaluated +in the fixed order `manifest_bytes`, `manifest_entries`, `active_entries`, so the reported dimension is deterministic +when more than one binds. + +Because `MemoryCreateContent` and `MemoryReplaceContent` route through `plan_remember` in +`builtin/persistence/family_management.py`, the generic Artifact write API shares enforcement, configured budgets, and +the capacity-specific error. Head compare-and-swap rejects a prepared plan whose base moved. + +The check counts the loaded manifest and reuses the canonical bytes computed for the content hash. It adds no entry +body reads or projection writes, so #1709's append guarantees are untouched. + +## `compact()` + +New on `MemoryService`: + +```python +async def compact( + self, + memory: Memory, + *, + dry_run: bool = False, + limit: int | None = None, + reason: str | None = None, +) -> MemoryCompactionResult: + """Drop qualifying inactive tombstones from the current manifest.""" +``` + +`MemoryCompactionResult` carries `memory` (the new Revision, or the unchanged base for a dry run or a no-op), +`entry_ids`, `reclaimed_bytes`, and `dry_run`. + +`reclaimed_bytes` is the signed difference `len(base_content_bytes) - len(next_content_bytes)`, including every audit +change and reason. It may be negative even when tombstones were removed; it is zero for a no-op and never represents +physical database bytes freed. A dry run writes nothing and works while compaction is disabled. A real commit requires +`enabled=True` and an unchanged head. + +### Eligibility + +A manifest item qualifies only when all hold: + +1. `state == "inactive"` in the base manifest. +2. It has been inactive for at least `min_tombstone_revisions` completed Revision advances. An entry deactivated at + Revision 2 qualifies at Revision 12 with the default 10. Only changes in that recovery window are read, without + loading entry bodies. Reactivation followed by deactivation restarts the window. Zero bypasses only the age check. +3. No Artifact tag binds it. Tag targets resolve against the latest manifest including inactive items + (`builtin/persistence/tags.py:302`), so compacting a tagged entry would break tag resolution. A new backend method + `any_tagged_entry_ids(memory)` supplies the exclusion set. The existing `tagged_entry_ids` already spans the whole + manifest rather than only active items, but it requires a `TagFilter` and answers "which entries match this filter"; + eligibility needs "which entries carry any tag at all", so it is not reusable as it stands. +4. It is not reactivated, revised, or otherwise named by the same operation. + +`limit` caps how many tombstones one Revision drops, so a Memory with a large tombstone backlog can be compacted in +bounded steps rather than one very large Revision. + +Tags are rechecked under the owning Artifact's head lock before commit. A newly tagged candidate aborts the transaction +with `CapabilityNotSupportedError("compaction-tag-conflict")`; callers can preview again against the unchanged head. + +### The Revision it writes + +Compaction produces an ordinary Revision through `_commit_existing_transition`: the manifest omits the compacted items, +and each drop records a change. Both invariants that make this safe are structural rather than promised: + +- **No projection work.** Inactive entries have no rows in `pc_memory_entry_heads` or the search index — `_commit` + derives its active-head diff from `state == "active"` items only, so a compacted item is absent from both + `previous_active` and `current_active` and appears in neither the delete set nor the upsert set. Compaction writes + zero projection rows regardless of how many tombstones it drops. +- **No body deletion.** `pc_memory_entry_versions` rows stay. The foreign keys from `pc_memory_entry_heads` are + `ondelete="RESTRICT"` and prior Revisions still reference those versions, so retaining them is required, not merely + chosen. + +### The change operation + +Compaction records `op="compact"` with `from_entry_version_id` set to the dropped version and +`to_entry_version_id=None`. A manifest that silently shed items with no change record would break RFC 0014's rule that +`changes()` is the compact delta for a Revision, and would leave no audit trail for the one operation in Memory that +removes something from the current directory. + +This extends `MemoryChangeOp`, which is public: + +- `src/powercontext/builtin/artifacts/memory/models.py` — add `"compact"` to the `MemoryChangeOp` alias. +- `openapi/powercontext.yaml` — add `compact` to `EntryChangeOperation`. +- Regenerate with `make api-generate` and verify with `make contract-test`; the generated models use + `extra="forbid"` with enum validation, so this is a versioned additive change, not a silent one. + +This enum addition affects compatibility alongside write ceilings and the history read bound. A client that enumerates +`EntryChangeOperation` exhaustively must be updated before it reads a Memory whose history contains a compaction. Because compaction is +disabled by default, no existing deployment produces the new value until an operator opts in. + +## Public read API + +`POST /v1/memory/capacity` returns the `MemoryCapacity` of the current head, reusing the Scope authorization and +Memory-identity validation of the existing `/v1/memory/entries/list` handler. This is an additive OpenAPI operation: +specify it in `openapi/powercontext.yaml`, regenerate, and add a contract test. The service-level `capacity()` is what +SDK callers use directly. The route requires `scope.read` and returns 404 if the Scope has no Memory; it never creates +a Memory or fabricates a reference to report zeros. Python HTTP callers use `PowerContextClient.get_memory_capacity()`. + +Compaction is deliberately **not** exposed over HTTP in this RFC. It is a maintenance operation whose authorization +model belongs with the broader retention policy in #1425; exposing it as an unauthenticated-by-default Server route +ahead of that design would be the wrong order. SDK and in-process runtime callers can invoke it today. + +## Revision history bounding + +This RFC does not delete or bound stored Revisions. Deleting history would break lineage, exact citations, and Handoff +verification, and physical erasure is #1425's boundary. + +It does bound one unbounded *read*. `MemoryService.revisions()` loads every Revision from 1 to the head in a loop, one +backend `get()` each, so a Memory with 1,000 Revisions issues 1,000 loads for a single call. This RFC caps that fan-out +with `max_history_revisions` (default 100) and raises `CapabilityNotSupportedError("history-window")` before loading the +history past the cap, which the existing mapping already turns into a 422 naming the capability. The cursor-based +replacement is #1657's deliverable, and this cap is the explicit bound #1656's acceptance criteria asks callers to agree +on rather than discover. + +The result is never silently truncated. At 4 MiB per Revision, 1,000 snapshots approach 4 GiB before Python object +overhead; even 100 approach 400 MiB. The limit bounds read fan-out, not process memory: relief operations and lowered +budgets can leave Revisions above the byte budget. Callers may explicitly raise it when they can afford the snapshots. +Exact Revision reads and stored history remain available after the cap is reached. + +`entries()` remains unpaginated here and is #1656's to bound. Compaction reduces its cost as a side effect, because a +compacted tombstone is no longer a version that `entries()` loads. + +## Backend behavior + +Logic is backend-neutral: it lives in the service, operates on the manifest, and reaches storage only through existing +`MemoryBackend` methods plus the one new tag query. Counts, bytes, decisions, and errors are identical across backends, +and the same parametrized conformance tests cover both. OceanBase verification requires a disposable +`POWERCONTEXT_TEST_OCEANBASE_URL`; a skipped run is not evidence of backend parity. + +The difference is physical reclamation, and it must be reported separately because the two behave differently: + +- **SQLite.** Compaction reduces future directory size and retains every historical manifest. Checkpointing and + `VACUUM` cannot reclaim pages still occupied by that history. Measurements must distinguish checkpointed bytes from + post-`VACUUM` bytes; a smaller current manifest does not imply a smaller database file. +- **OceanBase.** Any physical reclamation is asynchronous and cannot reclaim retained history. Report observed database + bytes and the observation delay without promising a decrease after logical compaction. + +## Verification + +### Behavior tests + +`tests/builtin/artifacts/memory/test_capacity.py` covers the observable capacity contract on SQLite and, when +configured, OceanBase: + +- Exact canonical byte counts, deterministic refusal for each dimension, and unchanged storage after refusal. +- Runtime budget propagation, lowered budgets, active-only reactivation checks, and over-budget deduplication. +- Compaction previews, stale-head conflicts, audit changes, retained bodies and citations, and zero projection writes. +- Tombstone age and reactivation resets, tag protection, and rollback when a candidate gains a tag before commit. +- A full Memory recovering immediately with explicit age zero while active and tagged entries remain protected. +- Signed `reclaimed_bytes` when audit reasons outweigh removed pointers, and zero bytes for no-op compaction. +- History reads succeeding at 100 Revisions, refusing at 101 before expansion, and explicit configuration overrides. + +`tests/e2e/test_memory_capacity.py` covers the HTTP and client contract, including 404 for a Scope without Memory, +409 on explicit and generic writes, and read authorization. `tests/test_api_contract.py` verifies the operation and +additive `compact` enum value. + +### Regression guard + +The #1709 tests `test_memory_append_projection_writes_do_not_grow_with_entry_history` and +`test_memory_append_leaves_untouched_projection_rows_identical` must stay green unmodified. Enforcement adds no +projection work, so any change in their statement counts means the implementation put the check in the wrong place. + +### Scale benchmark + +A new `benchmark/memory_capacity/` module, alongside the existing `locomo` benchmarks and outside `tests/` for the +reasons `benchmark/README.md` gives, records at entry counts 200, 1,000, and 5,000, and across a compaction cycle: + +entry count, manifest bytes, database bytes, mean append latency, mean final-window append latency, projection row +writes per append, and search recall behavior before and after compaction. + +This is the full envelope #1718 asks for, superseding #1709's projection-statement-count measurement rather than +repeating it. SQLite and OceanBase results are reported separately, each stating its reclamation procedure. + +## Implementation order + +Each step is independently reviewable and leaves the tree green. + +1. **Measurement, no enforcement.** `MemoryCapacityBudget`, `MemoryCapacity`, `MemoryCapacityDimension`, + `MemoryService.capacity()`, exports. Test that reported bytes equal canonical bytes. +2. **Enforcement.** `MemoryCapacityExceededError`, `_require_capacity`, the two call sites, the `growth` table, the + HTTP mapping. Tests for refusal, for nothing persisted, and for relief over budget. +3. **Configuration.** The six `RuntimeConfig` fields and the constructor threading. Test that a configured ceiling + reaches the service. +4. **Compaction.** `compact()`, eligibility including the new tag query, the `compact` change op, the OpenAPI enum + addition, `make api-generate`, `make contract-test`. +5. **Read bounding.** The `revisions()` cap and its capability error. +6. **Public read endpoint.** `POST /v1/memory/capacity`, OpenAPI, contract test. +7. **Benchmark.** `benchmark/memory_capacity/` and the recorded SQLite and OceanBase results. + +Steps 1 through 3 alone close the "no observable ceiling" half of #1718 and are worth landing before compaction. + +Validation for the whole change: `make check`, `make test`, `make contract-test` after step 4 or 6, and `make docs-test` +for this document and its Chinese translation. + +## Acceptance criteria + +- A Memory's active entry count, manifest entry count, and manifest bytes are readable through a public API, and the + reported bytes equal the canonical bytes the Revision commits to. +- A write that would cross a budget dimension raises `MemoryCapacityExceededError`, maps to a 409 naming the dimension, + limit, and observed value, and persists nothing. +- Refusal is deterministic: the same write against the same head fails identically, and which dimension is reported is + fixed when several bind. +- Capacity never blocks `forget()`, `organize()`, or enabled `compact()`. A full Memory with eligible tombstones can + recover through lifecycle operations without direct storage access; age zero permits immediate recovery while tags + remain protected. +- `compact()` deletes no entry body row, no prior Revision, and no content hash, and writes zero projection rows. +- A citation created before compaction still validates against the Revision it names afterward. +- Compaction skips tagged tombstones and tombstones below the configured minimum age, and supports dry-run. +- `changes()` reports a `compact` operation for every dropped entry. +- The #1709 incremental projection write guarantees hold unchanged, verified by its existing tests. +- Scale results record entry count, manifest bytes, database bytes, append latency, final-window append latency, + projection row writes, and post-compaction search behavior, reported separately for SQLite and OceanBase. +- Compaction is disabled by default, with a 10-Revision recovery window. Budgets default to 5,000 / 10,000 / 4 MiB and + history reads to 100 Revisions; callers can explicitly configure these limits. + +# Drawbacks + +- **A ceiling can refuse a legitimate write.** A deployment that genuinely needs more than 5,000 active entries in one + Memory now fails where it previously degraded. That is the intended trade — silent superlinear degradation is worse + than a named limit — but it is a behavior change. A deployment may need to tune the defaults to its workload. +- **Compaction is irreversible for the active surface.** A compacted entry cannot be reactivated. Dry-run, the minimum + age, the tag exclusion, and the disabled default reduce this risk. Explicit age zero trades the recovery window for + immediate capacity relief. +- **A public enum grows.** `EntryChangeOperation` gaining `compact` obliges strict clients to update. +- **Total storage remains unbounded.** `flat-v1` still duplicates the directory per Revision and retains history. + Budgets constrain growth of each Revision, with relief exempted; neither compaction nor the history read limit caps + cumulative database size. +- **Three dimensions are more than one.** Two counts plus bytes is more contract surface than a single entry cap. The + counts are what operators reason about and bytes is what storage pays, and collapsing them would lose one or the + other. + +# Rationale and alternatives + +**Why refuse rather than split?** Splitting needs a second identity mapping one logical Memory to several Artifacts. +RFC 0014 makes that a non-goal and lists routing manifests as a future possibility. A split invented here would have to +answer which Memory a search covers, which Memory a Scope resolves, and how a citation survives a split — an RFC of its +own. Refusing is the honest intermediate: deterministic, observable, testable, and forward-compatible, since a later +routing design replaces `_require_capacity`'s raise with a route decision at exactly one call site. + +**Why not delete old Revisions?** It is the largest storage win available and the one thing we must not do here. It +breaks lineage, exact citations, and Handoff verification, and it is physical erasure, which #1425 owns. + +**Why not summarize or merge entries at the ceiling?** RFC 1652 rejected body compaction for a reason worth repeating: +entry bodies are self-contained citation-bearing records, and rewriting them risks losing names, dates, and quantities. +Capacity pressure must not become a license to rewrite content. + +**Why `manifest_bytes` rather than database bytes?** Database bytes are the number an operator actually cares about, +but they are backend-specific, lag behind writes, and need `VACUUM` on SQLite to mean anything. Manifest bytes are +exact, backend-neutral, already computed in the write path, and directly proportional to the cost this RFC bounds. The +benchmark reports database bytes so the relationship between the two is measured rather than assumed. + +**Why not a dedicated capacity table?** RFC 0014 requires entry counts to be derived from the manifest rather than +persisted redundantly. A counter table would add a second source of truth that could disagree with the manifest, for no +read we cannot serve from the manifest we already load. + +**Why enforce in the service rather than the backend?** The service is where the next manifest is assembled, so it is +able to refuse before persistence writes. Both SQL adapters then inherit identical semantics, and the generic +Artifact write path inherits them too. + +**Why enable the budget by default?** A capacity contract that is off by default does not bound anything, and #1321 is +a report about a system with no bound. Since compaction — the part that mutates — stays off, the default-on piece can +refuse growth past the configured ceiling; deployments that need a higher limit can configure it explicitly. + +**Impact of not doing this.** #1321's superlinear growth stays unbounded and unobservable. A long-lived Memory keeps +degrading with no signal, no ceiling, and no supported way to reclaim tombstone space. + +# Prior art + +Immutable-log systems separate logical deletion from physical reclamation the same way: a tombstone marks the deletion, +and a later compaction pass reclaims space without rewriting history readers may still hold. LSM-tree compaction and +Git's reachability-based garbage collection both make the reclamation step explicit and asynchronous rather than +implicit in the delete. + +Within PowerContext, `organize()` already establishes that maintenance is an ordinary Revision with recorded changes +rather than an out-of-band mutation, and the Topic Memory work budget in +`src/powercontext/builtin/persistence/topic_memory_budget.py` establishes server-owned ceilings enforced before +expensive work with a stable exhaustion reason. This RFC follows both patterns: compaction is a normal Revision, and +the budget is a ceiling checked before persistence with a named reason. + +RFC 1652's retention tiers and reversible automated deactivation are the logical counterpart to this physical contract: +that RFC decides which entries should leave the active surface, this one decides when the manifest may stop carrying +them. + +# Unresolved questions + +- **Deployment calibration.** The defaults remain 5,000 / 10,000 / 4 MiB. Isolated latency measurements are still + needed to recommend tighter budgets for particular workloads. +- **Should compaction ever be automatic?** This RFC makes it explicit and operator-driven. Whether a scheduled + compaction below a headroom threshold is safe depends on the authorization and dry-run model in #1425. +- **What is the recovery path for a compacted entry?** Today: none through `reactivate()`. Whether a `restore` + operation that re-adds a retained body as a new entry is worth defining, and what identity it would take, is + deliberately left open. +- **Per-Scope budgets?** Budgets are per-deployment here. Per-Scope overrides need the Scope-level configuration story + that #1219 and #1345 own. +- **Cursor and consistency sharing.** The `revisions()` cap must agree with the cursor and pinned-Revision semantics + #1656 and #1657 settle. If they land first, this cap becomes their default page size rather than a separate limit. + +# Future possibilities + +A delta manifest format — `manifest-v2`, storing a base reference plus the changes since it, with periodic full +snapshots — is the actual fix for `flat-v1`'s per-Revision duplication, and would turn the storage growth this RFC +bounds into growth proportional to changes rather than entries. It is a persisted-format change requiring migration and +a rebuild path, so it needs its own RFC; this RFC's measured dimensions are what would demonstrate the need and verify +the result. + +Beyond that: routing manifests and automatic split, plugging into `_require_capacity`'s single decision point; scheduled +compaction under #1425's authorization model; capacity signals feeding RFC 1652's cleanup proposals, so that pressure +selects low-importance entries rather than merely refusing; and per-Scope budget overrides once Scope-level +configuration exists. diff --git a/docs/zh/development/memory-layer.md b/docs/zh/development/memory-layer.md index 0bb11b2fbb..3f7df7defa 100644 --- a/docs/zh/development/memory-layer.md +++ b/docs/zh/development/memory-layer.md @@ -70,6 +70,62 @@ revised = await memory.revise( `retire()` 将 entry 标记为 inactive,但不删除不可变 content。`changes()` 返回紧凑的 revision change。 expected revision 和 citation 保留 optimistic concurrency,调用方无需重新构造 reference。 +## 容量与墓碑压缩 + +`await runtime.memory.for_scope(scope_id).capacity()` 返回当前版本的活跃条目数、清单条目总数、精确的规范化内容 +字节数、可压缩墓碑数、预算和超限维度。直接调用 `await service.capacity(memory)` 则测量传入的精确版本。 +远程调用使用 `POST /v1/memory/capacity`,请求体为 `{"scope_id": "project-alpha"}`;Python 客户端提供 +`PowerContextClient.get_memory_capacity(GetMemoryCapacityRequest(scope_id="project-alpha"))`。 +Scope 尚无 Memory 时返回 404,查询不会创建 Memory。 + +`RuntimeConfig` 提供以下部署级默认值: + +| 配置项 | 默认值 | +| --- | --- | +| `memory_max_active_entries` | 5,000 | +| `memory_max_manifest_entries` | 10,000 | +| `memory_max_manifest_bytes` | 4,194,304 | +| `memory_compaction_enabled` | `False` | +| `memory_compaction_min_tombstone_revisions` | 10 | +| `memory_max_history_revisions` | 100 | + +容量默认值约束每个版本的增长,不保证追加延迟,也不限制数据库总大小;保留的历史清单仍会持续累积。 +部署时应结合对应后端的代表性测量调整预算。 + +活跃条目上限不得大于清单条目上限。显式写入、提取和通用 Artifact 管理共用预算。只有某维度既超过上限、又比 +基础版本更大时才拒绝写入;错误维度按字节数、清单条目数、活跃条目数的固定顺序选择。HTTP 返回 +`409 memory_capacity_exceeded`,详情包含 `dimension`、`limit` 和 `observed`,拒绝后不持久化内容。 +`manifest_bytes` 计入完整规范化版本内容,包括变更记录及其原因。 + +超限时仍可执行 `forget()` 和 `organize()`;`reactivate()` 仅检查活跃条目数增长。压缩从当前清单移除达到保留 +年龄且未绑定标签的非活跃条目。通过 `RuntimeConfig` 显式启用,或使用 `MemoryCompactionPolicy(enabled=True)` +构造 `MemoryService`,执行前先预览: + +```python +preview = await service.compact(memory, dry_run=True, limit=100) +result = await service.compact(memory, limit=100) +memory = result.memory +``` + +压缩关闭时仍可预览,预览不写入版本。年龄按已推进的版本数计算:默认保留 10 个版本时,在版本 2 停用的条目 +从版本 12 起可压缩。重新激活并再次停用会重置保留窗口。资格检查只读取这一近期窗口。无变化的维护操作不会 +推进版本;如果所有墓碑都过新,可显式配置 `memory_compaction_min_tombstone_revisions=0`,或使用 +`MemoryCompactionPolicy(enabled=True, min_tombstone_revisions=0)`,先预览再立即压缩。零年龄只跳过保留窗口, +活跃条目和带标签墓碑仍受保护。除非需要立即恢复容量,否则建议保留默认窗口,因为压缩后无法重新激活条目。 +标签会保护非活跃条目;压缩期间新增标签会使整个事务回滚并抛出 +`CapabilityNotSupportedError("compaction-tag-conflict")`,调用方可重新预览。 + +压缩保留所有条目正文、历史版本和精确引用。被移除的条目无法重新激活,也不会出现在当前清单中。 +每次移除都记录新增的 `compact` 变更类型;启用前应更新穷举变更类型的消费者。压缩仅提供进程内接口。 +`reclaimed_bytes` 是完整规范化内容的有符号字节差;审计记录或较长原因可能抵消小规模清单缩减,因此该值可能 +为负。后续版本不再携带本次压缩的变更记录。 + +`MemoryService.revisions()` 在历史超过 `memory_max_history_revisions` 时,在展开历史前抛出 +`CapabilityNotSupportedError("history-window")`,不会静默截断结果;历史内容和精确版本读取仍然保留。 +在接近 4 MiB 字节预算时,默认 100 个版本已可能包含约 400 MiB 规范内容,尚未计入对象开销。这是读取展开次数 +上限,不是内存硬上限;调低预算或执行补救操作后,版本也可能超过字节预算。只有调用方能承担完整快照开销时 +才应提高历史读取上限。这一上限不为 `entries()` 或 `changes()` 提供分页。 + ## 检索、展开与引用 SQLite 和 OceanBase 都会初始化全文索引,因此不配置 embedding model 也可以检索: diff --git a/docs/zh/rfcs/1718_memory_capacity_contract.md b/docs/zh/rfcs/1718_memory_capacity_contract.md new file mode 100644 index 0000000000..aad3c05c90 --- /dev/null +++ b/docs/zh/rfcs/1718_memory_capacity_contract.md @@ -0,0 +1,565 @@ +- 提案名称:`memory_capacity_contract` +- 起始日期:2026-09-23 +- RFC PR:[oceanbase/powercontext#1718](https://github.com/oceanbase/powercontext/pull/1718) +- 跟踪 Issue:[oceanbase/powercontext#1718](https://github.com/oceanbase/powercontext/issues/1718) +- 相关 RFC:[RFC 0014](/zh/rfcs/0014_memory_layer_design) 和 [RFC 1652](/zh/rfcs/1652_memory_quality_and_lifecycle) +- 相关工作:[#1321](https://github.com/oceanbase/powercontext/issues/1321)、 + [#1709](https://github.com/oceanbase/powercontext/pull/1709)、 + [#1656](https://github.com/oceanbase/powercontext/issues/1656)、 + [#1657](https://github.com/oceanbase/powercontext/issues/1657) 和 + [#1425](https://github.com/oceanbase/powercontext/issues/1425) + +# 摘要 + +本 RFC 定义长期存续的 Memory Artifact 的容量契约:度量什么、上限在哪里、触及上限时发生什么,以及运维如何重新 +获得余量。 + +Memory 获得三个可度量维度(活跃 entry 数、manifest entry 数、manifest 字节数)、一份覆盖这些维度的可配置预算, +以及一个在写入将越过预算时的确定性拒绝。恢复途径是显式的:`forget()` 把 entry 移出活跃检索面,新增的可选 +`compact()` 操作从**当前** manifest 中丢弃符合条件的 inactive 墓碑,同时不删除任何 entry 正文、任何历史 +Revision 和任何精确引用。 + +自动 Memory 拆分与路由**不在**本 RFC 范围内。它们需要 RFC 0014 列为后续可能性的 routing manifest,以及 RFC 0014 +明确列为非目标的第二层身份。本 RFC 转而定义触及上限时稳定、可观测的失败行为,并说明后续路由设计可以接入的接缝。 + +# 动机 + +PR #1709 消除了 #1321 报告的投影写放大:一次 append 现在只重写它实际改变的那个 entry 的行。该 PR 刻意没有定义容量 +边界,#1321 在这个缺口仍然存在的情况下被关闭。 + +遗留的问题是 Memory 没有上限。每个 Revision 都保存完整的 `flat-v1` manifest,因此追加第 *N* 条 entry 会写入包含 +*N* 项的 manifest;inactive entry 作为墓碑永久留在该 manifest 中;并且公开 API 无法告诉调用方一个 Memory 距离实际 +限制有多近,也不会在越过限制时拒绝。RFC 0014 把前半部分记录为缺点: + +> `flat-v1` 会复制目录并无限累积 inactive 墓碑,因此 manifest 成本随 entry 数量线性增长。 + +并把后半部分推迟给实测工作: + +> Memory 拆分、inactive 墓碑压缩和公开 routing manifest 的阈值将由实测结果决定。 + +RFC 1652 随后提供了**逻辑**生命周期(重要性、保留档位、可恢复的自动停用),并明确把物理部分排除在外,指向后续 +设计: + +> 物理墓碑/manifest 压缩、法律保留、外部擦除和跨 Artifact 清理不在范围内。 + +本 RFC 就是那份设计,并收窄到一个问题:单个 Memory Artifact 的容量边界。它正是 RFC 1652 所预期的"仅当盘点与评测 +证明存在存储问题时才提出独立 RFC"这一步——#1321 就是该证明。 + +我们期望的结果:运维可以通过公开 API 读取一个 Memory 的容量;失控的写入方会以具体、可操作的错误失败,而不是静默 +退化;并且运维可以通过受支持的生命周期操作和策略配置恢复容量,同时不丢失任何引用。 + +## 非目标 + +- **不做自动拆分或路由。** Memory 身份就是 Artifact ID(RFC 0014),一个 Scope 恰好解析到一个 Memory Artifact ID。 + 路由到第二个 Memory 需要 RFC 0014 排除的 routing manifest 和映射身份。本 RFC 规定拒绝行为,以及后续路由设计需要 + 满足的接口。 +- **不删除 Revision 或 entry 正文。** 旧 Revision 和旧 entry 版本既不被重写**也不被删除**。物理擦除、法律保留和 + 跨 Artifact 清理属于 #1425。 +- **不改变 `flat-v1` 的存储增长。** 每个 Revision 复制 manifest 是该格式固有的特性。本 RFC 约束并观测这种增长; + 增量 manifest 格式在后续可能性中给出草图。 +- **不引入游标分页。** #1656 负责有界 entry 列举,#1657 负责历史分页。本 RFC 约束这些 issue 依赖的**内部**读放大, + 自身不定义任何游标。 +- **不做质量或重要性打分。** 哪条 entry 值得留下是 RFC 1652 的问题。本 RFC 只做计数。 + +# 使用说明 + +## Memory 现在具有可读的容量 + +每个 Memory 都会报告自己的位置: + +```python +capacity = await memory_service.capacity(memory) + +capacity.active_entry_count # 412 活跃检索面上的 entry 数 +capacity.manifest_entry_count # 468 当前 manifest 中 active + inactive 的项数 +capacity.manifest_bytes # 104_568 本 Revision 提交的规范字节数 +capacity.compactable_entry_count # 31 当前符合 compact() 条件的墓碑数 +capacity.budget # 已配置的上限 +capacity.exceeded # () —— 没有维度超出预算 +``` + +`manifest_bytes` 不是估算值,而是 `len(memory_content_bytes(content))`,即 Revision content hash 本来就承诺的那串 +精确字节。因此运维读到的数字,就是存储层写入的数字。 + +## 越过上限是具体、可操作的失败 + +会把 Memory 推过预算的写入,在任何内容被持久化之前就被拒绝: + +``` +409 Conflict +{ + "code": "memory_capacity_exceeded", + "message": "The Memory has reached its capacity budget.", + "details": {"dimension": "manifest_bytes", "limit": 4194304, "observed": 4194527} +} +``` + +拒绝会指明触发的维度、已配置的上限,以及被拒写入本会产生的值。原样重试会以完全相同的方式失败——这是状态冲突, +不是临时性错误。 + +## 触及预算后的容量恢复 + +一个会阻断自身补救手段的容量上限会让 Memory 变成砖头。因此补救操作始终可执行,即使已超出预算: + +- `forget()` 把 entry 移出活跃检索面。始终允许。 +- `organize(mode="dedupe")` 停用完全重复的 entry。始终允许。 +- `compact()` 从当前 manifest 丢弃符合条件的墓碑。不受容量预算阻断,但仍需显式启用并满足资格条件。 + +只有会**增长**当前超限维度的操作才被拒绝:追加或修订 entry,以及活跃检索面已满时的 reactivate。 + +如果写满的 Memory 只有过新的墓碑,运维可将最小年龄设为零后预览压缩,无需人为制造 Revision。 +带标签条目仍受保护:若需保留这些标签,可能需要提高预算。容量契约不会覆盖保留决策。 + +## 压缩移除墓碑,而不是历史 + +`forget()` 会在 manifest 中保留一个 inactive 项,以便该 entry 可以被 reactivate,并保持其历史可读。这是正确的 +默认行为,同时也正是 manifest 只增不减的原因。`compact()` 是回收这部分空间的显式手段: + +```python +plan = await memory_service.compact(memory, dry_run=True) +plan.entry_ids # 符合条件的墓碑 +plan.reclaimed_bytes # 完整规范内容的有符号字节减少量 + +result = await memory_service.compact(memory) # 需要先启用压缩 +memory = result.memory +``` + +压缩**不触碰**的,正是让 Memory 可被引用的那部分。entry 正文保留在 `pc_memory_entry_versions` 中。每个历史 +Revision 保留各自的 manifest。Handoff 引用按 `ArtifactRef + entry_id + entry_version_id` 解析到**它所指名的那个 +精确 Revision**,因此压缩之前写下的引用在压缩之后仍然逐字节验证通过。 + +它确实改变的、运维在启用前必须接受的是:被压缩的 entry 已从**当前** manifest 中消失,因此 `reactivate()` 无法再 +恢复它,`list(include_inactive=True)` 不再展示它。带标签条目被排除,其标签继续正常解析。所以压缩是 +**默认关闭、先 dry-run、且对活跃检索面不可逆**的。默认保留窗口为停用后推进 10 个 Revision,给误操作留下恢复 +时间;显式将年龄设为零可立即压缩。压缩关闭时仍可预览。 + +## 默认值 + +默认预算为 5,000 条活跃 entry、10,000 个 manifest 项和 4 MiB 完整规范内容。这些上限约束每个 Revision 的增长, +不保证延迟,也不限制数据库总大小。压缩默认关闭,因为从当前清单移除条目后无法重新激活。历史读取默认上限为 +100 个 Revision,超限时明确报错,不会静默截断。 + +# 参考级说明 + +## 设计不变量 + +1. **权威内容永不被销毁。** 不删除任何 entry 正文行,不删除或重写任何历史 Revision,不改变任何 content hash。 + 压缩只决定**下一个** manifest 携带哪些项。 +2. **引用保持稳定。** 本 RFC 中的任何操作都不影响精确 Revision 引用的验证,因为它按所指名的 Revision 解析,而不是 + 按当前 head。 +3. **预算不阻断补救。** 停用、去重和已启用的压缩无论预算状态如何都可执行。 +4. **拒绝无副作用且确定。** 预算在完整准备好的下一个 manifest 上、在事务开启之前完成评估。相同输入产生相同决策。 +5. **#1709 的保证不变。** 强制检查不给 append 路径增加任何按 entry 的 I/O,压缩完全不写投影行。 +6. **后端中立语义。** 计数、字节数、决策和错误在 SQLite 与 OceanBase 上完全一致,只有物理空间回收不同。 + +## 可度量维度 + +三者都从一个精确 Revision 的 manifest 派生。均不冗余持久化,符合 RFC 0014 关于 entry 计数从 manifest 派生的规定。 + +| 维度 | 定义 | 增长于 | 缩减于 | +| --- | --- | --- | --- | +| `active_entry_count` | `state == "active"` 的 manifest 项 | add、reactivate | deactivate | +| `manifest_entry_count` | 全部 manifest 项 | add | compact | +| `manifest_bytes` | `len(memory_content_bytes(content))` | 取决于内容,包括审计变更 | 取决于内容 | + +`manifest_bytes` 是最关键的维度:它是每个 Revision 实际写入的量,也是唯一计入标识符和 hash 宽度、而非假定固定 +单条成本的维度。两个计数维度存在的理由是:它们才是运维实际推理的对象,并且 entry 数上限比字节上限更早捕获病态 +写入方。 + +`deactivate` 降低活跃条目数,但不改变清单条目数。每次操作都会替换该 Revision 的审计变更和原因,因此完整规范 +字节数的增减不完全由这两个计数决定。压缩缩小目录,但新审计记录可能使该 Revision 更大;后续 Revision 不再重复 +这些记录。因此 `reclaimed_bytes` 必须保留符号,而释放清单项容量仍需 `compact()`。 + +## 预算值 + +新增于 `src/powercontext/builtin/artifacts/memory/models.py`: + +```python +class MemoryCapacityBudget(BaseModel): + """The capacity ceiling applied to one Memory Artifact.""" + + max_active_entries: int = Field(default=5_000, ge=1) + max_manifest_entries: int = Field(default=10_000, ge=1) + max_manifest_bytes: int = Field(default=4_194_304, ge=1_024) + + @model_validator(mode="after") + def validate_entry_ceiling_order(self): + if self.max_active_entries > self.max_manifest_entries: + raise ValueError("max_active_entries cannot exceed max_manifest_entries") + return self + + +class MemoryCapacity(BaseModel): + """Observed capacity of one exact Memory Revision against its budget.""" + + memory_ref: ArtifactRef + active_entry_count: int = Field(ge=0) + manifest_entry_count: int = Field(ge=0) + manifest_bytes: int = Field(ge=0) + compactable_entry_count: int = Field(ge=0) + budget: MemoryCapacityBudget + exceeded: tuple[MemoryCapacityDimension, ...] = () +``` + +其中 `MemoryCapacityDimension: TypeAlias = Literal["active_entries", "manifest_entries", "manifest_bytes"]`。 + +### 默认预算与校准 + +默认值是增长上限。具体部署的延迟目标需要对应后端的代表性测量来校准。 + +按当前 service 生成的标识符宽度,一个规范 manifest 项实测为 **223 字节**:`mem_ent_` 加 32 字符十六进制 UUID、 +同样宽度的 `mem_ver_`、一个 64 字符十六进制 content hash、一个 state 和 JSON 框架。以规范编码器实测,5,000 项为 +1,115,092 字节,10,000 项为 2,230,092 字节。 + +因此字节上限取 4 MiB,刻意高于两个 entry 上限自身的占用。若取 1 MiB,两个 entry 上限都将永不可达——字节上限 +总会在约 4,700 项处先行触发——从而留下两个永不生效的已文档化维度。在 4 MiB 和当前标识符宽度下, +`manifest_entries` 对应的目录本身约为 2.13 MiB。`manifest_bytes` 还计入审计记录和原因,因此即使使用默认 +标识符宽度,大批量写入也可能先触发字节上限。 + +保留可配置的 5,000 / 10,000 / 4 MiB 增长上限。部署预算应通过对应后端上隔离、具有代表性的测量校准,并计入保留 +历史的成本。基准方法与测量摘要集中在 `benchmark/memory_capacity/README.md`,原始运行结果随验收证据保存。 + +## 配置 + +`src/powercontext/builtin/runtime/config.py` 的 `RuntimeConfig` 中,紧邻现有 memory 配置项: + +```python +memory_max_active_entries: int = Field(default=5_000, ge=1, le=100_000) +memory_max_manifest_entries: int = Field(default=10_000, ge=1, le=200_000) +memory_max_manifest_bytes: int = Field(default=4_194_304, ge=1_024, le=67_108_864) +memory_compaction_enabled: bool = False +memory_compaction_min_tombstone_revisions: int = Field(default=10, ge=0) +memory_max_history_revisions: int = Field(default=100, ge=1) +``` + +它们按 `memory_rerank_candidate_limit` 今天的完全相同方式向下传递:在 `builtin/runtime/composition.py` 中读取, +作为字段挂在 `builtin/runtime/relational.py` 的关系型 runtime 上,再传入 `MemoryService` 构造函数。 +`MemoryService.__init__` 新增 `capacity_budget: MemoryCapacityBudget | None` 和 +`compaction: MemoryCompactionPolicy | None`,以及 `max_history_revisions: int = 100`;`None` 表示默认预算且 +压缩关闭。`family_management.py` 的通用 Artifact 写入同样接收已配置的容量预算,不能绕过部署上限。 + +## 错误 + +新增于 `src/powercontext/builtin/artifacts/memory/errors.py`: + +```python +class MemoryCapacityExceededError(MemoryLayerError, RuntimeError): + def __init__(self, dimension: str, limit: int, observed: int) -> None: + self.dimension = dimension + self.limit = limit + self.observed = observed + super().__init__(f"memory capacity budget is exceeded: {dimension} {observed} > {limit}") +``` + +它继承 `MemoryLayerError`。通用 Artifact 写入保留这一特定异常,不将其转换为 `InvalidBaseAccessRequestError`, +从而返回相同的容量冲突。从 `src/powercontext/builtin/artifacts/memory/__init__.py` 导出。 + +`src/powercontext/server/app.py` 的 `_map_domain_error` 在宽泛的 `InvalidMemoryCandidateError` 分支之前映射它: + +```python +if isinstance(error, MemoryCapacityExceededError): + return ( + status.HTTP_409_CONFLICT, + "memory_capacity_exceeded", + "The Memory has reached its capacity budget.", + {"dimension": error.dimension, "limit": error.limit, "observed": error.observed}, + ) +``` + +选 409 而非 422 或 503:请求本身格式正确、Server 也健康,但目标资源的状态拒绝它,并且原样重试会同样失败——这与 +现有把 `RevisionConflictError` 和 `MemoryEntryInactiveError` 映射为 409 的理由一致。 + +## 强制检查点 + +强制检查在准备好的 manifest 上执行一次,位于 `src/powercontext/builtin/artifacts/memory/service.py`: + +- `_prepare_commit` 在构造 `Memory` 之前生成 `sorted_manifest` 和 `content`。检查紧接在 `content` 构造之后、 + 返回 `MemoryCommit` 之前。 +- `_commit_existing_transition` 为 `forget`、`reactivate`、`organize` 和 `compact` 做同样的事。 + +两条路径调用同一个辅助函数: + +```python +def _require_capacity( + self, base: Memory | None, content: MemoryContent, *, growth: frozenset[str], content_bytes: bytes +) -> None: + """Refuse a prepared Revision that grows a dimension past its budget.""" +``` + +`growth` 指明本操作可能增长的维度;不在 `growth` 中的维度永不被检查,这使不变量 3 由结构而非约定来保证: + +| 操作 | `growth` | +| --- | --- | +| append / revise(`_prepare_commit`) | `active_entries`、`manifest_entries`、`manifest_bytes` | +| `reactivate` | `active_entries` | +| `forget`、`organize`、`compact` | `frozenset()` | + +只有当准备值既超过上限**又**超过基准 Revision 在该维度上的值时,该维度才被拒绝。因此一个已经超出预算的 +Memory——例如在配置下调上限之后——仍然接受不会让超限变得更糟的写入,也仍然接受所有补救操作。维度按固定顺序 +`manifest_bytes`、`manifest_entries`、`active_entries` 求值,因此多个维度同时触发时上报的维度是确定的。 + +由于 `MemoryCreateContent` 和 `MemoryReplaceContent` 都经由 `builtin/persistence/family_management.py` 中的 +`plan_remember`,通用 Artifact 写入 API 共用强制检查、已配置预算和容量专用错误。head 的比较交换拒绝 base +已经移动的预备 plan。 + +检查对已加载的清单计数,并复用计算 content hash 的规范字节串。不增加 entry 正文读取或投影写入,因此 #1709 的 +append 保证不受影响。 + +## `compact()` + +`MemoryService` 新增: + +```python +async def compact( + self, + memory: Memory, + *, + dry_run: bool = False, + limit: int | None = None, + reason: str | None = None, +) -> MemoryCompactionResult: + """Drop qualifying inactive tombstones from the current manifest.""" +``` + +`MemoryCompactionResult` 携带 `memory`(新 Revision;dry-run 或无变化时为原基准)、`entry_ids`、 +`reclaimed_bytes` 和 `dry_run`。 + +`reclaimed_bytes` 是 `len(base_content_bytes) - len(next_content_bytes)`,计入每条审计变更及原因。即使移除了 +墓碑也可能为负;无变化时为零;它不代表实际释放的数据库字节。dry-run 不写入任何内容,压缩关闭时也可使用。 +真实提交要求 `enabled=True`,且 head 未发生变化。 + +### 资格条件 + +一个 manifest 项必须同时满足以下全部条件才符合条件: + +1. 在基准 manifest 中 `state == "inactive"`。 +2. 停用后已推进至少 `min_tombstone_revisions` 个 Revision。默认年龄为 10 时,版本 2 停用的条目从版本 12 起 + 符合条件。仅读取保留窗口内的变更,不加载 entry 正文。重新激活并再次停用会重置窗口。零年龄只跳过年龄检查。 +3. 没有任何 Artifact tag 绑定它。tag 目标按包含 inactive 项的最新 manifest 解析 + (`builtin/persistence/tags.py:302`),因此压缩带 tag 的 entry 会破坏 tag 解析。新增后端方法 + `any_tagged_entry_ids(memory)` 提供排除集合。现有 `tagged_entry_ids` 本身已覆盖整个 manifest 而非仅活跃项, + 但它要求传入 `TagFilter`,回答的是"哪些 entry 匹配该过滤条件";资格判定需要的是"哪些 entry 带有任意 tag", + 因此按现状不可直接复用。 +4. 未被同一次操作 reactivate、revise 或以其他方式指名。 + +`limit` 限制单个 Revision 丢弃的墓碑数量,因此墓碑积压很多的 Memory 可以分有界的多步压缩,而不是产生一个超大 +Revision。 + +提交前会在持有所属 Artifact 的 head 锁时重新检查标签。若候选条目刚被绑定标签,整个事务回滚并抛出 +`CapabilityNotSupportedError("compaction-tag-conflict")`;调用方可针对未变化的 head 重新预览。 + +### 它写入的 Revision + +压缩通过 `_commit_existing_transition` 产生一个普通 Revision:manifest 省略被压缩的项,每次丢弃记录一条变更。 +让这一点安全的两个不变量都是结构性的,而非口头承诺: + +- **无投影工作。** inactive entry 在 `pc_memory_entry_heads` 和搜索索引中没有行——`_commit` 仅从 + `state == "active"` 的项派生活跃 head 差异,因此被压缩的项既不在 `previous_active` 也不在 `current_active`, + 既不出现在删除集合也不出现在 upsert 集合。无论丢弃多少墓碑,压缩写入零投影行。 +- **不删除正文。** `pc_memory_entry_versions` 的行保留。`pc_memory_entry_heads` 上的外键是 + `ondelete="RESTRICT"`,且历史 Revision 仍然引用这些版本,因此保留它们是必需的,而不仅是一种选择。 + +### 变更操作 + +压缩记录 `op="compact"`,`from_entry_version_id` 为被丢弃的版本,`to_entry_version_id=None`。一个静默减少项数却 +不留变更记录的 manifest 会破坏 RFC 0014 关于 `changes()` 是 Revision 紧凑增量的规定,并且会让 Memory 中唯一一个 +从当前目录移除内容的操作失去审计轨迹。 + +这会扩展公开的 `MemoryChangeOp`: + +- `src/powercontext/builtin/artifacts/memory/models.py` —— 向 `MemoryChangeOp` 别名添加 `"compact"`。 +- `openapi/powercontext.yaml` —— 向 `EntryChangeOperation` 添加 `compact`。 +- 用 `make api-generate` 重新生成,并用 `make contract-test` 验证;生成的模型使用 `extra="forbid"` 加枚举校验, + 因此这是带版本的增量变更,而不是静默变更。 + +除写入上限与历史读取边界外,枚举扩展也影响兼容性。穷举 `EntryChangeOperation` 的客户端必须在读取历史中含压缩记录的 Memory +之前更新。由于压缩默认关闭,在运维主动启用之前没有任何现有部署会产生这个新值。 + +## 公开读取 API + +`POST /v1/memory/capacity` 返回当前 head 的 `MemoryCapacity`,复用现有 `/v1/memory/entries/list` 处理器的 Scope +授权与 Memory 身份校验。这是一个增量 OpenAPI 操作:在 `openapi/powercontext.yaml` 中定义、重新生成、并添加契约 +测试。SDK 调用方直接使用 service 层的 `capacity()`。该路由要求 `scope.read` 权限;Scope 尚无 Memory 时返回 404, +不会创建 Memory 或虚构引用来返回零值。Python HTTP 客户端使用 `PowerContextClient.get_memory_capacity()`。 + +压缩刻意**不**在本 RFC 中通过 HTTP 暴露。它是一个维护操作,其授权模型属于 #1425 的更广保留策略;在那份设计之前 +就把它作为默认无认证的 Server 路由暴露出去顺序是错的。SDK 与进程内 runtime 调用方现在即可调用它。 + +## Revision 历史约束 + +本 RFC 不删除也不限制已存储的 Revision。删除历史会破坏血缘、精确引用和 Handoff 验证,而物理擦除是 #1425 的边界。 + +它确实约束了一处无界**读取**。`MemoryService.revisions()` 在循环中从 1 加载到 head 的每个 Revision,每次一个后端 +`get()`,因此一个有 1,000 个 Revision 的 Memory 单次调用会发出 1,000 次加载。本 RFC 用 `max_history_revisions` +为该读放大设上限,默认 100 个 Revision;超限时在展开历史前抛出 `CapabilityNotSupportedError("history-window")`, +现有映射将其转成指明该 capability 的 422。基于游标的替代方案是 #1657 的交付物,而这个上限正是 #1656 的验收标准所要求的、让调用方事先 +约定而非自行摸索的显式边界。 + +结果不会静默截断。每个 Revision 为 4 MiB 时,1,000 个快照接近 4 GiB,100 个也接近 400 MiB,尚未计入 Python +对象开销。这是读取展开次数上限,不是进程内存上限:补救操作和调低预算可能使版本超过字节预算。只有调用方能够 +承担完整快照开销时才应显式提高上限。达到上限后,精确 Revision 读取和已存储历史仍然可用。 + +`entries()` 在此仍不分页,由 #1656 负责约束。压缩会作为副作用降低它的成本,因为被压缩的墓碑不再是 `entries()` +需要加载的版本。 + +## 后端行为 + +逻辑是后端中立的:它位于 service 层、作用于 manifest,并且只通过现有 `MemoryBackend` 方法加一个新的 tag 查询 +访问存储。相同参数化测试覆盖两种后端的计数、字节数、决策和错误。OceanBase 验证需要可清理的 +`POWERCONTEXT_TEST_OCEANBASE_URL` 数据库;跳过执行不能证明后端一致性。 + +差异在于物理回收,并且必须分别报告,因为两者行为不同: + +- **SQLite。** 压缩降低未来目录大小,但保留所有历史清单。checkpoint 和 `VACUUM` 无法回收历史仍占用的页面。 + 测量必须区分 checkpoint 后与 `VACUUM` 后的字节数;当前清单更小不代表数据库文件更小。 +- **OceanBase。** 物理回收是异步的,且不能回收保留的历史。报告数据库字节数时说明观察延迟,不承诺逻辑压缩后 + 数据库占用一定下降。 + +## 验证 + +### 行为测试 + +`tests/builtin/artifacts/memory/test_capacity.py` 在 SQLite 及已配置时的 OceanBase 上验证可观察的容量契约: + +- 精确规范字节数、各维度的确定性拒绝,以及拒绝后存储不变。 +- runtime 预算传递、下调预算、仅检查活跃数的重新激活,以及超预算去重。 +- 压缩预览、过期 head 冲突、审计变更、正文和引用保留,以及零投影写入。 +- 墓碑年龄、重新激活后的窗口重置、标签保护,以及提交前新加标签时事务回滚。 +- 满容量 Memory 通过显式零年龄立即恢复,同时保护活跃条目与带标签墓碑。 +- 审计原因超过被移除指针大小时 `reclaimed_bytes` 为负,无变化压缩返回零。 +- 历史读取在 100 个 Revision 时成功,在 101 个时于展开前拒绝,并支持显式配置覆盖。 + +`tests/e2e/test_memory_capacity.py` 覆盖 HTTP 与客户端契约,包括 Scope 尚无 Memory 时的 404、显式及通用写入的 +409 和读取权限。`tests/test_api_contract.py` 验证新增操作和 `compact` 枚举值。 + +### 回归防护 + +PR #1709 的 `test_memory_append_projection_writes_do_not_grow_with_entry_history` 和 +`test_memory_append_leaves_untouched_projection_rows_identical` 必须在不修改的前提下保持通过。强制检查不增加任何 +投影工作,因此它们的语句计数一旦变化,就说明实现把检查放错了位置。 + +### 规模基准 + +新增 `benchmark/memory_capacity/` 模块,与现有 `locomo` 基准并列、并按 `benchmark/README.md` 给出的理由置于 +`tests/` 之外,在 entry 数 200、1,000、5,000 以及一个完整压缩周期上记录: + +entry 数、manifest 字节数、数据库字节数、平均 append 延迟、末窗平均 append 延迟、每次 append 的投影行写入数, +以及压缩前后的搜索召回行为。 + +这是 #1718 所要求的完整边界,取代而非重复 #1709 的投影语句计数测量。SQLite 与 OceanBase 结果分别报告,各自说明 +其回收流程。 + +## 实施顺序 + +每一步都可独立评审,并让代码树保持通过。 + +1. **只度量,不强制。** `MemoryCapacityBudget`、`MemoryCapacity`、`MemoryCapacityDimension`、 + `MemoryService.capacity()` 和导出。测试上报字节数等于规范字节数。 +2. **强制检查。** `MemoryCapacityExceededError`、`_require_capacity`、两个调用点、`growth` 表、HTTP 映射。 + 补齐拒绝、无持久化、以及超预算时补救可执行的测试。 +3. **配置。** 六个 `RuntimeConfig` 字段与构造函数传递。测试配置的上限能到达 service。 +4. **压缩。** `compact()`、含新 tag 查询的资格判定、`compact` 变更 op、OpenAPI 枚举新增、`make api-generate`、 + `make contract-test`。 +5. **读取约束。** `revisions()` 的上限及其 capability 错误。 +6. **公开读取端点。** `POST /v1/memory/capacity`、OpenAPI、契约测试。 +7. **基准。** `benchmark/memory_capacity/` 以及记录的 SQLite 与 OceanBase 结果。 + +仅第 1 至 3 步就能闭合 #1718 中"没有可观测上限"的那一半,值得在压缩之前先合入。 + +整体变更的验证命令:`make check`、`make test`,第 4 或 6 步之后的 `make contract-test`,以及针对本文档及其中文 +翻译的 `make docs-test`。 + +## 验收标准 + +- Memory 的活跃 entry 数、manifest entry 数和 manifest 字节数可通过公开 API 读取,且上报的字节数等于该 Revision + 所承诺的规范字节数。 +- 会越过某个预算维度的写入抛出 `MemoryCapacityExceededError`,映射为指明维度、上限和观测值的 409,并且不持久化 + 任何内容。 +- 拒绝是确定的:针对同一 head 的同一写入以相同方式失败,多个维度同时触发时上报的维度固定。 +- 容量预算不阻断 `forget()`、`organize()` 和已启用的 `compact()`。存在符合条件墓碑的满容量 Memory 能通过 + 生命周期操作恢复,无需直接访问存储;零年龄允许立即恢复,但标签始终受保护。 +- `compact()` 不删除任何 entry 正文行、任何历史 Revision 和任何 content hash,且写入零投影行。 +- 压缩之前创建的引用在压缩之后仍能针对其指名的 Revision 验证通过。 +- 压缩跳过带 tag 的墓碑和低于配置最小年龄的墓碑,并支持 dry-run。 +- `changes()` 为每个被丢弃的 entry 上报 `compact` 操作。 +- #1709 的增量投影写入保证保持不变,由其现有测试验证。 +- 规模结果记录 entry 数、manifest 字节数、数据库字节数、append 延迟、末窗 append 延迟、投影行写入数和压缩后的 + 搜索行为,SQLite 与 OceanBase 分别报告。 +- 压缩默认关闭,保留窗口默认 10 个 Revision。预算默认为 5,000 / 10,000 / 4 MiB,历史读取默认 100 个 Revision; + 调用方可显式配置这些限制。 + +# 缺点 + +- **上限可能拒绝合理的写入。** 确实需要在单个 Memory 中放超过 5,000 条活跃 entry 的部署,现在会失败,而此前只是 + 退化。这是有意的取舍——静默的超线性退化比一个具名的限制更糟——但它确实是行为变更,部署可能需要按负载调整默认值。 +- **压缩对活跃检索面不可逆。** 被压缩的 entry 无法 reactivate。dry-run、最小年龄、tag 排除和默认关闭共同降低 + 风险。显式零年龄以放弃恢复窗口换取立即释放容量。 +- **一个公开枚举被扩展。** `EntryChangeOperation` 新增 `compact` 会要求严格校验的客户端更新。 +- **总存储仍然无界。** `flat-v1` 仍然按 Revision 复制目录并保留历史。预算约束每个 Revision 的增长,补救操作豁免; + 压缩和历史读取上限均不限制数据库累计大小。 +- **三个维度比一个多。** 两个计数加字节数的契约面比单一 entry 上限更大。计数是运维推理的对象,字节是存储支付的 + 代价,合并两者必然丢掉其中一方。 + +# 设计理由与替代方案 + +**为什么拒绝而不是拆分?** 拆分需要把一个逻辑 Memory 映射到多个 Artifact 的第二层身份。RFC 0014 把它列为非目标, +并把 routing manifest 列为后续可能性。在这里临时发明的拆分方案必须回答:一次搜索覆盖哪些 Memory、一个 Scope +解析到哪个 Memory、以及引用如何跨拆分存续——这本身就是一份独立 RFC。拒绝是诚实的中间状态:确定、可观测、可测试, +并且向前兼容,因为后续路由设计只需在唯一一个调用点把 `_require_capacity` 的抛出替换为路由决策。 + +**为什么不删除旧 Revision?** 那是可获得的最大存储收益,也是这里绝不能做的事。它会破坏血缘、精确引用和 Handoff +验证,而且属于物理擦除,由 #1425 负责。 + +**为什么不在上限处汇总或合并 entry?** RFC 1652 拒绝正文压缩的理由值得重述:entry 正文是自包含、承载引用的记录, +重写它们有丢失姓名、日期和数量的风险。容量压力不能成为改写内容的许可。 + +**为什么用 `manifest_bytes` 而不是数据库字节数?** 数据库字节数才是运维真正关心的量,但它依赖具体后端、滞后于 +写入,并且在 SQLite 上需要 `VACUUM` 才有意义。manifest 字节数精确、后端中立、已在写入路径中被计算,并与本 RFC +所约束的成本直接成正比。基准测试会报告数据库字节数,使二者关系是被测量的而非被假定的。 + +**为什么不用专门的容量表?** RFC 0014 要求 entry 计数从 manifest 派生而非冗余持久化。计数表会引入可能与 manifest +不一致的第二个事实来源,而它并不能服务任何我们无法从已加载的 manifest 直接得出的读取。 + +**为什么在 service 而不是后端强制?** service 是组装下一个 manifest 的层,能在持久化写入之前拒绝。 +两个 SQL 适配器由此继承完全一致的语义,通用 Artifact 写入路径也一并继承。 + +**为什么预算默认启用?** 默认关闭的容量契约约束不了任何东西,而 #1321 报告的正是一个没有任何边界的系统。 +启用预算让增长越界显式可见;确需更高上限的部署可以配置预算。压缩仍需单独启用。 + +**不做此事的影响。** #1321 的超线性增长仍然无界且不可观测。长期存续的 Memory 会持续退化,没有信号、没有上限, +也没有受支持的墓碑空间回收手段。 + +# 先例 + +不可变日志系统以同样方式区分逻辑删除与物理回收:墓碑标记删除,随后的压缩过程回收空间,而不重写读者可能仍持有的 +历史。LSM-tree 压缩和 Git 基于可达性的垃圾回收都把回收步骤做成显式且异步的,而不是隐含在删除动作里。 + +在 PowerContext 内部,`organize()` 已经确立了维护操作是带变更记录的普通 Revision、而非带外改动这一范式; +`src/powercontext/builtin/persistence/topic_memory_budget.py` 中的 Topic Memory 工作预算则确立了在昂贵工作之前 +强制服务端上限、并给出稳定耗尽原因的范式。本 RFC 同时遵循两者:压缩是普通 Revision,预算是带具名原因、在持久化前检查的 +上限。 + +RFC 1652 的保留档位与可恢复自动停用是本物理契约的逻辑对应面:那份 RFC 决定哪些 entry 应当离开活跃检索面,本 +RFC 决定 manifest 何时可以不再携带它们。 + +# 未解决问题 + +- **部署校准。** 默认值保留 5,000 / 10,000 / 4 MiB。仍需隔离的延迟测量,才能针对具体负载推荐更严格的预算。 +- **压缩是否应当自动化?** 本 RFC 让它显式、由运维驱动。低于余量阈值时执行定时压缩是否安全,取决于 #1425 的 + 授权与 dry-run 模型。 +- **被压缩 entry 的恢复路径是什么?** 目前通过 `reactivate()` 没有恢复路径。是否值得定义一个把保留的正文作为新 + entry 重新加入的 `restore` 操作、以及它采用何种身份,刻意留待后续。 +- **按 Scope 的预算?** 这里的预算按部署生效。按 Scope 覆盖需要 #1219 和 #1345 负责的 Scope 级配置能力。 +- **游标与一致性的共享决策。** `revisions()` 的上限必须与 #1656、#1657 最终确定的游标和固定 Revision 语义一致。 + 若它们先落地,本上限应变成它们的默认页大小,而不是一个独立限制。 + +# 后续可能性 + +增量 manifest 格式——`manifest-v2`,保存一个基准引用加自该基准以来的变更,并周期性写入完整快照——才是 +`flat-v1` 按 Revision 复制问题的真正解法,它会把本 RFC 所约束的存储增长从与 entry 数成正比变为与变更数成正比。 +这是需要迁移和重建路径的持久化格式变更,因此需要独立 RFC;本 RFC 的可度量维度正是证明其必要性并验证其结果的 +依据。 + +再往后:接入 `_require_capacity` 单一决策点的 routing manifest 与自动拆分;在 #1425 授权模型下的定时压缩;把 +容量信号接入 RFC 1652 的清理提案,使压力去选择低重要性的 entry 而不仅仅是拒绝写入;以及在 Scope 级配置就绪后 +按 Scope 覆盖预算。 diff --git a/integrations/codex/plugins/powercontext/hooks/bind_tools.py b/integrations/codex/plugins/powercontext/hooks/bind_tools.py index 76cec95abb..7ca7919246 100644 --- a/integrations/codex/plugins/powercontext/hooks/bind_tools.py +++ b/integrations/codex/plugins/powercontext/hooks/bind_tools.py @@ -57,6 +57,7 @@ "generate_skill", "get_artifact_candidate", "get_experience", + "get_memory_capacity", "get_memory_entry", "get_skill", "get_skill_package_manifest", diff --git a/integrations/dsh/plugins/powercontext/lib/index.js b/integrations/dsh/plugins/powercontext/lib/index.js index 468c62ac8f..6011ecf489 100644 --- a/integrations/dsh/plugins/powercontext/lib/index.js +++ b/integrations/dsh/plugins/powercontext/lib/index.js @@ -605,6 +605,17 @@ const OPERATIONS$1 = { successStatuses: [200], emptyStatuses: [] }, + get_memory_capacity: { + method: "POST", + path: "/v1/memory/capacity", + location: "body", + scopeMode: "current", + pathParameters: [], + queryParams: [], + headerParams: [], + successStatuses: [200], + emptyStatuses: [] + }, list_memory_entries: { method: "POST", path: "/v1/memory/entries/list", diff --git a/integrations/dsh/plugins/powercontext/src/operations.generated.ts b/integrations/dsh/plugins/powercontext/src/operations.generated.ts index 32e2223d57..63be154eff 100644 --- a/integrations/dsh/plugins/powercontext/src/operations.generated.ts +++ b/integrations/dsh/plugins/powercontext/src/operations.generated.ts @@ -57,6 +57,7 @@ export const OPERATIONS = { flush_memory: { method: 'POST', path: '/v1/memory/flush', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, remember_memory: { method: 'POST', path: '/v1/memory/remember', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, search_memory: { method: 'POST', path: '/v1/memory/search', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, + get_memory_capacity: { method: 'POST', path: '/v1/memory/capacity', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, list_memory_entries: { method: 'POST', path: '/v1/memory/entries/list', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, get_memory_entry: { method: 'POST', path: '/v1/memory/entries/get', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, revise_memory_entry: { method: 'POST', path: '/v1/memory/entries/revise', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, diff --git a/integrations/opencode/plugins/powercontext/src/operations.generated.ts b/integrations/opencode/plugins/powercontext/src/operations.generated.ts index 32e2223d57..63be154eff 100644 --- a/integrations/opencode/plugins/powercontext/src/operations.generated.ts +++ b/integrations/opencode/plugins/powercontext/src/operations.generated.ts @@ -57,6 +57,7 @@ export const OPERATIONS = { flush_memory: { method: 'POST', path: '/v1/memory/flush', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, remember_memory: { method: 'POST', path: '/v1/memory/remember', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, search_memory: { method: 'POST', path: '/v1/memory/search', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, + get_memory_capacity: { method: 'POST', path: '/v1/memory/capacity', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, list_memory_entries: { method: 'POST', path: '/v1/memory/entries/list', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, get_memory_entry: { method: 'POST', path: '/v1/memory/entries/get', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, revise_memory_entry: { method: 'POST', path: '/v1/memory/entries/revise', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, diff --git a/integrations/pi/plugins/powercontext/src/operations.generated.ts b/integrations/pi/plugins/powercontext/src/operations.generated.ts index 32e2223d57..63be154eff 100644 --- a/integrations/pi/plugins/powercontext/src/operations.generated.ts +++ b/integrations/pi/plugins/powercontext/src/operations.generated.ts @@ -57,6 +57,7 @@ export const OPERATIONS = { flush_memory: { method: 'POST', path: '/v1/memory/flush', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, remember_memory: { method: 'POST', path: '/v1/memory/remember', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, search_memory: { method: 'POST', path: '/v1/memory/search', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, + get_memory_capacity: { method: 'POST', path: '/v1/memory/capacity', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, list_memory_entries: { method: 'POST', path: '/v1/memory/entries/list', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, get_memory_entry: { method: 'POST', path: '/v1/memory/entries/get', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, revise_memory_entry: { method: 'POST', path: '/v1/memory/entries/revise', location: "body", scopeMode: 'current', pathParameters: [], queryParams: [], headerParams: [], successStatuses: [200], emptyStatuses: [] }, diff --git a/openapi/powercontext.yaml b/openapi/powercontext.yaml index 1461bf30ef..b61daffeb3 100644 --- a/openapi/powercontext.yaml +++ b/openapi/powercontext.yaml @@ -1555,6 +1555,44 @@ paths: $ref: "#/components/responses/Unavailable" "500": $ref: "#/components/responses/InternalError" + /v1/memory/capacity: + post: + tags: [memory] + summary: Read Memory capacity + description: >- + Measure the current Memory head against the deployment budget, including exact canonical content bytes + and the number of aged, untagged tombstones eligible for compaction. Returns 404 when no Memory exists. + operationId: get_memory_capacity + x-powercontext-access: {action: scope.read, resource: {type: scope, scope-id-from: scope_id}} + x-powercontext-scope-mode: current + requestBody: + required: true + content: + application/json: + schema: + $ref: "#/components/schemas/GetMemoryCapacityRequest" + responses: + "200": + description: Capacity of one exact current Memory Revision. + headers: + X-PowerContext-Request-ID: + $ref: "#/components/headers/RequestId" + content: + application/json: + schema: + $ref: "#/components/schemas/MemoryCapacity" + "404": + $ref: "#/components/responses/NotFound" + "401": + $ref: "#/components/responses/Unauthorized" + "403": + $ref: "#/components/responses/Forbidden" + "422": + $ref: "#/components/responses/InvalidRequest" + "503": + $ref: "#/components/responses/Unavailable" + "500": + $ref: "#/components/responses/InternalError" /v1/memory/entries/list: post: tags: [memory] @@ -8170,6 +8208,59 @@ components: properties: status: $ref: "#/components/schemas/TopicMemoryFlushStatus" + GetMemoryCapacityRequest: + type: object + additionalProperties: false + required: [scope_id] + properties: + scope_id: + type: string + minLength: 1 + maxLength: 256 + pattern: '.*\S.*' + MemoryCapacityDimension: + type: string + enum: [active_entries, manifest_entries, manifest_bytes] + MemoryCapacityBudget: + type: object + additionalProperties: false + required: [max_active_entries, max_manifest_entries, max_manifest_bytes] + properties: + max_active_entries: + type: integer + minimum: 1 + max_manifest_entries: + type: integer + minimum: 1 + max_manifest_bytes: + type: integer + minimum: 1024 + MemoryCapacity: + type: object + additionalProperties: false + required: [memory_ref, active_entry_count, manifest_entry_count, manifest_bytes, compactable_entry_count, budget, exceeded] + properties: + memory_ref: + $ref: "#/components/schemas/ArtifactReference" + active_entry_count: + type: integer + minimum: 0 + manifest_entry_count: + type: integer + minimum: 0 + manifest_bytes: + type: integer + minimum: 0 + description: Exact canonical bytes of the complete Revision content, including its change records. + compactable_entry_count: + type: integer + minimum: 0 + budget: + $ref: "#/components/schemas/MemoryCapacityBudget" + exceeded: + type: array + items: + $ref: "#/components/schemas/MemoryCapacityDimension" GetMemoryEntryRequest: type: object additionalProperties: false @@ -10038,7 +10129,7 @@ components: enum: [ready, empty] EntryChangeOperation: type: string - enum: [add, revise, deactivate, reactivate] + enum: [add, revise, deactivate, reactivate, compact] FlushStatus: type: string enum: [idle, processed] diff --git a/src/powercontext/builtin/artifacts/memory/__init__.py b/src/powercontext/builtin/artifacts/memory/__init__.py index ed1ce9e4e7..a3c28e2981 100644 --- a/src/powercontext/builtin/artifacts/memory/__init__.py +++ b/src/powercontext/builtin/artifacts/memory/__init__.py @@ -21,6 +21,7 @@ InvalidMemoryCitationError, InvalidMemoryEvidenceError, MemoryBackendConfigurationError, + MemoryCapacityExceededError, MemoryEntryInactiveError, MemoryEntryNotFoundError, MemoryLayerError, @@ -41,10 +42,15 @@ EmbeddingProfile, Memory, MemoryCapabilities, + MemoryCapacity, + MemoryCapacityBudget, + MemoryCapacityDimension, MemoryChange, MemoryChangeOp, MemoryChannelHit, MemoryCitation, + MemoryCompactionPolicy, + MemoryCompactionResult, MemoryContent, MemoryEntryInput, MemoryEntryState, @@ -116,11 +122,17 @@ "MemoryBackendConfigurationError", "MemoryCandidateRequest", "MemoryCapabilities", + "MemoryCapacity", + "MemoryCapacityBudget", + "MemoryCapacityDimension", + "MemoryCapacityExceededError", "MemoryChange", "MemoryChangeOp", "MemoryChannelHit", "MemoryCitation", "MemoryCommit", + "MemoryCompactionPolicy", + "MemoryCompactionResult", "MemoryContent", "MemoryEntryInactiveError", "MemoryEntryInput", diff --git a/src/powercontext/builtin/artifacts/memory/errors.py b/src/powercontext/builtin/artifacts/memory/errors.py index e7b61d1070..de507eb781 100644 --- a/src/powercontext/builtin/artifacts/memory/errors.py +++ b/src/powercontext/builtin/artifacts/memory/errors.py @@ -21,6 +21,14 @@ class MemoryLayerError(PowerContextError): """Base exception for Memory domain and repository failures.""" +class MemoryCapacityExceededError(MemoryLayerError, RuntimeError): + def __init__(self, dimension: str, limit: int, observed: int) -> None: + self.dimension = dimension + self.limit = limit + self.observed = observed + super().__init__(f"memory capacity budget is exceeded: {dimension} {observed} > {limit}") + + class CapabilityNotSupportedError(MemoryLayerError, RuntimeError): def __init__(self, capability: str, detail: str | None = None) -> None: self.capability = capability diff --git a/src/powercontext/builtin/artifacts/memory/models.py b/src/powercontext/builtin/artifacts/memory/models.py index 2e1f52eacb..e7b5c553a6 100644 --- a/src/powercontext/builtin/artifacts/memory/models.py +++ b/src/powercontext/builtin/artifacts/memory/models.py @@ -19,7 +19,7 @@ from dataclasses import dataclass from typing import ClassVar, Literal, TypeAlias -from pydantic import BaseModel, Field +from pydantic import BaseModel, Field, model_validator from powercontext.artifacts import Artifact, ArtifactRef from powercontext.artifacts import MemoryCitation as MemoryCitation @@ -28,7 +28,8 @@ from powercontext.sources import Source, SourceRef MemoryEntryState: TypeAlias = Literal["active", "inactive"] -MemoryChangeOp: TypeAlias = Literal["add", "revise", "deactivate", "reactivate"] +MemoryChangeOp: TypeAlias = Literal["add", "revise", "deactivate", "reactivate", "compact"] +MemoryCapacityDimension: TypeAlias = Literal["active_entries", "manifest_entries", "manifest_bytes"] MemorySearchMode: TypeAlias = Literal["fts", "vector", "hybrid", "auto"] MemoryUsedSearchMode: TypeAlias = Literal["fts", "vector", "hybrid"] MemoryMatchedBy: TypeAlias = Literal["fts", "vector"] @@ -112,6 +113,53 @@ class Memory(Artifact[MemoryContent]): family: ClassVar[str] = "memory" +class MemoryCapacityBudget(BaseModel): + """The capacity ceiling applied to one Memory Artifact.""" + + max_active_entries: int = Field(default=5_000, ge=1) + max_manifest_entries: int = Field(default=10_000, ge=1) + max_manifest_bytes: int = Field(default=4_194_304, ge=1_024) + + @model_validator(mode="after") + def validate_entry_ceiling_order(self) -> MemoryCapacityBudget: + if self.max_active_entries > self.max_manifest_entries: + raise ValueError("max_active_entries cannot exceed max_manifest_entries") # noqa: TRY003 + return self + + +class MemoryCapacity(BaseModel): + """Observed capacity of one exact Memory Revision against its budget.""" + + memory_ref: ArtifactRef + active_entry_count: int = Field(ge=0) + manifest_entry_count: int = Field(ge=0) + manifest_bytes: int = Field(ge=0) + compactable_entry_count: int = Field(ge=0) + budget: MemoryCapacityBudget + exceeded: tuple[MemoryCapacityDimension, ...] = () + + +class MemoryCompactionPolicy(BaseModel): + """Opt-in removal of inactive manifest pointers after a recovery window.""" + + enabled: bool = False + min_tombstone_revisions: int = Field( + default=10, ge=0, description="Completed Revision advances since deactivation; zero permits immediate removal." + ) + + +class MemoryCompactionResult(BaseModel): + """A compaction preview or committed Revision, retaining all historical bodies.""" + + memory: Memory + entry_ids: tuple[str, ...] = () + reclaimed_bytes: int = Field( + default=0, + description="Signed decrease in complete canonical content bytes, including compaction audit records.", + ) + dry_run: bool + + class MemoryEntryInput(BaseModel): """An untrusted proposed entry addition or content revision.""" diff --git a/src/powercontext/builtin/artifacts/memory/protocols.py b/src/powercontext/builtin/artifacts/memory/protocols.py index c79303fcb0..7efda1bcc5 100644 --- a/src/powercontext/builtin/artifacts/memory/protocols.py +++ b/src/powercontext/builtin/artifacts/memory/protocols.py @@ -111,6 +111,11 @@ async def commit(self, value: MemoryCommit, /) -> Memory: class MemoryBackend(Protocol): + async def any_tagged_entry_ids(self, memory: ArtifactRef, /) -> frozenset[str]: + """Return logical entry IDs carrying any tag in this Memory.""" + + ... + async def tagged_entry_ids(self, memory: ArtifactRef, tag_filter: TagFilter) -> frozenset[str]: """Return exact tag matches without imposing a candidate limit.""" diff --git a/src/powercontext/builtin/artifacts/memory/service.py b/src/powercontext/builtin/artifacts/memory/service.py index 84f0541e24..d9b068919b 100644 --- a/src/powercontext/builtin/artifacts/memory/service.py +++ b/src/powercontext/builtin/artifacts/memory/service.py @@ -19,6 +19,7 @@ from collections.abc import Callable, Sequence from contextlib import nullcontext from dataclasses import dataclass +from hashlib import sha256 from time import perf_counter from typing import Literal, Protocol, TypeAlias, TypeVar, overload from uuid import uuid4 @@ -31,7 +32,7 @@ embedding_content_hash, entry_content_bytes, entry_content_hash, - memory_content_hash, + memory_content_bytes, normalize_kind, normalize_query, normalize_reason, @@ -44,6 +45,7 @@ InvalidMemoryCandidateError, InvalidMemoryCitationError, InvalidMemoryEvidenceError, + MemoryCapacityExceededError, MemoryEntryInactiveError, MemoryEntryNotFoundError, ) @@ -56,8 +58,13 @@ EmbeddingProfile, Memory, MemoryCapabilities, + MemoryCapacity, + MemoryCapacityBudget, + MemoryCapacityDimension, MemoryChange, MemoryCitation, + MemoryCompactionPolicy, + MemoryCompactionResult, MemoryContent, MemoryEntryInput, MemoryEntryVersion, @@ -140,6 +147,8 @@ def __init__(self, code: str) -> None: "search-limit": "memory search limit must be positive", "search-mode": "unsupported memory search mode", "search-query": "memory search query must be non-empty text", + "compaction-limit": "memory compaction limit must be positive", + "history-limit": "memory history revision limit must be positive", } super().__init__(messages[code]) @@ -169,6 +178,9 @@ def __init__( artifact_resolver: _ArtifactResolver | None = None, id_factory: IdFactory | None = None, prompt_context: ScopedPrompts | None = None, + capacity_budget: MemoryCapacityBudget | None = None, + compaction: MemoryCompactionPolicy | None = None, + max_history_revisions: int = 100, ) -> None: self._backend = backend self._prompt_context = prompt_context @@ -181,6 +193,11 @@ def __init__( self._source_resolver = source_resolver self._artifact_resolver = artifact_resolver self._id_factory = _default_id if id_factory is None else id_factory + self._capacity_budget = MemoryCapacityBudget() if capacity_budget is None else capacity_budget.model_copy() + self._compaction = MemoryCompactionPolicy() if compaction is None else compaction.model_copy() + if max_history_revisions < 1: + raise _InvalidMemoryOperationError("history-limit") + self._max_history_revisions = max_history_revisions # One entry, describing the projections of the most recently written Memory # revision. Revisions are immutable, so a hit is always valid for that exact # reference; a rebuilt or externally advanced Memory simply misses. @@ -202,6 +219,8 @@ async def revisions(self, memory: Memory, /) -> tuple[Memory, ...]: canonical = await self.get(memory) latest = await self._backend.latest(canonical.artifact_id) + if latest.revision > self._max_history_revisions: + raise CapabilityNotSupportedError("history-window") history = [] for revision in range(1, latest.revision + 1): history.append( @@ -216,6 +235,120 @@ async def head(self, artifact_id: str, /) -> Memory: return await self._backend.latest(artifact_id) + async def capacity(self, memory: Memory, /) -> MemoryCapacity: + """Measure an exact Revision, including eligible tombstones even when compaction is disabled.""" + + canonical = await self._canonical_memory(memory) + values = self._capacity_values(canonical.content) + return MemoryCapacity( + memory_ref=canonical.as_ref(), + active_entry_count=values["active_entries"], + manifest_entry_count=values["manifest_entries"], + manifest_bytes=values["manifest_bytes"], + compactable_entry_count=len(await self._compactable_entry_ids(canonical)), + budget=self._capacity_budget.model_copy(), + exceeded=tuple(dimension for dimension, limit in self._capacity_limits() if values[dimension] > limit), + ) + + def _capacity_limits(self) -> tuple[tuple[MemoryCapacityDimension, int], ...]: + budget = self._capacity_budget + return ( + ("manifest_bytes", budget.max_manifest_bytes), + ("manifest_entries", budget.max_manifest_entries), + ("active_entries", budget.max_active_entries), + ) + + @staticmethod + def _capacity_values( + content: MemoryContent, content_bytes: bytes | None = None + ) -> dict[MemoryCapacityDimension, int]: + return { + "manifest_bytes": len(memory_content_bytes(content) if content_bytes is None else content_bytes), + "manifest_entries": len(content.manifest.entries), + "active_entries": sum(item.state == "active" for item in content.manifest.entries), + } + + def _require_capacity( + self, base: Memory | None, content: MemoryContent, *, growth: frozenset[str], content_bytes: bytes + ) -> None: + if not growth: + return + values = self._capacity_values(content, content_bytes) + previous = None + for dimension, limit in self._capacity_limits(): + observed = values[dimension] + if dimension not in growth or observed <= limit: + continue + if previous is None: + previous = {} if base is None else self._capacity_values(base.content) + if observed > previous.get(dimension, 0): + raise MemoryCapacityExceededError(dimension, limit, observed) + + async def _compactable_entry_ids(self, memory: Memory) -> tuple[str, ...]: + inactive = {item.entry_id for item in memory.content.manifest.entries if item.state == "inactive"} + if not inactive: + return () + # Only the recovery window matters. An inactive entry untouched throughout + # that window was already inactive at its lower bound. + lower = max(0, memory.revision - self._compaction.min_tombstone_revisions) + if lower < 1: + return () + recent = { + change.entry_id + for revision in await self._backend.changes(memory.as_ref(), lower) + for change in revision.changes + if change.op in {"add", "deactivate", "reactivate"} + } + tagged = await self._backend.any_tagged_entry_ids(memory.as_ref()) + return tuple(sorted(inactive - recent - tagged, key=str.encode)) + + async def compact( + self, memory: Memory, *, dry_run: bool = False, limit: int | None = None, reason: str | None = None + ) -> MemoryCompactionResult: + """Drop aged, untagged tombstones; retain every prior Revision and entry body. + + Previews are available while compaction is disabled. Reclaimed bytes are + the signed difference of complete canonical contents, including the audit + changes and reason, which can outweigh a small manifest reduction. + """ + + if limit is not None and limit < 1: + raise _InvalidMemoryOperationError("compaction-limit") + if not dry_run and not self._compaction.enabled: + raise CapabilityNotSupportedError("compaction") + normalized_reason = normalize_reason(reason) + base = await self._canonical_base(memory) + entry_ids = (await self._compactable_entry_ids(base))[:limit] + selected = frozenset(entry_ids) + manifest = {item.entry_id: item for item in base.content.manifest.entries if item.entry_id not in selected} + changes = tuple( + MemoryChange( + op="compact", + entry_id=item.entry_id, + from_entry_version_id=item.entry_version_id, + to_entry_version_id=None, + reason=normalized_reason, + ) + for item in base.content.manifest.entries + if item.entry_id in selected + ) + if not entry_ids: + return MemoryCompactionResult(memory=base, dry_run=dry_run) + content = MemoryContent(manifest=MemoryManifest(entries=tuple(manifest.values())), changes=changes) + reclaimed = len(memory_content_bytes(base.content)) - len(memory_content_bytes(content)) + result = ( + base + if dry_run + else await self._commit_existing_transition( + base=base, + manifest=manifest, + changes=changes, + current_by_entry={}, + entry_versions=(), + ) + ) + return MemoryCompactionResult(memory=result, entry_ids=entry_ids, reclaimed_bytes=reclaimed, dry_run=dry_run) + async def head_entries(self, artifact_id: str, /) -> tuple[Memory, tuple[MemoryEntryVersion, ...]]: """Return the current Memory head together with its validated entry objects. @@ -843,6 +976,7 @@ async def _set_entry_state( changes=changes, current_by_entry=current_by_entry, entry_versions=(), + growth=frozenset({"active_entries"}) if target_state == "active" else frozenset(), ) async def _commit_existing_transition( @@ -853,10 +987,13 @@ async def _commit_existing_transition( changes: Sequence[MemoryChange], current_by_entry: dict[str, MemoryEntryVersion], entry_versions: tuple[MemoryEntryVersion, ...], + growth: frozenset[str] = frozenset(), ) -> Memory: sorted_manifest = tuple(sorted(manifest.values(), key=lambda item: item.entry_id.encode("utf-8"))) sorted_changes = tuple(sorted(changes, key=lambda change: change.entry_id.encode("utf-8"))) content = MemoryContent(manifest=MemoryManifest(entries=sorted_manifest), changes=sorted_changes) + content_bytes = memory_content_bytes(content) + self._require_capacity(base, content, growth=growth, content_bytes=content_bytes) memory = Memory( artifact_id=base.artifact_id, revision=base.revision + 1, @@ -872,7 +1009,7 @@ async def _commit_existing_transition( commit = MemoryCommit( base=base, memory=memory, - content_hash=memory_content_hash(content), + content_hash=sha256(content_bytes).hexdigest(), entry_versions=entry_versions, projections=projections, ) @@ -1205,6 +1342,13 @@ async def _prepare_commit( sorted_manifest = tuple(sorted(manifest.values(), key=lambda item: item.entry_id.encode("utf-8"))) sorted_changes = tuple(sorted(changes, key=lambda change: change.entry_id.encode("utf-8"))) content = MemoryContent(manifest=MemoryManifest(entries=sorted_manifest), changes=sorted_changes) + content_bytes = memory_content_bytes(content) + self._require_capacity( + base, + content, + growth=frozenset({"active_entries", "manifest_entries", "manifest_bytes"}), + content_bytes=content_bytes, + ) memory = Memory( artifact_id=memory_id, revision=next_revision, @@ -1223,7 +1367,7 @@ async def _prepare_commit( return MemoryCommit( base=base, memory=memory, - content_hash=memory_content_hash(content), + content_hash=sha256(content_bytes).hexdigest(), entry_versions=tuple(new_versions), projections=projections, ) diff --git a/src/powercontext/builtin/persistence/artifacts.py b/src/powercontext/builtin/persistence/artifacts.py index 25fc175ad5..5c018d765a 100644 --- a/src/powercontext/builtin/persistence/artifacts.py +++ b/src/powercontext/builtin/persistence/artifacts.py @@ -20,7 +20,7 @@ from typing import Any from pydantic import BaseModel, ConfigDict, JsonValue, ValidationError -from sqlalchemy import insert, select, tuple_, update +from sqlalchemy import insert, select, true, tuple_, update from sqlalchemy.exc import IntegrityError from sqlalchemy.ext.asyncio import AsyncConnection @@ -306,6 +306,9 @@ async def revisions( family: str, artifact_id: str, /, + *, + since_revision: int = 0, + through_revision: int | None = None, ) -> tuple[Artifact[Any], ...]: """Return an artifact lifecycle in ascending revision order.""" @@ -318,6 +321,8 @@ async def revisions( ARTIFACTS_TABLE.c.scope_id == scope_id, ARTIFACTS_TABLE.c.family == family, ARTIFACTS_TABLE.c.artifact_id == artifact_id, + ARTIFACTS_TABLE.c.revision > since_revision, + true() if through_revision is None else ARTIFACTS_TABLE.c.revision <= through_revision, ) .order_by(ARTIFACTS_TABLE.c.revision) ) diff --git a/src/powercontext/builtin/persistence/family_management.py b/src/powercontext/builtin/persistence/family_management.py index 090876b78d..8d5bc5c918 100644 --- a/src/powercontext/builtin/persistence/family_management.py +++ b/src/powercontext/builtin/persistence/family_management.py @@ -30,6 +30,8 @@ from powercontext.builtin.artifacts.handoff import Handoff, HandoffContent, HandoffService, PreparedHandoff from powercontext.builtin.artifacts.memory import ( Memory, + MemoryCapacityBudget, + MemoryCapacityExceededError, MemoryEntryInput, MemoryEntryNotFoundError, MemoryLayerError, @@ -387,12 +389,14 @@ def __init__( index: MemoryIndex, embedding_model: EmbeddingModel | None, id_factory: IdFactory, + capacity_budget: MemoryCapacityBudget | None = None, ) -> None: self._database = database self._artifacts = artifacts self._index = index self._embedding_model = embedding_model self._id_factory = id_factory + self._capacity_budget = capacity_budget def artifact_id_for_create(self, generated: str, /) -> str: return generated @@ -416,6 +420,7 @@ def id_factory(kind: str) -> str: connection=connection, ), embedding_model=self._embedding_model, + capacity_budget=self._capacity_budget, id_factory=id_factory, ) @@ -460,6 +465,8 @@ async def replace( inputs.append(MemoryEntryInput(kind=item.kind, text=item.text, entry=existing)) try: plan = await service.plan_remember(memory=memory, entries=tuple(inputs), mode="append") + except MemoryCapacityExceededError: + raise except (MemoryEntryNotFoundError, MemoryLayerError) as error: raise InvalidBaseAccessRequestError("content.entries", "cannot be applied to the current Memory") from error return await _apply_memory_plan(service, plan, direct_source) diff --git a/src/powercontext/builtin/persistence/memory.py b/src/powercontext/builtin/persistence/memory.py index 6e015f5424..8ce878cc09 100644 --- a/src/powercontext/builtin/persistence/memory.py +++ b/src/powercontext/builtin/persistence/memory.py @@ -48,6 +48,7 @@ memory_content_hash, ) from powercontext.builtin.artifacts.memory.errors import ( + CapabilityNotSupportedError, InvalidMemoryCitationError, MemoryBackendConfigurationError, ) @@ -60,6 +61,7 @@ from powercontext.builtin.persistence.memory_index import MemoryIndex, NoMemoryIndex from powercontext.builtin.persistence.tables import ( ARTIFACT_HEADS_TABLE, + ARTIFACT_TAGS_TABLE, MEMORY_ENTRY_HEADS_TABLE, MEMORY_ENTRY_VERSIONS_TABLE, ) @@ -178,6 +180,23 @@ async def tagged_entry_ids(self, memory: ArtifactRef, tag_filter: TagFilter) -> ) return frozenset(rows) + async def any_tagged_entry_ids(self, memory: ArtifactRef, /) -> frozenset[str]: + await self.get(memory) + async with self._database.connection(self._bound_connection) as connection: + return await self._any_tagged_entry_ids(connection, memory) + + async def _any_tagged_entry_ids( + self, connection: AsyncConnection, memory: ArtifactRef, *, for_update: bool = False + ) -> frozenset[str]: + table = ARTIFACT_TAGS_TABLE + query = select(table.c.target_id).where( + table.c.scope_id == self._scope_id, + table.c.family == Memory.family, + table.c.artifact_id == memory.artifact_id, + table.c.target_type == "memory_entry", + ) + return frozenset(await connection.scalars(query.with_for_update() if for_update else query)) + async def entries(self, memory: ArtifactRef, /) -> tuple[MemoryEntryVersion, ...]: canonical = await self.get(memory) version_ids = tuple(item.entry_version_id for item in canonical.content.manifest.entries) @@ -297,8 +316,10 @@ async def changes( self._scope_id, Memory.family, memory.artifact_id, + since_revision=lower, + through_revision=target.revision, ) - selected = (_require_memory(value) for value in revisions if lower < value.revision <= target.revision) + selected = (_require_memory(value) for value in revisions) return tuple( MemoryRevisionChanges(memory_ref=value.as_ref(), changes=value.content.changes) for value in selected ) @@ -487,6 +508,15 @@ async def _commit(self, connection: AsyncConnection, value: MemoryCommit) -> Mem if committed != value.memory: raise _InvalidMemoryCommitError("artifact-result") + compacted = {change.entry_id for change in value.memory.content.changes if change.op == "compact"} + if compacted: + # Artifact revision CAS holds the same head lock as tag replacement. + # Recheck with a current read so a newly tagged entry rolls back the + # entire compaction instead of leaving a dangling tag. + tagged = await self._any_tagged_entry_ids(connection, value.memory.as_ref(), for_update=True) + if compacted & tagged: + raise CapabilityNotSupportedError("compaction-tag-conflict") + if value.entry_versions: await connection.execute( insert(MEMORY_ENTRY_VERSIONS_TABLE), diff --git a/src/powercontext/builtin/runtime/application.py b/src/powercontext/builtin/runtime/application.py index a734414545..757acf9914 100644 --- a/src/powercontext/builtin/runtime/application.py +++ b/src/powercontext/builtin/runtime/application.py @@ -56,6 +56,7 @@ from powercontext.builtin.artifacts.memory import ( EmbeddingProfile, Memory, + MemoryCapacity, MemoryCitation, MemoryEntryInput, MemoryEntryVersion, @@ -2458,6 +2459,15 @@ async def search(self, request: SearchMemoryRequest, /) -> MemorySearchPage: rerank=result.rerank, ) + async def capacity(self) -> MemoryCapacity: + """Read capacity of the Scope's current Memory, or raise when it does not exist.""" + + async with self._runtime._context(self.scope_id) as context: + service = context.artifacts.memory + current = await service.head(context.artifacts.memory_artifact_id) + _validate_memory_identity(context.artifacts.memory_artifact_id, current) + return await service.capacity(current) + async def list(self, *, include_inactive: bool = False, tag_filter: TagFilter | None = None) -> MemoryEntriesPage: async with self._runtime._context(self.scope_id) as context: service = context.artifacts.memory diff --git a/src/powercontext/builtin/runtime/composition.py b/src/powercontext/builtin/runtime/composition.py index 8b7b72c04c..8fe63d9089 100644 --- a/src/powercontext/builtin/runtime/composition.py +++ b/src/powercontext/builtin/runtime/composition.py @@ -39,6 +39,8 @@ CandidatePipeline, DefaultMemoryEvidenceProjector, MemoryCapabilities, + MemoryCapacityBudget, + MemoryCompactionPolicy, MemoryHit, MemoryRerankDecision, MemoryReranker, @@ -876,6 +878,16 @@ async def open_builtin_contexts( memory_reranker=memory_reranker, decision_model=decision_model, memory_rerank_candidate_limit=config.runtime.memory_rerank_candidate_limit, + memory_capacity_budget=MemoryCapacityBudget( + max_active_entries=config.runtime.memory_max_active_entries, + max_manifest_entries=config.runtime.memory_max_manifest_entries, + max_manifest_bytes=config.runtime.memory_max_manifest_bytes, + ), + memory_compaction=MemoryCompactionPolicy( + enabled=config.runtime.memory_compaction_enabled, + min_tombstone_revisions=config.runtime.memory_compaction_min_tombstone_revisions, + ), + memory_max_history_revisions=config.runtime.memory_max_history_revisions, prompt_registry=prompt_registry, prompt_demonstrators=prompt_demonstrators, handoff_verification_keys=handoff_verification_keys, @@ -934,6 +946,16 @@ async def open_builtin_contexts( memory_reranker=memory_reranker, decision_model=decision_model, memory_rerank_candidate_limit=config.runtime.memory_rerank_candidate_limit, + memory_capacity_budget=MemoryCapacityBudget( + max_active_entries=config.runtime.memory_max_active_entries, + max_manifest_entries=config.runtime.memory_max_manifest_entries, + max_manifest_bytes=config.runtime.memory_max_manifest_bytes, + ), + memory_compaction=MemoryCompactionPolicy( + enabled=config.runtime.memory_compaction_enabled, + min_tombstone_revisions=config.runtime.memory_compaction_min_tombstone_revisions, + ), + memory_max_history_revisions=config.runtime.memory_max_history_revisions, prompt_registry=prompt_registry, prompt_demonstrators=prompt_demonstrators, handoff_verification_keys=handoff_verification_keys, diff --git a/src/powercontext/builtin/runtime/config.py b/src/powercontext/builtin/runtime/config.py index 8b714bb334..0d5916851d 100644 --- a/src/powercontext/builtin/runtime/config.py +++ b/src/powercontext/builtin/runtime/config.py @@ -142,6 +142,19 @@ def reject_boolean_worker_quota(cls, value: Any) -> Any: memory_rerank_enabled: bool = False memory_rerank_candidate_limit: int = Field(default=30, ge=1, le=100) decision_assistance_enabled: bool = False + memory_max_active_entries: int = Field(default=5_000, ge=1, le=100_000) + memory_max_manifest_entries: int = Field(default=10_000, ge=1, le=200_000) + memory_max_manifest_bytes: int = Field(default=4_194_304, ge=1_024, le=67_108_864) + memory_compaction_enabled: bool = False + memory_compaction_min_tombstone_revisions: int = Field(default=10, ge=0) + memory_max_history_revisions: int = Field(default=100, ge=1) + + @model_validator(mode="after") + def validate_memory_capacity_order(self) -> RuntimeConfig: + if self.memory_max_active_entries > self.memory_max_manifest_entries: + raise ValueError("memory_max_active_entries cannot exceed memory_max_manifest_entries") # noqa: TRY003 + return self + recall_gate_enabled: bool = False recall_gate_max_rounds: int = Field(default=2, ge=0, le=2) recall_gate_min_candidates: int = Field(default=2, ge=1) diff --git a/src/powercontext/builtin/runtime/relational.py b/src/powercontext/builtin/runtime/relational.py index 0ed72c576c..1b96afc8d4 100644 --- a/src/powercontext/builtin/runtime/relational.py +++ b/src/powercontext/builtin/runtime/relational.py @@ -62,6 +62,8 @@ CandidatePipeline, EmbeddingProfile, Memory, + MemoryCapacityBudget, + MemoryCompactionPolicy, MemoryQueryEmbedding, MemoryReranker, MemoryService, @@ -299,6 +301,9 @@ class _ScopedServices: memory_reranker: MemoryReranker | None memory_rerank_candidate_limit: int decision_model: DecisionModel | None + memory_capacity_budget: MemoryCapacityBudget + memory_compaction: MemoryCompactionPolicy + memory_max_history_revisions: int id_factory: IdFactory handoff_artifact_id: str memory_artifact_id: str @@ -344,6 +349,9 @@ def memory( embedding_model=self.embedding_model, reranker=self.memory_reranker, rerank_candidate_limit=self.memory_rerank_candidate_limit, + capacity_budget=self.memory_capacity_budget, + compaction=self.memory_compaction, + max_history_revisions=self.memory_max_history_revisions, source_resolver=_RelationalMemorySourceResolver( database=self.database, scope_id=self.scope_id, @@ -514,6 +522,9 @@ def __init__( memory_reranker: MemoryReranker | None = None, decision_model: DecisionModel | None = None, memory_rerank_candidate_limit: int = 30, + memory_capacity_budget: MemoryCapacityBudget | None = None, + memory_compaction: MemoryCompactionPolicy | None = None, + memory_max_history_revisions: int = 100, id_factory: IdFactory | None = None, handoff_artifact_id: str = "handoff", memory_artifact_id: str = "memory", @@ -594,6 +605,7 @@ def __init__( PromptManagementWriter(self.repositories.artifacts, self.prompt_registry), ProfileManagementWriter(self.repositories.artifacts), MemoryManagementWriter( + capacity_budget=memory_capacity_budget, database=database, artifacts=self.repositories.artifacts, index=self.index, @@ -664,6 +676,11 @@ def __init__( self._memory_reranker = memory_reranker self._decision_model = decision_model self._memory_rerank_candidate_limit = memory_rerank_candidate_limit + self._memory_capacity_budget = ( + MemoryCapacityBudget() if memory_capacity_budget is None else memory_capacity_budget + ) + self._memory_compaction = MemoryCompactionPolicy() if memory_compaction is None else memory_compaction + self._memory_max_history_revisions = memory_max_history_revisions self._handoff_artifact_id = handoff_artifact_id self._memory_artifact_id = memory_artifact_id self._tracing = tracing @@ -1401,6 +1418,9 @@ def _services_for(self, scope_id: str) -> _ScopedServices: memory_reranker=self._memory_reranker, memory_rerank_candidate_limit=self._memory_rerank_candidate_limit, decision_model=self._decision_model, + memory_capacity_budget=self._memory_capacity_budget, + memory_compaction=self._memory_compaction, + memory_max_history_revisions=self._memory_max_history_revisions, id_factory=self._id_factory, handoff_artifact_id=self._handoff_artifact_id, memory_artifact_id=self._memory_artifact_id, diff --git a/src/powercontext/client/client.py b/src/powercontext/client/client.py index f450861bc2..4c3473f3b9 100644 --- a/src/powercontext/client/client.py +++ b/src/powercontext/client/client.py @@ -92,6 +92,7 @@ GetConnectorCheckpointRequest, GetExperienceRequest, GetHandoffReportRequest, + GetMemoryCapacityRequest, GetMemoryEntryRequest, GetSkillPackageRequest, GetSkillRequest, @@ -125,6 +126,7 @@ ListRemoteSkillTargetsResponse, ListScopesRequest, ListSourcesRequest, + MemoryCapacity, MemoryEntry, MemoryMutationResponse, PrepareContextRequest, @@ -240,6 +242,7 @@ GET_EXPERIENCE, GET_HANDOFF_REPORT, GET_LIVENESS, + GET_MEMORY_CAPACITY, GET_MEMORY_ENTRY, GET_MEMORY_ENTRY_TAGS, GET_PROFILE_POLICY, @@ -949,6 +952,11 @@ async def continue_handoff(self, request: ContinueHandoffRequest) -> HandoffReso return await self._request(CONTINUE_HANDOFF, request) + async def get_memory_capacity(self, request: GetMemoryCapacityRequest) -> MemoryCapacity: + """Read capacity of the current Memory head.""" + + return await self._request(GET_MEMORY_CAPACITY, request) + async def list_memory_entries(self, request: ListMemoryEntriesRequest) -> ListMemoryEntriesResponse: """List active entries, optionally including inactive entries for audit.""" diff --git a/src/powercontext/http/__init__.py b/src/powercontext/http/__init__.py index 1a41691998..302628a54f 100644 --- a/src/powercontext/http/__init__.py +++ b/src/powercontext/http/__init__.py @@ -166,6 +166,7 @@ GetConnectorCheckpointRequest, GetExperienceRequest, GetHandoffReportRequest, + GetMemoryCapacityRequest, GetMemoryEntryRequest, GetSkillPackageRequest, GetSkillRequest, @@ -226,6 +227,9 @@ ListSourcesRequest, LiveStateCheckStatus, ManagedSkillLibraryEntry, + MemoryCapacity, + MemoryCapacityBudget, + MemoryCapacityDimension, MemoryCitation, MemoryEntry, MemoryEntryAccessSelector, @@ -560,6 +564,7 @@ "GetConnectorCheckpointRequest", "GetExperienceRequest", "GetHandoffReportRequest", + "GetMemoryCapacityRequest", "GetMemoryEntryRequest", "GetSkillPackageRequest", "GetSkillRequest", @@ -620,6 +625,9 @@ "ListSourcesRequest", "LiveStateCheckStatus", "ManagedSkillLibraryEntry", + "MemoryCapacity", + "MemoryCapacityBudget", + "MemoryCapacityDimension", "MemoryCitation", "MemoryEntry", "MemoryEntryAccessSelector", diff --git a/src/powercontext/http/_generated/models.py b/src/powercontext/http/_generated/models.py index 5ac8ed0246..7d41c3f5af 100644 --- a/src/powercontext/http/_generated/models.py +++ b/src/powercontext/http/_generated/models.py @@ -1333,6 +1333,46 @@ class FlushTopicMemoryRequest(BaseModel): scope_id: Annotated[StrictStr, Field(max_length=256, min_length=1, pattern=".*\\S.*")] +class GetMemoryCapacityRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + ) + scope_id: Annotated[StrictStr, Field(max_length=256, min_length=1, pattern=".*\\S.*")] + + +class MemoryCapacityDimension(StrEnum): + ACTIVE_ENTRIES = "active_entries" + MANIFEST_ENTRIES = "manifest_entries" + MANIFEST_BYTES = "manifest_bytes" + + +class MemoryCapacityBudget(BaseModel): + model_config = ConfigDict( + extra="forbid", + ) + max_active_entries: Annotated[StrictInt, Field(ge=1)] + max_manifest_entries: Annotated[StrictInt, Field(ge=1)] + max_manifest_bytes: Annotated[StrictInt, Field(ge=1024)] + + +class MemoryCapacity(BaseModel): + model_config = ConfigDict( + extra="forbid", + ) + memory_ref: ArtifactReference + active_entry_count: Annotated[StrictInt, Field(ge=0)] + manifest_entry_count: Annotated[StrictInt, Field(ge=0)] + manifest_bytes: Annotated[ + StrictInt, + Field( + description="Exact canonical bytes of the complete Revision content, including its change records.", ge=0 + ), + ] + compactable_entry_count: Annotated[StrictInt, Field(ge=0)] + budget: MemoryCapacityBudget + exceeded: list[MemoryCapacityDimension] + + class GetTopicMemoryRequest(BaseModel): model_config = ConfigDict( extra="forbid", @@ -2297,6 +2337,7 @@ class EntryChangeOperation(StrEnum): REVISE = "revise" DEACTIVATE = "deactivate" REACTIVATE = "reactivate" + COMPACT = "compact" class FlushStatus(StrEnum): diff --git a/src/powercontext/http/_generated/operations.py b/src/powercontext/http/_generated/operations.py index 221caae794..0e2b558617 100644 --- a/src/powercontext/http/_generated/operations.py +++ b/src/powercontext/http/_generated/operations.py @@ -70,6 +70,7 @@ GetConnectorCheckpointRequest, GetExperienceRequest, GetHandoffReportRequest, + GetMemoryCapacityRequest, GetMemoryEntryRequest, GetSkillPackageRequest, GetSkillRequest, @@ -103,6 +104,7 @@ ListRemoteSkillTargetsResponse, ListScopesRequest, ListSourcesRequest, + MemoryCapacity, MemoryEntry, MemoryMutationResponse, PrepareContextRequest, @@ -1255,6 +1257,33 @@ class AccessRequirement(BaseModel): access=AccessRequirement(action="scope.read", resource="scope", scope_id_field="scope_id", resolver="request"), ) +GET_MEMORY_CAPACITY = Operation[GetMemoryCapacityRequest, MemoryCapacity]( + method="POST", + path="/v1/memory/capacity", + operation_id="get_memory_capacity", + request_type=GetMemoryCapacityRequest, + request_location="body", + path_parameters=(), + response_type=MemoryCapacity, + success_status=200, + summary="Read Memory capacity", + tags=("memory",), + scope_mode="current", + responses={ + 200: { + "description": "Capacity of one exact current Memory Revision.", + "headers": {"X-PowerContext-Request-ID": {"$ref": "#/components/headers/RequestId"}}, + }, + 404: {"$ref": "#/components/responses/NotFound"}, + 401: {"$ref": "#/components/responses/Unauthorized"}, + 403: {"$ref": "#/components/responses/Forbidden"}, + 422: {"$ref": "#/components/responses/InvalidRequest"}, + 503: {"$ref": "#/components/responses/Unavailable"}, + 500: {"$ref": "#/components/responses/InternalError"}, + }, + access=AccessRequirement(action="scope.read", resource="scope", scope_id_field="scope_id", resolver="request"), +) + LIST_MEMORY_ENTRIES = Operation[ListMemoryEntriesRequest, ListMemoryEntriesResponse]( method="POST", path="/v1/memory/entries/list", diff --git a/src/powercontext/http/_generated/schema.py b/src/powercontext/http/_generated/schema.py index 599f756efa..fa73aa5f62 100644 --- a/src/powercontext/http/_generated/schema.py +++ b/src/powercontext/http/_generated/schema.py @@ -1592,6 +1592,43 @@ "x-powercontext-scope-mode": "current", } }, + "/v1/memory/capacity": { + "post": { + "tags": ["memory"], + "summary": "Read Memory capacity", + "description": "Measure the current Memory head " + "against the deployment budget, " + "including exact canonical content " + "bytes and the number of aged, untagged " + "tombstones eligible for compaction. " + "Returns 404 when no Memory exists.", + "operationId": "get_memory_capacity", + "requestBody": { + "content": { + "application/json": {"schema": {"$ref": "#/components/schemas/GetMemoryCapacityRequest"}} + }, + "required": True, + }, + "responses": { + "200": { + "description": "Capacity of one exact current Memory Revision.", + "headers": {"X-PowerContext-Request-ID": {"$ref": "#/components/headers/RequestId"}}, + "content": {"application/json": {"schema": {"$ref": "#/components/schemas/MemoryCapacity"}}}, + }, + "404": {"$ref": "#/components/responses/NotFound"}, + "401": {"$ref": "#/components/responses/Unauthorized"}, + "403": {"$ref": "#/components/responses/Forbidden"}, + "422": {"$ref": "#/components/responses/InvalidRequest"}, + "503": {"$ref": "#/components/responses/Unavailable"}, + "500": {"$ref": "#/components/responses/InternalError"}, + }, + "x-powercontext-access": { + "action": "scope.read", + "resource": {"type": "scope", "scope-id-from": "scope_id"}, + }, + "x-powercontext-scope-mode": "current", + } + }, "/v1/memory/entries/list": { "post": { "tags": ["memory"], @@ -7368,6 +7405,63 @@ "type": "object", "required": ["status"], }, + "GetMemoryCapacityRequest": { + "properties": {"scope_id": {"type": "string", "maxLength": 256, "minLength": 1, "pattern": ".*\\S.*"}}, + "additionalProperties": False, + "type": "object", + "required": ["scope_id"], + }, + "MemoryCapacityDimension": { + "type": "string", + "enum": ["active_entries", "manifest_entries", "manifest_bytes"], + }, + "MemoryCapacityBudget": { + "properties": { + "max_active_entries": {"type": "integer", "minimum": 1.0}, + "max_manifest_entries": {"type": "integer", "minimum": 1.0}, + "max_manifest_bytes": {"type": "integer", "minimum": 1024.0}, + }, + "additionalProperties": False, + "type": "object", + "required": ["max_active_entries", "max_manifest_entries", "max_manifest_bytes"], + }, + "MemoryCapacity": { + "properties": { + "memory_ref": {"$ref": "#/components/schemas/ArtifactReference"}, + "active_entry_count": {"type": "integer", "minimum": 0.0}, + "manifest_entry_count": {"type": "integer", "minimum": 0.0}, + "manifest_bytes": { + "type": "integer", + "minimum": 0.0, + "description": "Exact " + "canonical " + "bytes " + "of " + "the " + "complete " + "Revision " + "content, " + "including " + "its " + "change " + "records.", + }, + "compactable_entry_count": {"type": "integer", "minimum": 0.0}, + "budget": {"$ref": "#/components/schemas/MemoryCapacityBudget"}, + "exceeded": {"items": {"$ref": "#/components/schemas/MemoryCapacityDimension"}, "type": "array"}, + }, + "additionalProperties": False, + "type": "object", + "required": [ + "memory_ref", + "active_entry_count", + "manifest_entry_count", + "manifest_bytes", + "compactable_entry_count", + "budget", + "exceeded", + ], + }, "GetMemoryEntryRequest": { "properties": { "scope_id": {"type": "string", "maxLength": 256, "minLength": 1, "pattern": ".*\\S.*"}, @@ -9283,7 +9377,10 @@ "CandidateStatus": {"type": "string", "enum": ["pending", "approved", "rejected"]}, "PreparedContextSchema": {"type": "string", "enum": ["powercontext.prepared-context.v1"]}, "PreparedContextStatus": {"type": "string", "enum": ["ready", "empty"]}, - "EntryChangeOperation": {"type": "string", "enum": ["add", "revise", "deactivate", "reactivate"]}, + "EntryChangeOperation": { + "type": "string", + "enum": ["add", "revise", "deactivate", "reactivate", "compact"], + }, "FlushStatus": {"type": "string", "enum": ["idle", "processed"]}, "TopicMemoryFlushStatus": {"type": "string", "enum": ["accepted", "idle"]}, "TopicMemoryMatchedBy": { diff --git a/src/powercontext/server/app.py b/src/powercontext/server/app.py index c8cfa7ddfd..58e5d7649a 100644 --- a/src/powercontext/server/app.py +++ b/src/powercontext/server/app.py @@ -63,9 +63,11 @@ InvalidMemoryCandidateError, InvalidMemoryCitationError, InvalidMemoryEvidenceError, + MemoryCapacityExceededError, MemoryEntryInactiveError, MemoryEntryNotFoundError, ) +from powercontext.builtin.artifacts.memory.models import MemoryCapacity as RuntimeMemoryCapacity from powercontext.builtin.artifacts.prompt import GeneratePromptDemonstrations, PromptError from powercontext.builtin.artifacts.skill import ( AgentKind, @@ -448,6 +450,7 @@ GetConnectorCheckpointRequest, GetExperienceRequest, GetHandoffReportRequest, + GetMemoryCapacityRequest, GetMemoryEntryRequest, GetSkillPackageRequest, GetSkillRequest, @@ -479,6 +482,7 @@ ListRemoteSkillTargetsResponse, ListScopesRequest, ListSourcesRequest, + MemoryCapacity, MemoryEntry, MemoryEntryAccessSelector, MemoryMutationResponse, @@ -677,6 +681,7 @@ GET_EXPERIENCE, GET_HANDOFF_REPORT, GET_LIVENESS, + GET_MEMORY_CAPACITY, GET_MEMORY_ENTRY, GET_MEMORY_ENTRY_TAGS, GET_PROFILE_POLICY, @@ -1152,6 +1157,8 @@ def for_scope(self, scope_id: str, /) -> _ScopedWorkApplication: ... class _ScopedMemoryApplication(Protocol): + async def capacity(self) -> RuntimeMemoryCapacity: ... + async def remember(self, request: RuntimeRememberMemoryRequest, /) -> MemoryMutationResult: ... async def search(self, request: RuntimeSearchMemoryRequest, /) -> MemorySearchPage: ... @@ -1425,6 +1432,7 @@ async def unexpected_error(request: Request, error: Exception) -> JSONResponse: _add_route(app, COMMIT_HANDOFF, commit_handoff) _add_route(app, CONTINUE_HANDOFF, continue_handoff) _add_route(app, LIST_MEMORY_ENTRIES, list_memory_entries) + _add_route(app, GET_MEMORY_CAPACITY, get_memory_capacity) _add_route(app, GET_MEMORY_ENTRY, get_memory_entry) _add_route(app, REVISE_MEMORY_ENTRY, revise_memory_entry) _add_route(app, RETIRE_MEMORY_ENTRY, retire_memory_entry) @@ -3085,6 +3093,14 @@ async def continue_handoff( return mapping.handoff_resolution_response(result) +async def get_memory_capacity( + request: GetMemoryCapacityRequest, + application: Annotated[ServerApplication, Depends(_require_application)], +) -> MemoryCapacity: + result = await application.memory.for_scope(request.scope_id).capacity() + return MemoryCapacity.model_validate_json(result.model_dump_json()) + + async def list_memory_entries( request: ListMemoryEntriesRequest, application: Annotated[ServerApplication, Depends(_require_application)], @@ -4318,6 +4334,7 @@ def _add_route( _COLLECTION_CONTENT_OPERATIONS = frozenset({ "search_memory", "list_memory_entries", + "get_memory_capacity", "list_memory_changes", "prepare_context", "list_managed_skills", @@ -5468,10 +5485,9 @@ def _map_domain_error(error: Exception) -> tuple[int, str, str, dict[str, Any] | return status.HTTP_404_NOT_FOUND, "artifact_not_found", "The requested Artifact was not found.", None if isinstance(error, MemoryEntryNotFoundError): return status.HTTP_404_NOT_FOUND, "memory_not_found", "The requested Memory value was not found.", None - if isinstance(error, RevisionConflictError): - return status.HTTP_409_CONFLICT, "revision_conflict", "The Memory Revision is stale.", None - if isinstance(error, MemoryEntryInactiveError): - return status.HTTP_409_CONFLICT, "memory_entry_inactive", "The Memory entry is inactive.", None + memory_conflict = _map_memory_conflict_error(error) + if memory_conflict is not None: + return memory_conflict if isinstance(error, CapabilityNotSupportedError): return ( status.HTTP_422_UNPROCESSABLE_CONTENT, @@ -5503,6 +5519,21 @@ def _map_domain_error(error: Exception) -> tuple[int, str, str, dict[str, Any] | return status.HTTP_500_INTERNAL_SERVER_ERROR, "internal_error", "The Server failed.", None +def _map_memory_conflict_error(error: Exception) -> tuple[int, str, str, dict[str, Any] | None] | None: + if isinstance(error, RevisionConflictError): + return status.HTTP_409_CONFLICT, "revision_conflict", "The Memory Revision is stale.", None + if isinstance(error, MemoryCapacityExceededError): + return ( + status.HTTP_409_CONFLICT, + "memory_capacity_exceeded", + "The Memory has reached its capacity budget.", + {"dimension": error.dimension, "limit": error.limit, "observed": error.observed}, + ) + if isinstance(error, MemoryEntryInactiveError): + return status.HTTP_409_CONFLICT, "memory_entry_inactive", "The Memory entry is inactive.", None + return None + + def _invalid_request_details(error: Exception) -> dict[str, Any] | None: if ( isinstance(error, InvalidMemoryCandidateError) diff --git a/tests/builtin/artifacts/memory/test_capacity.py b/tests/builtin/artifacts/memory/test_capacity.py new file mode 100644 index 0000000000..fb0e6fd175 --- /dev/null +++ b/tests/builtin/artifacts/memory/test_capacity.py @@ -0,0 +1,466 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import asyncio +import os +from contextlib import asynccontextmanager +from uuid import uuid4 + +import pytest +from pydantic import SecretStr, ValidationError +from sqlalchemy import event, func, select + +from powercontext.artifacts import MemoryCitation +from powercontext.builtin.artifacts.memory import ( + CapabilityNotSupportedError, + MemoryCapacityBudget, + MemoryCapacityExceededError, + MemoryCompactionPolicy, + MemoryEntryInput, + MemoryEntryNotFoundError, + MemoryService, +) +from powercontext.builtin.artifacts.memory.canonical import memory_content_bytes +from powercontext.builtin.persistence.memory import RelationalMemoryBackend +from powercontext.builtin.persistence.oceanbase import OceanBaseConfig +from powercontext.builtin.persistence.sqlite import SQLiteConfig +from powercontext.builtin.persistence.tables import ( + ARTIFACTS_TABLE, + MEMORY_ENTRY_HEADS_TABLE, + MEMORY_ENTRY_VERSIONS_TABLE, +) +from powercontext.builtin.runtime import BuiltinConfig, open_builtin_contexts +from powercontext.builtin.runtime.config import RuntimeConfig +from powercontext.builtin.tags import MemoryEntryTagTarget +from powercontext.errors import RevisionConflictError + + +@pytest.fixture(params=("sqlite", "oceanbase")) +def database_config(request): + if request.param == "sqlite": + return SQLiteConfig() + url = os.environ.get("POWERCONTEXT_TEST_OCEANBASE_URL") + if not url: + pytest.skip("requires a disposable POWERCONTEXT_TEST_OCEANBASE_URL database") + return OceanBaseConfig(url=SecretStr(url)) + + +@asynccontextmanager +async def memory_context(database_config, **settings): + config = BuiltinConfig(database=database_config, runtime=RuntimeConfig(**settings)) + async with open_builtin_contexts(config) as contexts: + scope_id = "capacity-" + uuid4().hex + context = await contexts.get(scope_id) + backend = RelationalMemoryBackend( + database=contexts.database, + scope_id=scope_id, + artifacts=contexts.repositories.artifacts, + index=contexts.index, + ) + yield contexts, scope_id, context.artifacts.memory, backend + + +def fact(number, **values): + return MemoryEntryInput(kind="fact", text=f"Capacity project fact {number}.", **values) + + +async def row_counts(contexts, scope_id): + async with contexts.database.connection() as connection: + return tuple([ + await connection.scalar(select(func.count()).select_from(table).where(table.c.scope_id == scope_id)) + for table in (ARTIFACTS_TABLE, MEMORY_ENTRY_VERSIONS_TABLE, MEMORY_ENTRY_HEADS_TABLE) + ]) + + +@pytest.mark.parametrize("dimension", ["manifest_entries", "active_entries", "manifest_bytes"]) +def test_refusal_is_deterministic_and_persists_nothing(database_config, dimension): + async def scenario(): + async with memory_context(database_config) as (contexts, scope_id, service, backend): + memory = await service.remember(memory=None, entries=(fact(1), fact(2)), mode="append") + assert memory is not None + budget = MemoryCapacityBudget( + max_active_entries=2 if dimension != "manifest_bytes" else 100, + max_manifest_entries=2 if dimension == "manifest_entries" else 100, + max_manifest_bytes=1024 if dimension == "manifest_bytes" else 4_194_304, + ) + limited = MemoryService(backend=backend, capacity_budget=budget) + before = await row_counts(contexts, scope_id) + errors = [] + for _ in range(2): + with pytest.raises(MemoryCapacityExceededError) as caught: + await limited.remember(memory=memory, entries=(fact(3, reason="x" * 512),), mode="append") + errors.append((caught.value.dimension, caught.value.limit, caught.value.observed)) + assert errors[0] == errors[1] + assert errors[0][0] == dimension + assert errors[0][2] > errors[0][1] + if dimension == "manifest_bytes": + plan = await service.plan_remember(memory=memory, entries=(fact(3, reason="x" * 512),), mode="append") + assert errors[0][2] == len(memory_content_bytes(plan.result.content)) + assert await limited.head(memory.artifact_id) == memory + assert await row_counts(contexts, scope_id) == before + capacity = await limited.capacity(memory) + assert capacity.manifest_bytes == len(memory_content_bytes(memory.content)) + assert capacity.active_entry_count == capacity.manifest_entry_count == 2 + + asyncio.run(scenario()) + + +def test_runtime_budget_and_over_budget_non_growth(database_config): + async def scenario(): + async with memory_context(database_config, memory_max_active_entries=2, memory_max_manifest_entries=2) as ( + _, + _, + service, + backend, + ): + memory = await service.remember(memory=None, entries=(fact(1), fact(2)), mode="append") + assert memory is not None + assert (await service.capacity(memory)).budget.max_manifest_entries == 2 + with pytest.raises(MemoryCapacityExceededError): + await service.remember(memory=memory, entries=(fact(3),), mode="append") + # Lower all ceilings below the existing state. A same-size revision + # is allowed; its longer audit reason is then refused by the byte cap. + entry = (await service.entries(memory))[0] + large = await service.remember( + memory=memory, entries=(fact(10, entry=entry, reason="x" * 512),), mode="append" + ) + assert large is not None + limited = MemoryService( + backend=backend, + capacity_budget=MemoryCapacityBudget( + max_active_entries=1, + max_manifest_entries=1, + max_manifest_bytes=1024, + ), + ) + assert (await limited.capacity(large)).exceeded == ("manifest_bytes", "manifest_entries", "active_entries") + current_entry = next(value for value in await service.entries(large) if value.entry_id == entry.entry_id) + smaller = await limited.remember(memory=large, entries=(fact(11, entry=current_entry),), mode="append") + assert smaller is not None + assert (await limited.capacity(smaller)).exceeded == ("manifest_entries", "active_entries") + current_entry = next(value for value in await service.entries(smaller) if value.entry_id == entry.entry_id) + with pytest.raises(MemoryCapacityExceededError, match="manifest_bytes"): + await limited.remember( + memory=smaller, entries=(fact(12, entry=current_entry, reason="x" * 512),), mode="append" + ) + + asyncio.run(scenario()) + + +def test_reactivation_checks_only_active_growth(database_config): + async def scenario(): + async with memory_context(database_config) as (_, _, service, backend): + memory = await service.remember(memory=None, entries=tuple(fact(i) for i in range(5)), mode="append") + entries = await service.entries(memory) + retired = await service.forget(memory, entries=entries[:2]) + limited = MemoryService( + backend=backend, + capacity_budget=MemoryCapacityBudget( + max_active_entries=4, + max_manifest_entries=4, + max_manifest_bytes=1024, + ), + ) + restored = await limited.reactivate(retired, entries=entries[:1]) + assert (await limited.capacity(restored)).active_entry_count == 4 + with pytest.raises(MemoryCapacityExceededError, match="active_entries"): + await limited.reactivate(restored, entries=entries[1:2]) + relieved = await limited.forget(restored, entries=entries[:1], reason="relief") + assert (await limited.capacity(relieved)).active_entry_count == 3 + + asyncio.run(scenario()) + + +def test_compaction_preserves_history_tags_and_projection_budget(database_config): + async def scenario(): + async with memory_context( + database_config, memory_compaction_enabled=True, memory_compaction_min_tombstone_revisions=1 + ) as (contexts, scope_id, service, backend): + initial = await service.remember(memory=None, entries=tuple(fact(i) for i in range(12)), mode="append") + entries = await service.entries(initial) + tagged_entry, recent_entry, *old_entries = entries + target = MemoryEntryTagTarget(artifact_id=initial.artifact_id, entry_id=tagged_entry.entry_id) + empty = await contexts.records.get_tags(scope_id, target) + tags = await contexts.records.replace_tags(scope_id, target, ("keep",), expected_etag=empty.etag) + retired = await service.forget(initial, entries=(tagged_entry, *old_entries)) + current = await service.forget(retired, entries=(recent_entry,)) + assert (await service.capacity(current)).compactable_entry_count == len(old_entries) + preview = await service.compact(current, dry_run=True, limit=3) + assert len(preview.entry_ids) == 3 + assert preview.memory == current and preview.dry_run + assert await service.head(current.artifact_id) == current + before = await row_counts(contexts, scope_id) + statements = [] + + def record(_connection, _cursor, statement, _parameters, _context, _executemany): + statements.append(statement.lower()) + + engine = contexts.database.engine.sync_engine + event.listen(engine, "before_cursor_execute", record) + try: + result = await service.compact(current, limit=3) + finally: + event.remove(engine, "before_cursor_execute", record) + assert result.entry_ids == preview.entry_ids + assert result.reclaimed_bytes == preview.reclaimed_bytes + assert result.reclaimed_bytes == len(memory_content_bytes(current.content)) - len( + memory_content_bytes(result.memory.content) + ) + assert not any( + statement.lstrip().startswith(("insert", "update", "delete")) + and any(table in statement for table in ("pc_memory_entry_heads", "pc_memory_entry_fts")) + for statement in statements + ) + after = await row_counts(contexts, scope_id) + assert after == (before[0] + 1, before[1], before[2]) + assert await service.get(initial) == initial + dropped = next(entry for entry in old_entries if entry.entry_id in result.entry_ids) + citation = MemoryCitation( + memory_ref=initial.as_ref(), entry_id=dropped.entry_id, entry_version_id=dropped.entry_version_id + ) + assert await service.validate_citation(citation) == dropped + assert dropped.entry_id not in {entry.entry_id for entry in await service.entries(result.memory)} + with pytest.raises(MemoryEntryNotFoundError): + await service.reactivate(result.memory, entries=(dropped,)) + assert await contexts.records.get_tags(scope_id, target) == tags + changes = await service.changes(result.memory) + assert {change.op for change in changes[0].changes} == {"compact"} + assert all(change.to_entry_version_id is None for change in changes[0].changes) + with pytest.raises(RevisionConflictError): + await service.compact(current) + # A lowered deployment budget cannot block any relief operation. + limited = MemoryService( + backend=backend, + capacity_budget=MemoryCapacityBudget(max_active_entries=2, max_manifest_entries=2), + compaction=MemoryCompactionPolicy(enabled=True, min_tombstone_revisions=1), + ) + compacted = await limited.compact(result.memory) + assert len(compacted.memory.content.manifest.entries) == 1 # tagged tombstone survives + appended = await limited.remember(memory=compacted.memory, entries=(fact("after relief"),), mode="append") + assert appended is not None + assert (await limited.capacity(appended)).exceeded == () + + asyncio.run(scenario()) + + +def test_compaction_defaults_age_and_reactivation_reset(database_config): + async def scenario(): + async with memory_context(database_config, memory_compaction_min_tombstone_revisions=1) as ( + _, + _, + service, + backend, + ): + initial = await service.remember(memory=None, entries=(fact(1), fact(2)), mode="append") + entry, other = await service.entries(initial) + retired = await service.forget(initial, entries=(entry,)) + assert not (await service.compact(retired, dry_run=True)).entry_ids + aged = await service.forget(retired, entries=(other,)) + assert (await service.compact(aged, dry_run=True)).entry_ids == (entry.entry_id,) + with pytest.raises(CapabilityNotSupportedError, match="compaction"): + await service.compact(aged) + restored = await service.reactivate(aged, entries=(entry,)) + retired_again = await service.forget(restored, entries=(entry,)) + preview = await service.compact(retired_again, dry_run=True) + assert entry.entry_id not in preview.entry_ids + enabled = MemoryService(backend=backend, compaction=MemoryCompactionPolicy(enabled=True)) + assert (await enabled.compact(retired_again)).memory == retired_again + with pytest.raises(ValueError, match="limit"): + await service.compact(retired_again, dry_run=True, limit=0) + + asyncio.run(scenario()) + + +def test_history_window_refuses_before_loading_history(database_config, monkeypatch): + async def scenario(): + async with memory_context(database_config, memory_max_history_revisions=2) as (_, _, service, backend): + first = await service.remember(memory=None, entries=(fact(1),), mode="append") + second = await service.remember(memory=first, entries=(fact(2),), mode="append") + assert await service.revisions(first) == (first, second) + third = await service.remember(memory=second, entries=(fact(3),), mode="append") + with pytest.raises(CapabilityNotSupportedError, match="history-window"): + await service.revisions(first) + loaded = [] + original_get = backend.get + + async def record_get(reference): + loaded.append(reference) + return await original_get(reference) + + monkeypatch.setattr(backend, "get", record_get) + bounded = MemoryService(backend=backend, max_history_revisions=2) + with pytest.raises(CapabilityNotSupportedError, match="history-window"): + await bounded.revisions(first) + assert len(loaded) <= 2 + assert await service.revision(first.as_ref()) == first + assert (await service.changes(third, since_revision=2))[0].memory_ref == third.as_ref() + + asyncio.run(scenario()) + + +def test_default_history_window_and_explicit_override(database_config): + async def scenario(): + async with memory_context(database_config) as (_, _, service, backend): + first = await service.remember(memory=None, entries=(fact(1),), mode="append") + current = first + for number in range(2, 101): + current = await service.remember(memory=current, entries=(fact(number),), mode="append") + readers = (service, MemoryService(backend=backend)) + for reader in readers: + assert len(await reader.revisions(first)) == 100 + current = await service.remember(memory=current, entries=(fact(101),), mode="append") + for reader in readers: + with pytest.raises(CapabilityNotSupportedError, match="history-window"): + await reader.revisions(first) + assert await reader.get(first) == first + expanded = MemoryService(backend=backend, max_history_revisions=101) + history = await expanded.revisions(first) + assert len(history) == 101 + assert history[0] == first and history[-1] == current + + asyncio.run(scenario()) + + +def test_zero_age_compaction_recovers_full_memory_and_preserves_tags(database_config): + async def scenario(): + async with memory_context( + database_config, + memory_max_active_entries=3, + memory_max_manifest_entries=3, + memory_compaction_enabled=True, + memory_compaction_min_tombstone_revisions=0, + ) as (contexts, scope_id, service, backend): + initial = await service.remember(memory=None, entries=(fact(1), fact(2), fact(3)), mode="append") + tagged, dropped, active = await service.entries(initial) + target = MemoryEntryTagTarget(artifact_id=initial.artifact_id, entry_id=tagged.entry_id) + empty = await contexts.records.get_tags(scope_id, target) + tags = await contexts.records.replace_tags(scope_id, target, ("keep",), expected_etag=empty.etag) + retired = await service.forget(initial, entries=(tagged, dropped)) + with pytest.raises(MemoryCapacityExceededError, match="manifest_entries"): + await service.remember(memory=retired, entries=(fact(4),), mode="append") + defaults = MemoryService(backend=backend) + assert not (await defaults.compact(retired, dry_run=True)).entry_ids + disabled = MemoryService(backend=backend, compaction=MemoryCompactionPolicy(min_tombstone_revisions=0)) + preview = await disabled.compact(retired, dry_run=True) + assert preview.entry_ids == (dropped.entry_id,) + assert await service.head(initial.artifact_id) == retired + with pytest.raises(CapabilityNotSupportedError, match="compaction"): + await disabled.compact(retired) + result = await service.compact(retired) + assert result.entry_ids == preview.entry_ids + assert {item.entry_id for item in result.memory.content.manifest.entries} == { + tagged.entry_id, + active.entry_id, + } + appended = await service.remember(memory=result.memory, entries=(fact(4),), mode="append") + assert (await service.capacity(appended)).exceeded == () + assert await contexts.records.get_tags(scope_id, target) == tags + citation = MemoryCitation( + memory_ref=initial.as_ref(), entry_id=dropped.entry_id, entry_version_id=dropped.entry_version_id + ) + assert await service.validate_citation(citation) == dropped + with pytest.raises(MemoryEntryNotFoundError): + await service.reactivate(appended, entries=(dropped,)) + + asyncio.run(scenario()) + + +def test_compaction_reports_signed_complete_content_bytes(database_config): + async def scenario(): + async with memory_context( + database_config, memory_compaction_enabled=True, memory_compaction_min_tombstone_revisions=0 + ) as (_, _, service, _): + initial = await service.remember(memory=None, entries=(fact(1),), mode="append") + retired = await service.forget(initial, entries=await service.entries(initial)) + reason = "审" * 512 + preview = await service.compact(retired, dry_run=True, reason=reason) + assert preview.reclaimed_bytes < 0 + result = await service.compact(retired, reason=reason) + assert not result.memory.content.manifest.entries + assert result.reclaimed_bytes == preview.reclaimed_bytes + assert result.reclaimed_bytes == len(memory_content_bytes(retired.content)) - len( + memory_content_bytes(result.memory.content) + ) + assert (await service.capacity(result.memory)).manifest_bytes > ( + await service.capacity(retired) + ).manifest_bytes + unchanged = await service.compact(result.memory, reason=reason) + assert unchanged.memory == result.memory + assert unchanged.reclaimed_bytes == 0 and not unchanged.entry_ids + + asyncio.run(scenario()) + + +@pytest.mark.parametrize( + "values", + [ + {"memory_max_active_entries": 2, "memory_max_manifest_entries": 1}, + {"memory_max_manifest_bytes": 1023}, + {"memory_compaction_min_tombstone_revisions": -1}, + {"memory_max_history_revisions": 0}, + ], +) +def test_capacity_configuration_rejects_invalid_limits(values): + with pytest.raises(ValidationError): + RuntimeConfig(**values) + + +def test_compaction_rechecks_tags_added_after_eligibility(database_config, monkeypatch): + async def scenario(): + async with memory_context(database_config) as (contexts, scope_id, writer, backend): + service = MemoryService( + backend=backend, compaction=MemoryCompactionPolicy(enabled=True, min_tombstone_revisions=1) + ) + initial = await writer.remember(memory=None, entries=(fact(1), fact(2)), mode="append") + entry, other = await writer.entries(initial) + retired = await writer.forget(initial, entries=(entry,)) + current = await writer.forget(retired, entries=(other,)) + target = MemoryEntryTagTarget(artifact_id=current.artifact_id, entry_id=entry.entry_id) + empty = await contexts.records.get_tags(scope_id, target) + original = backend.any_tagged_entry_ids + + async def concurrent_tag(memory): + observed = await original(memory) + await contexts.records.replace_tags(scope_id, target, ("newly protected",), expected_etag=empty.etag) + return observed + + monkeypatch.setattr(backend, "any_tagged_entry_ids", concurrent_tag) + before = await row_counts(contexts, scope_id) + with pytest.raises(CapabilityNotSupportedError, match="compaction-tag-conflict"): + await service.compact(current) + assert await service.head(current.artifact_id) == current + assert await row_counts(contexts, scope_id) == before + assert (await contexts.records.get_tags(scope_id, target)).tags == ("newly protected",) + + asyncio.run(scenario()) + + +def test_deduplication_remains_available_over_budget(database_config): + async def scenario(): + async with memory_context(database_config) as (_, _, writer, backend): + first = await writer.remember(memory=None, entries=(fact(1), fact(2)), mode="append") + entry = next(item for item in await writer.entries(first) if item.text == fact(2).text) + duplicate = await writer.remember(memory=first, entries=(fact(1, entry=entry),), mode="append") + service = MemoryService( + backend=backend, capacity_budget=MemoryCapacityBudget(max_active_entries=1, max_manifest_entries=1) + ) + deduplicated = await service.organize(duplicate, mode="dedupe") + capacity = await service.capacity(deduplicated) + assert capacity.active_entry_count == 1 + assert capacity.manifest_entry_count == 2 + assert capacity.exceeded == ("manifest_entries",) + + asyncio.run(scenario()) diff --git a/tests/e2e/test_memory_capacity.py b/tests/e2e/test_memory_capacity.py new file mode 100644 index 0000000000..853fd2bc57 --- /dev/null +++ b/tests/e2e/test_memory_capacity.py @@ -0,0 +1,122 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import asyncio + +import httpx +import pytest + +from powercontext.builtin.persistence.sqlite import SQLiteConfig +from powercontext.builtin.runtime.config import RuntimeConfig +from powercontext.client import PowerContextClient, ServerResponseError +from powercontext.http import GetMemoryCapacityRequest, RememberMemoryRequest +from powercontext.server.authentication import StaticBearerAuthenticationProvider +from powercontext.server.authz import PrincipalRef +from powercontext.server.factory import create_server_app +from powercontext.server.settings import AccessControlConfig, BearerAuthConfig, McpConfig, ServerSettings + + +def test_capacity_and_refusal_through_server_and_client(tmp_path): + async def scenario(): + app = create_server_app( + settings=ServerSettings( + database=SQLiteConfig(url=f"sqlite+aiosqlite:///{tmp_path / 'capacity.db'}"), + runtime=RuntimeConfig(memory_max_active_entries=1, memory_max_manifest_entries=1), + auth=BearerAuthConfig(enabled=False), + mcp=McpConfig(enabled=False), + ) + ) + async with ( + app.router.lifespan_context(app), + httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), + base_url="http://testserver", + ) as transport, + ): + client = PowerContextClient("http://testserver", http_client=transport, trust_transport_security=True) + scope_id = (await client.get_default_scope()).scope_id + request = GetMemoryCapacityRequest(scope_id=scope_id) + with pytest.raises(ServerResponseError) as missing: + await client.get_memory_capacity(request) + assert missing.value.status_code == 404 + written = await client.remember_memory( + RememberMemoryRequest(scope_id=scope_id, kind="fact", text="First fact.") + ) + capacity = await client.get_memory_capacity(request) + assert capacity.memory_ref == written.memory + assert capacity.active_entry_count == capacity.manifest_entry_count == 1 + assert capacity.budget.max_manifest_entries == 1 + assert capacity.exceeded == [] + rejected = await transport.post( + "/v1/memory/remember", json={"scope_id": scope_id, "kind": "fact", "text": "Second fact."} + ) + assert rejected.status_code == 409, rejected.text + error = rejected.json()["error"] + assert error["code"] == "memory_capacity_exceeded" + assert error["details"] == {"dimension": "manifest_entries", "limit": 1, "observed": 2} + assert await client.get_memory_capacity(request) == capacity + # Generic Artifact management must inherit the same deployment limit. + create = await transport.post( + f"/v1/scopes/{scope_id}/artifacts", + json={ + "family": "memory", + "content": { + "entries": [{"kind": "fact", "text": "Generic one."}, {"kind": "fact", "text": "Generic two."}] + }, + }, + ) + assert create.status_code == 409, create.text + assert create.json()["error"]["code"] == "memory_capacity_exceeded" + record = await transport.get(f"/v1/scopes/{scope_id}/artifacts/memory/{written.memory.artifact_id}") + replace = await transport.put( + f"/v1/scopes/{scope_id}/artifacts/memory/{written.memory.artifact_id}", + headers={"If-Match": record.headers["etag"]}, + json={"content": {"entries": [{"kind": "fact", "text": "Generic append."}]}}, + ) + assert replace.status_code == 409, replace.text + assert replace.json()["error"]["code"] == "memory_capacity_exceeded" + assert await client.get_memory_capacity(request) == capacity + + asyncio.run(scenario()) + + +def test_capacity_requires_scope_access(tmp_path): + async def scenario(): + app = create_server_app( + settings=ServerSettings( + database=SQLiteConfig(url=f"sqlite+aiosqlite:///{tmp_path / 'access.db'}"), + access=AccessControlConfig(mode="enforced"), + mcp=McpConfig(enabled=False), + ), + authentication_provider=StaticBearerAuthenticationProvider( + "test-token", PrincipalRef(type="user", id="outsider") + ), + ) + async with ( + app.router.lifespan_context(app), + httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), + base_url="http://testserver", + ) as client, + ): + anonymous = await client.post("/v1/memory/capacity", json={"scope_id": "private"}) + assert anonymous.status_code == 401 + denied = await client.post( + "/v1/memory/capacity", json={"scope_id": "private"}, headers={"Authorization": "Bearer test-token"} + ) + assert denied.status_code == 403 + + asyncio.run(scenario()) diff --git a/tests/test_api_contract.py b/tests/test_api_contract.py index 872ad33b47..7df86ab2f3 100644 --- a/tests/test_api_contract.py +++ b/tests/test_api_contract.py @@ -40,6 +40,7 @@ CreateMemoryArtifactRequest, CreateSourceRequest, CreateWorkContractRequest, + EntryChangeOperation, ExternalSkillResolution, FinalizeHandoffRequest, FlushTopicMemoryRequest, @@ -47,6 +48,7 @@ GeneratedCandidateResponse, GenerateExperienceRequest, GenerateSkillRequest, + GetMemoryCapacityRequest, GetMemoryEntryRequest, GetStatsRequest, GetTopicMemoryRequest, @@ -60,6 +62,7 @@ ListExternalSkillsRequest, ListExternalSkillsResponse, ListMemoryEntriesRequest, + MemoryCapacity, PrepareContextRequest, PreparedContext, PreparedHandoff, @@ -107,6 +110,7 @@ GET_ARTIFACT_CANDIDATE, GET_ARTIFACT_REVISION, GET_EXPERIENCE, + GET_MEMORY_CAPACITY, GET_MEMORY_ENTRY, GET_READINESS, GET_SKILL, @@ -974,3 +978,13 @@ def test_server_publishes_the_canonical_openapi_schema() -> None: create_server_app(settings=ServerSettings(handoff_report=HandoffReportConfig(enabled=True))).openapi() == contract ) + + +def test_memory_capacity_contract_and_compact_change_are_public(): + assert GET_MEMORY_CAPACITY.path == "/v1/memory/capacity" + assert GET_MEMORY_CAPACITY.request_type is GetMemoryCapacityRequest + assert GET_MEMORY_CAPACITY.response_type is MemoryCapacity + assert GET_MEMORY_CAPACITY.access == LIST_MEMORY_ENTRIES.access + assert EntryChangeOperation.COMPACT.value == "compact" + with pytest.raises(ValidationError): + GetMemoryCapacityRequest.model_validate({"scope_id": "scope", "budget": {}}) From e81719429f7063d13907f8c2f5284a34cf69b8c3 Mon Sep 17 00:00:00 2001 From: 222twotwotwo Date: Sun, 27 Sep 2026 16:12:54 +0800 Subject: [PATCH 2/2] fix(memory): address capacity review feedback --- docs/en/development/memory-layer.md | 28 +++++--- docs/en/rfcs/1718_memory_capacity_contract.md | 45 ++++++++---- docs/zh/development/memory-layer.md | 23 ++++-- docs/zh/rfcs/1718_memory_capacity_contract.md | 38 +++++++--- integrations/capabilities.toml | 3 + openapi/powercontext.yaml | 3 + .../builtin/artifacts/memory/service.py | 39 ++++++++--- .../builtin/runtime/application.py | 22 ++++++ src/powercontext/http/_generated/schema.py | 11 ++- src/powercontext/server/mcp.py | 3 + .../builtin/artifacts/memory/test_capacity.py | 15 ++++ tests/e2e/test_mcp_transport.py | 17 +++++ tests/e2e/test_memory_capacity.py | 70 +++++++++++++++++++ tests/test_mcp.py | 6 +- 14 files changed, 274 insertions(+), 49 deletions(-) diff --git a/docs/en/development/memory-layer.md b/docs/en/development/memory-layer.md index 55a2b7d1c4..2c0ed5d7d3 100644 --- a/docs/en/development/memory-layer.md +++ b/docs/en/development/memory-layer.md @@ -79,6 +79,10 @@ method `await service.capacity(memory)` measures the exact Revision supplied. Re `POST /v1/memory/capacity` with `{"scope_id": "project-alpha"}`, or `PowerContextClient.get_memory_capacity(GetMemoryCapacityRequest(scope_id="project-alpha"))`. A Scope without a Memory returns 404; reading capacity does not create one. +MCP exposes the same read as `get_memory_capacity`, with read-only and idempotent annotations. +Tombstone eligibility can load complete manifests across the configured recovery window (10 Revisions by default), +in addition to reading the target Revision. Read and decode cost scales with their combined size. Use this operation +for explicit capacity inspection, not frequent polling; it is not a constant-cost counter. `RuntimeConfig` supplies deployment-wide defaults: @@ -101,15 +105,20 @@ The deterministic priority is bytes, manifest entries, then active entries. HTTP `manifest_bytes` includes the complete canonical Revision content, including its changes and reasons. `forget()` and `organize()` remain available over budget. `reactivate()` checks active-entry growth only. -Compaction removes aged, untagged inactive entries from the current manifest. Enable it explicitly on `RuntimeConfig`, -or construct a `MemoryService` with `MemoryCompactionPolicy(enabled=True)`, then preview the operation: +Compaction removes aged, untagged inactive entries from the current manifest. Set +`RuntimeConfig(memory_compaction_enabled=True)` when constructing the Runtime to permit explicit in-process commits. +This flag does not schedule or automatically trigger compaction. Call the scoped Runtime entry point to preview and +commit, using the preview's Revision to reject a head that has changed: ```python -preview = await service.compact(memory, dry_run=True, limit=100) -result = await service.compact(memory, limit=100) -memory = result.memory +scoped = runtime.memory.for_scope(scope_id) +preview = await scoped.compact(dry_run=True, limit=100) +result = await scoped.compact(limit=100, expected_revision=preview.memory.revision) ``` +Direct service callers can construct `MemoryService` with `MemoryCompactionPolicy(enabled=True)` and call +`service.compact(memory, ...)` against an exact Memory Revision. No HTTP, MCP, or CLI compaction operation is provided. + A preview works while compaction is disabled and writes no Revision. Eligibility counts completed Revision advances: an entry deactivated at Revision 2 qualifies at Revision 12 with the default age of 10. Reactivation and a subsequent deactivation restart the window. Only this recent window is read. No-op maintenance does not advance the Revision; @@ -126,9 +135,12 @@ exhaustively enumerate change operations before enabling compaction. Compaction `reclaimed_bytes` is the signed difference between complete canonical contents. New audit records or a long reason can outweigh a small directory reduction; subsequent revisions no longer carry those compaction records. -`MemoryService.revisions()` refuses histories longer than `memory_max_history_revisions` with -`CapabilityNotSupportedError("history-window")` before loading them. Stored history and exact Revision reads remain -available; results are never silently truncated. The default 100 Revisions can already contain about 400 MiB of +`MemoryService.revisions(memory, since_revision=0, through_revision=None)` reads the interval +`(since_revision, through_revision]`, defaulting to the current head as the upper bound. It refuses intervals longer +than `memory_max_history_revisions` with `CapabilityNotSupportedError("history-window")` before expanding them. +For example, `through_revision=1` still reads the first Revision of a Memory with more than 100 Revisions, and +`since_revision=100, through_revision=200` reads its next 100. Results are never silently truncated. +The default 100 Revisions can already contain about 400 MiB of canonical content near the 4 MiB budget, before object overhead. This is a read fan-out bound, not a hard memory limit; lowered budgets and relief operations can leave Revisions above the byte budget. Increase the configurable history limit only when the caller can afford the complete snapshots. This bound does not paginate `entries()` or `changes()`. diff --git a/docs/en/rfcs/1718_memory_capacity_contract.md b/docs/en/rfcs/1718_memory_capacity_contract.md index 31efa66b59..bf06c19acd 100644 --- a/docs/en/rfcs/1718_memory_capacity_contract.md +++ b/docs/en/rfcs/1718_memory_capacity_contract.md @@ -305,7 +305,7 @@ Both paths call one helper: ```python def _require_capacity( - self, base: Memory | None, content: MemoryContent, *, growth: frozenset[str], content_bytes: bytes + self, base: Memory | None, content: MemoryContent, *, growth: frozenset[MemoryCapacityDimension], content_bytes: bytes ) -> None: """Refuse a prepared Revision that grows a dimension past its budget.""" ``` @@ -415,22 +415,38 @@ Memory-identity validation of the existing `/v1/memory/entries/list` handler. Th specify it in `openapi/powercontext.yaml`, regenerate, and add a contract test. The service-level `capacity()` is what SDK callers use directly. The route requires `scope.read` and returns 404 if the Scope has no Memory; it never creates a Memory or fabricates a reference to report zeros. Python HTTP callers use `PowerContextClient.get_memory_capacity()`. +The scoped Runtime exposes `capacity()`, and Server MCP exposes `get_memory_capacity` with read-only and idempotent +annotations. The `server-mcp` capability manifest classifies it as `memory_read`. + +Eligibility is not a constant-cost counter: besides reading the target Revision, it can load complete manifests for +up to `memory_compaction_min_tombstone_revisions` recent Revisions (10 by default). Read and decode cost scales with +their combined size. The operation description states this cost so callers use it for explicit inspection rather +than frequent polling. This RFC does not add a changes-only storage projection or deduplicate exact-revision reads. Compaction is deliberately **not** exposed over HTTP in this RFC. It is a maintenance operation whose authorization model belongs with the broader retention policy in #1425; exposing it as an unauthenticated-by-default Server route -ahead of that design would be the wrong order. SDK and in-process runtime callers can invoke it today. +ahead of that design would be the wrong order. In-process callers use +`runtime.memory.for_scope(scope_id).compact(dry_run=False, limit=None, reason=None, expected_revision=None)`. +The Runtime serializes scoped writes and delegates eligibility and commit rules to `MemoryService.compact()`. +`memory_compaction_enabled=True` permits these explicit commits; it does not schedule or trigger compaction. +Previews work while disabled. Passing `expected_revision=preview.memory.revision` rejects a changed head before +committing. Direct service callers pass an exact Memory Revision and a configured `MemoryCompactionPolicy`. +No HTTP, MCP, or CLI compaction operation is provided. ## Revision history bounding This RFC does not delete or bound stored Revisions. Deleting history would break lineage, exact citations, and Handoff verification, and physical erasure is #1425's boundary. -It does bound one unbounded *read*. `MemoryService.revisions()` loads every Revision from 1 to the head in a loop, one -backend `get()` each, so a Memory with 1,000 Revisions issues 1,000 loads for a single call. This RFC caps that fan-out -with `max_history_revisions` (default 100) and raises `CapabilityNotSupportedError("history-window")` before loading the -history past the cap, which the existing mapping already turns into a 422 naming the capability. The cursor-based -replacement is #1657's deliverable, and this cap is the explicit bound #1656's acceptance criteria asks callers to agree -on rather than discover. +`MemoryService.revisions(memory, since_revision=0, through_revision=None)` reads the interval +`(since_revision, through_revision]` in ascending order, defaulting to the current head as the upper bound. +`since_revision` must be nonnegative, `through_revision` must be between 1 and the current head, and the lower bound +must not exceed the upper bound. Equal bounds return an empty tuple. The expansion count is +`through_revision - since_revision`; it must not exceed `max_history_revisions` (default 100). Oversized requests raise +`CapabilityNotSupportedError("history-window")` before expanding history. Thus a Memory with 1,000 Revisions remains +readable in explicit bounded intervals without raising the deployment limit: `through_revision=1` reads its first +Revision, and `since_revision=900` reads its last 100. Calling without bounds still requests the whole visible history. +The cursor-based replacement remains #1657's deliverable; this API defines no cursor. The result is never silently truncated. At 4 MiB per Revision, 1,000 snapshots approach 4 GiB before Python object overhead; even 100 approach 400 MiB. The limit bounds read fan-out, not process memory: relief operations and lowered @@ -468,11 +484,14 @@ configured, OceanBase: - Tombstone age and reactivation resets, tag protection, and rollback when a candidate gains a tag before commit. - A full Memory recovering immediately with explicit age zero while active and tagged entries remain protected. - Signed `reclaimed_bytes` when audit reasons outweigh removed pointers, and zero bytes for no-op compaction. -- History reads succeeding at 100 Revisions, refusing at 101 before expansion, and explicit configuration overrides. +- History intervals succeeding at 100 Revisions, refusing at 101 before expansion, bounded reads of longer histories, + invalid interval rejection, and explicit configuration overrides. `tests/e2e/test_memory_capacity.py` covers the HTTP and client contract, including 404 for a Scope without Memory, -409 on explicit and generic writes, and read authorization. `tests/test_api_contract.py` verifies the operation and -additive `compact` enum value. +409 on explicit and generic writes, and read authorization. It also covers the scoped Runtime compaction entry point: +disabled commits, previews, expected-head conflicts, recovered capacity, audit changes, and historical citations. +`tests/e2e/test_mcp_transport.py` verifies the capacity tool's read-only annotation and equality with HTTP results +through the MCP transport. `tests/test_api_contract.py` verifies the operation and additive `compact` enum value. ### Regression guard @@ -603,8 +622,8 @@ them. # Unresolved questions -- **Deployment calibration.** The defaults remain 5,000 / 10,000 / 4 MiB. Isolated latency measurements are still - needed to recommend tighter budgets for particular workloads. +- **Deployment calibration.** The defaults remain 5,000 / 10,000 / 4 MiB. OceanBase scale measurements and isolated + latency measurements are still needed to recommend tighter budgets for particular workloads. - **Should compaction ever be automatic?** This RFC makes it explicit and operator-driven. Whether a scheduled compaction below a headroom threshold is safe depends on the authorization and dry-run model in #1425. - **What is the recovery path for a compacted entry?** Today: none through `reactivate()`. Whether a `restore` diff --git a/docs/zh/development/memory-layer.md b/docs/zh/development/memory-layer.md index 3f7df7defa..0ced194c4c 100644 --- a/docs/zh/development/memory-layer.md +++ b/docs/zh/development/memory-layer.md @@ -77,6 +77,9 @@ expected revision 和 citation 保留 optimistic concurrency,调用方无需 远程调用使用 `POST /v1/memory/capacity`,请求体为 `{"scope_id": "project-alpha"}`;Python 客户端提供 `PowerContextClient.get_memory_capacity(GetMemoryCapacityRequest(scope_id="project-alpha"))`。 Scope 尚无 Memory 时返回 404,查询不会创建 Memory。 +MCP 通过 `get_memory_capacity` 暴露相同的查询,并标记为只读、幂等。 +墓碑资格检查除了读取目标版本,还可能加载配置的保留窗口内的完整清单(默认 10 个版本);读取与解码开销随这些 +清单的总大小增长。这不是固定开销的计数器,适合显式检查容量,不适合频繁轮询。 `RuntimeConfig` 提供以下部署级默认值: @@ -98,15 +101,19 @@ Scope 尚无 Memory 时返回 404,查询不会创建 Memory。 `manifest_bytes` 计入完整规范化版本内容,包括变更记录及其原因。 超限时仍可执行 `forget()` 和 `organize()`;`reactivate()` 仅检查活跃条目数增长。压缩从当前清单移除达到保留 -年龄且未绑定标签的非活跃条目。通过 `RuntimeConfig` 显式启用,或使用 `MemoryCompactionPolicy(enabled=True)` -构造 `MemoryService`,执行前先预览: +年龄且未绑定标签的非活跃条目。构造 Runtime 时设置 `RuntimeConfig(memory_compaction_enabled=True)`,允许显式 +提交进程内压缩。该开关不会调度或自动触发压缩;调用 Scope 的 Runtime 入口预览,再用预览版本提交,避免处理 +已发生变化的 head: ```python -preview = await service.compact(memory, dry_run=True, limit=100) -result = await service.compact(memory, limit=100) -memory = result.memory +scoped = runtime.memory.for_scope(scope_id) +preview = await scoped.compact(dry_run=True, limit=100) +result = await scoped.compact(limit=100, expected_revision=preview.memory.revision) ``` +直接使用 service 的调用方可通过 `MemoryCompactionPolicy(enabled=True)` 构造 `MemoryService`,再以精确的 +Memory 版本调用 `service.compact(memory, ...)`。当前没有 HTTP、MCP 或 CLI 压缩操作。 + 压缩关闭时仍可预览,预览不写入版本。年龄按已推进的版本数计算:默认保留 10 个版本时,在版本 2 停用的条目 从版本 12 起可压缩。重新激活并再次停用会重置保留窗口。资格检查只读取这一近期窗口。无变化的维护操作不会 推进版本;如果所有墓碑都过新,可显式配置 `memory_compaction_min_tombstone_revisions=0`,或使用 @@ -120,8 +127,10 @@ memory = result.memory `reclaimed_bytes` 是完整规范化内容的有符号字节差;审计记录或较长原因可能抵消小规模清单缩减,因此该值可能 为负。后续版本不再携带本次压缩的变更记录。 -`MemoryService.revisions()` 在历史超过 `memory_max_history_revisions` 时,在展开历史前抛出 -`CapabilityNotSupportedError("history-window")`,不会静默截断结果;历史内容和精确版本读取仍然保留。 +`MemoryService.revisions(memory, since_revision=0, through_revision=None)` 读取区间 +`(since_revision, through_revision]`,默认以当前 head 为上界。请求区间超过 `memory_max_history_revisions` 时, +在展开前抛出 `CapabilityNotSupportedError("history-window")`,不会静默截断结果。例如,历史超过 100 个版本时, +仍可用 `through_revision=1` 读取首个版本,用 `since_revision=100, through_revision=200` 读取后续 100 个版本。 在接近 4 MiB 字节预算时,默认 100 个版本已可能包含约 400 MiB 规范内容,尚未计入对象开销。这是读取展开次数 上限,不是内存硬上限;调低预算或执行补救操作后,版本也可能超过字节预算。只有调用方能承担完整快照开销时 才应提高历史读取上限。这一上限不为 `entries()` 或 `changes()` 提供分页。 diff --git a/docs/zh/rfcs/1718_memory_capacity_contract.md b/docs/zh/rfcs/1718_memory_capacity_contract.md index aad3c05c90..3e0a31ee75 100644 --- a/docs/zh/rfcs/1718_memory_capacity_contract.md +++ b/docs/zh/rfcs/1718_memory_capacity_contract.md @@ -281,7 +281,7 @@ if isinstance(error, MemoryCapacityExceededError): ```python def _require_capacity( - self, base: Memory | None, content: MemoryContent, *, growth: frozenset[str], content_bytes: bytes + self, base: Memory | None, content: MemoryContent, *, growth: frozenset[MemoryCapacityDimension], content_bytes: bytes ) -> None: """Refuse a prepared Revision that grows a dimension past its budget.""" ``` @@ -381,19 +381,33 @@ Revision。 授权与 Memory 身份校验。这是一个增量 OpenAPI 操作:在 `openapi/powercontext.yaml` 中定义、重新生成、并添加契约 测试。SDK 调用方直接使用 service 层的 `capacity()`。该路由要求 `scope.read` 权限;Scope 尚无 Memory 时返回 404, 不会创建 Memory 或虚构引用来返回零值。Python HTTP 客户端使用 `PowerContextClient.get_memory_capacity()`。 +Scope 的 Runtime 提供 `capacity()`,Server MCP 提供带只读、幂等标记的 `get_memory_capacity`,并在 +`server-mcp` 能力清单中将其归为 `memory_read`。 + +资格判断不是固定开销的计数器:除了读取目标版本,还可能加载最多 +`memory_compaction_min_tombstone_revisions` 个近期版本的完整清单(默认 10 个)。读取与解码开销随这些清单的 +总大小增长。操作描述明确这一成本,供调用方显式检查容量,避免频繁轮询。本 RFC 不增加仅存变更的投影,也不 +合并重复的精确版本读取。 压缩刻意**不**在本 RFC 中通过 HTTP 暴露。它是一个维护操作,其授权模型属于 #1425 的更广保留策略;在那份设计之前 -就把它作为默认无认证的 Server 路由暴露出去顺序是错的。SDK 与进程内 runtime 调用方现在即可调用它。 +就把它作为默认无认证的 Server 路由暴露出去顺序是错的。进程内调用方使用 +`runtime.memory.for_scope(scope_id).compact(dry_run=False, limit=None, reason=None, expected_revision=None)`。 +Runtime 串行执行 Scope 写入,并将资格与提交规则交给 `MemoryService.compact()`。 +`memory_compaction_enabled=True` 只允许这些显式提交,不会调度或触发压缩。关闭时仍可预览;提交时传入 +`expected_revision=preview.memory.revision`,可在 head 变化后拒绝写入。直接使用 service 的调用方传入精确的 +Memory 版本并配置 `MemoryCompactionPolicy`。当前没有 HTTP、MCP 或 CLI 压缩操作。 ## Revision 历史约束 本 RFC 不删除也不限制已存储的 Revision。删除历史会破坏血缘、精确引用和 Handoff 验证,而物理擦除是 #1425 的边界。 -它确实约束了一处无界**读取**。`MemoryService.revisions()` 在循环中从 1 加载到 head 的每个 Revision,每次一个后端 -`get()`,因此一个有 1,000 个 Revision 的 Memory 单次调用会发出 1,000 次加载。本 RFC 用 `max_history_revisions` -为该读放大设上限,默认 100 个 Revision;超限时在展开历史前抛出 `CapabilityNotSupportedError("history-window")`, -现有映射将其转成指明该 capability 的 422。基于游标的替代方案是 #1657 的交付物,而这个上限正是 #1656 的验收标准所要求的、让调用方事先 -约定而非自行摸索的显式边界。 +`MemoryService.revisions(memory, since_revision=0, through_revision=None)` 按升序读取区间 +`(since_revision, through_revision]`,默认以当前 head 为上界。`since_revision` 必须非负,`through_revision` +必须在 1 与当前 head 之间,下界不得大于上界;相等时返回空元组。本次展开量为 +`through_revision - since_revision`,不得超过 `max_history_revisions`(默认 100)。超限请求在展开前抛出 +`CapabilityNotSupportedError("history-window")`。因此,含 1,000 个版本的 Memory 仍可按显式有界区间读取,无须 +提高部署上限:`through_revision=1` 读取首个版本,`since_revision=900` 读取最后 100 个版本。不传边界时仍请求 +全部可见历史。基于游标的替代方案由 #1657 负责,此接口不定义游标。 结果不会静默截断。每个 Revision 为 4 MiB 时,1,000 个快照接近 4 GiB,100 个也接近 400 MiB,尚未计入 Python 对象开销。这是读取展开次数上限,不是进程内存上限:补救操作和调低预算可能使版本超过字节预算。只有调用方能够 @@ -427,10 +441,13 @@ Revision。 - 墓碑年龄、重新激活后的窗口重置、标签保护,以及提交前新加标签时事务回滚。 - 满容量 Memory 通过显式零年龄立即恢复,同时保护活跃条目与带标签墓碑。 - 审计原因超过被移除指针大小时 `reclaimed_bytes` 为负,无变化压缩返回零。 -- 历史读取在 100 个 Revision 时成功,在 101 个时于展开前拒绝,并支持显式配置覆盖。 +- 历史请求区间为 100 个 Revision 时成功,为 101 个时于展开前拒绝;较长历史仍可有界读取,非法区间被拒绝, + 并支持显式配置覆盖。 `tests/e2e/test_memory_capacity.py` 覆盖 HTTP 与客户端契约,包括 Scope 尚无 Memory 时的 404、显式及通用写入的 -409 和读取权限。`tests/test_api_contract.py` 验证新增操作和 `compact` 枚举值。 +409 和读取权限;同时覆盖 Scope 的 Runtime 压缩入口:默认关闭时拒绝提交、预览、预期 head 冲突、容量恢复、 +审计变更和历史引用。`tests/e2e/test_mcp_transport.py` 通过 MCP 传输验证容量工具的只读标记及其与 HTTP 结果 +一致。`tests/test_api_contract.py` 验证新增操作和 `compact` 枚举值。 ### 回归防护 @@ -544,7 +561,8 @@ RFC 决定 manifest 何时可以不再携带它们。 # 未解决问题 -- **部署校准。** 默认值保留 5,000 / 10,000 / 4 MiB。仍需隔离的延迟测量,才能针对具体负载推荐更严格的预算。 +- **部署校准。** 默认值保留 5,000 / 10,000 / 4 MiB。仍需 OceanBase 规模测量和隔离的延迟测量,才能针对具体负载 + 推荐更严格的预算。 - **压缩是否应当自动化?** 本 RFC 让它显式、由运维驱动。低于余量阈值时执行定时压缩是否安全,取决于 #1425 的 授权与 dry-run 模型。 - **被压缩 entry 的恢复路径是什么?** 目前通过 `reactivate()` 没有恢复路径。是否值得定义一个把保留的正文作为新 diff --git a/integrations/capabilities.toml b/integrations/capabilities.toml index 5c5292f19d..b6c2505e2d 100644 --- a/integrations/capabilities.toml +++ b/integrations/capabilities.toml @@ -54,6 +54,9 @@ capabilities = ["memory_read"] id = "get_memory_entry" capabilities = ["memory_read"] [[toolsets.tools]] +id = "get_memory_capacity" +capabilities = ["memory_read"] +[[toolsets.tools]] id = "remember_memory" capabilities = ["memory_write"] [[toolsets.tools]] diff --git a/openapi/powercontext.yaml b/openapi/powercontext.yaml index b61daffeb3..2388150194 100644 --- a/openapi/powercontext.yaml +++ b/openapi/powercontext.yaml @@ -1562,6 +1562,9 @@ paths: description: >- Measure the current Memory head against the deployment budget, including exact canonical content bytes and the number of aged, untagged tombstones eligible for compaction. Returns 404 when no Memory exists. + Tombstone eligibility can load complete manifests for up to memory_compaction_min_tombstone_revisions + recent revisions (10 by default), in addition to reading the target revision. Read and decode cost scales + with their combined size; this is not a constant-cost counter and is unsuitable for frequent polling. operationId: get_memory_capacity x-powercontext-access: {action: scope.read, resource: {type: scope, scope-id-from: scope_id}} x-powercontext-scope-mode: current diff --git a/src/powercontext/builtin/artifacts/memory/service.py b/src/powercontext/builtin/artifacts/memory/service.py index d9b068919b..c168f98c7e 100644 --- a/src/powercontext/builtin/artifacts/memory/service.py +++ b/src/powercontext/builtin/artifacts/memory/service.py @@ -149,6 +149,7 @@ def __init__(self, code: str) -> None: "search-query": "memory search query must be non-empty text", "compaction-limit": "memory compaction limit must be positive", "history-limit": "memory history revision limit must be positive", + "through-range": "through_revision must be between 1 and the current head Revision", } super().__init__(messages[code]) @@ -214,15 +215,28 @@ async def latest(self, memory: Memory, /) -> Memory: canonical = await self.get(memory) return await self._backend.latest(canonical.artifact_id) - async def revisions(self, memory: Memory, /) -> tuple[Memory, ...]: - """Return the visible Memory history in ascending Revision order.""" + async def revisions( + self, memory: Memory, /, *, since_revision: int = 0, through_revision: int | None = None + ) -> tuple[Memory, ...]: + """Return history in ascending order within ``(since_revision, through_revision]``. + + The upper bound defaults to the current head. The history limit applies + to the requested interval, and oversized intervals are never truncated. + """ canonical = await self.get(memory) latest = await self._backend.latest(canonical.artifact_id) - if latest.revision > self._max_history_revisions: + upper = latest.revision if through_revision is None else through_revision + if upper < 1 or upper > latest.revision: + raise _InvalidMemoryOperationError("through-range") + if since_revision < 0: + raise _InvalidMemoryOperationError("since-negative") + if since_revision > upper: + raise _InvalidMemoryOperationError("since-greater") + if upper - since_revision > self._max_history_revisions: raise CapabilityNotSupportedError("history-window") history = [] - for revision in range(1, latest.revision + 1): + for revision in range(since_revision + 1, upper + 1): history.append( await self._backend.get( ArtifactRef(family=Memory.family, artifact_id=canonical.artifact_id, revision=revision) @@ -236,7 +250,11 @@ async def head(self, artifact_id: str, /) -> Memory: return await self._backend.latest(artifact_id) async def capacity(self, memory: Memory, /) -> MemoryCapacity: - """Measure an exact Revision, including eligible tombstones even when compaction is disabled.""" + """Measure an exact Revision, including eligible tombstones even when compaction is disabled. + + Eligibility can load complete manifests across the tombstone recovery + window. Cost scales with their combined size; this is not a cheap counter. + """ canonical = await self._canonical_memory(memory) values = self._capacity_values(canonical.content) @@ -269,7 +287,12 @@ def _capacity_values( } def _require_capacity( - self, base: Memory | None, content: MemoryContent, *, growth: frozenset[str], content_bytes: bytes + self, + base: Memory | None, + content: MemoryContent, + *, + growth: frozenset[MemoryCapacityDimension], + content_bytes: bytes, ) -> None: if not growth: return @@ -987,7 +1010,7 @@ async def _commit_existing_transition( changes: Sequence[MemoryChange], current_by_entry: dict[str, MemoryEntryVersion], entry_versions: tuple[MemoryEntryVersion, ...], - growth: frozenset[str] = frozenset(), + growth: frozenset[MemoryCapacityDimension] = frozenset(), ) -> Memory: sorted_manifest = tuple(sorted(manifest.values(), key=lambda item: item.entry_id.encode("utf-8"))) sorted_changes = tuple(sorted(changes, key=lambda change: change.entry_id.encode("utf-8"))) @@ -1346,7 +1369,7 @@ async def _prepare_commit( self._require_capacity( base, content, - growth=frozenset({"active_entries", "manifest_entries", "manifest_bytes"}), + growth=frozenset(dimension for dimension, _ in self._capacity_limits()), content_bytes=content_bytes, ) memory = Memory( diff --git a/src/powercontext/builtin/runtime/application.py b/src/powercontext/builtin/runtime/application.py index 757acf9914..02fc9c7262 100644 --- a/src/powercontext/builtin/runtime/application.py +++ b/src/powercontext/builtin/runtime/application.py @@ -58,6 +58,7 @@ Memory, MemoryCapacity, MemoryCitation, + MemoryCompactionResult, MemoryEntryInput, MemoryEntryVersion, MemoryHit, @@ -2468,6 +2469,27 @@ async def capacity(self) -> MemoryCapacity: _validate_memory_identity(context.artifacts.memory_artifact_id, current) return await service.capacity(current) + async def compact( + self, + *, + dry_run: bool = False, + limit: int | None = None, + reason: str | None = None, + expected_revision: int | None = None, + ) -> MemoryCompactionResult: + """Explicitly compact the Scope's current Memory under the configured policy. + + Enablement permits commits; it does not schedule them. Previews also work + while disabled. Pass the preview's revision to reject a changed head. + """ + + async with self._runtime._context(self.scope_id) as context, self._runtime._locked(self.scope_id): + service = context.artifacts.memory + current = await service.head(context.artifacts.memory_artifact_id) + _validate_memory_identity(context.artifacts.memory_artifact_id, current) + _validate_expected_revision(current, expected_revision) + return await service.compact(current, dry_run=dry_run, limit=limit, reason=reason) + async def list(self, *, include_inactive: bool = False, tag_filter: TagFilter | None = None) -> MemoryEntriesPage: async with self._runtime._context(self.scope_id) as context: service = context.artifacts.memory diff --git a/src/powercontext/http/_generated/schema.py b/src/powercontext/http/_generated/schema.py index fa73aa5f62..508871f72a 100644 --- a/src/powercontext/http/_generated/schema.py +++ b/src/powercontext/http/_generated/schema.py @@ -1601,7 +1601,16 @@ "including exact canonical content " "bytes and the number of aged, untagged " "tombstones eligible for compaction. " - "Returns 404 when no Memory exists.", + "Returns 404 when no Memory exists. " + "Tombstone eligibility can load " + "complete manifests for up to " + "memory_compaction_min_tombstone_revisions " + "recent revisions (10 by default), in " + "addition to reading the target " + "revision. Read and decode cost scales " + "with their combined size; this is not " + "a constant-cost counter and is " + "unsuitable for frequent polling.", "operationId": "get_memory_capacity", "requestBody": { "content": { diff --git a/src/powercontext/server/mcp.py b/src/powercontext/server/mcp.py index 8071b594b2..40f99a0b42 100644 --- a/src/powercontext/server/mcp.py +++ b/src/powercontext/server/mcp.py @@ -46,6 +46,7 @@ GET_ARTIFACT_CANDIDATE, GET_DREAM_RUN, GET_HANDOFF_REPORT, + GET_MEMORY_CAPACITY, GET_MEMORY_ENTRY, GET_SCOPE, GET_TOPIC_MEMORY, @@ -120,6 +121,7 @@ SEARCH_TOPIC_MEMORY.operation_id, GET_TOPIC_MEMORY.operation_id, LIST_MEMORY_ENTRIES.operation_id, + GET_MEMORY_CAPACITY.operation_id, GET_MEMORY_ENTRY.operation_id, REMEMBER_MEMORY.operation_id, REVISE_MEMORY_ENTRY.operation_id, @@ -147,6 +149,7 @@ SEARCH_TOPIC_MEMORY.operation_id, GET_TOPIC_MEMORY.operation_id, LIST_MEMORY_ENTRIES.operation_id, + GET_MEMORY_CAPACITY.operation_id, GET_MEMORY_ENTRY.operation_id, GET_HANDOFF_REPORT.operation_id, LIST_ARTIFACT_CANDIDATES.operation_id, diff --git a/tests/builtin/artifacts/memory/test_capacity.py b/tests/builtin/artifacts/memory/test_capacity.py index fb0e6fd175..0eca9bc53e 100644 --- a/tests/builtin/artifacts/memory/test_capacity.py +++ b/tests/builtin/artifacts/memory/test_capacity.py @@ -293,6 +293,19 @@ async def scenario(): third = await service.remember(memory=second, entries=(fact(3),), mode="append") with pytest.raises(CapabilityNotSupportedError, match="history-window"): await service.revisions(first) + assert await service.revisions(first, through_revision=1) == (first,) + assert await service.revisions(first, since_revision=1) == (second, third) + assert await service.revisions(first, since_revision=1, through_revision=2) == (second,) + assert await service.revisions(first, since_revision=3) == () + for bounds in ( + {"since_revision": -1}, + {"since_revision": 4}, + {"through_revision": 0}, + {"through_revision": 4}, + {"since_revision": 2, "through_revision": 1}, + ): + with pytest.raises(ValueError): + await service.revisions(first, **bounds) loaded = [] original_get = backend.get @@ -325,6 +338,8 @@ async def scenario(): for reader in readers: with pytest.raises(CapabilityNotSupportedError, match="history-window"): await reader.revisions(first) + assert await reader.revisions(first, through_revision=1) == (first,) + assert await reader.revisions(first, since_revision=100) == (current,) assert await reader.get(first) == first expanded = MemoryService(backend=backend, max_history_revisions=101) history = await expanded.revisions(first) diff --git a/tests/e2e/test_mcp_transport.py b/tests/e2e/test_mcp_transport.py index 104cf90587..f85fcbd8f6 100644 --- a/tests/e2e/test_mcp_transport.py +++ b/tests/e2e/test_mcp_transport.py @@ -150,6 +150,22 @@ async def exercise_tools() -> set[str]: ) assert empty_list.structured_content == {"memory": None, "entries": []} + capacity_tool = projected_tools["get_memory_capacity"] + assert capacity_tool.annotations is not None + assert capacity_tool.annotations.readOnlyHint is True + written = await client.call_tool( + "remember_memory", {"scope_id": scope["scope_id"], "kind": "fact", "text": "Inspect capacity via MCP."} + ) + capacity = await client.call_tool("get_memory_capacity", {"scope_id": scope["scope_id"]}) + assert not capacity.is_error + assert capacity.structured_content is not None + assert capacity.structured_content["active_entry_count"] == 1 + assert capacity.structured_content["manifest_entry_count"] == 1 + assert written.structured_content is not None + assert capacity.structured_content["memory_ref"] == written.structured_content["memory"] + http_capacity = await http_client.post("/v1/memory/capacity", json={"scope_id": scope["scope_id"]}) + assert http_capacity.json() == capacity.structured_content + created_review_scope = await client.call_tool( "create_scope", { @@ -214,6 +230,7 @@ async def exercise_tools() -> set[str]: "get_artifact_candidate", "get_handoff_report", "get_scope", + "get_memory_capacity", "get_memory_entry", "get_topic_memory", "handoff_current_work", diff --git a/tests/e2e/test_memory_capacity.py b/tests/e2e/test_memory_capacity.py index 853fd2bc57..3e2d961ac5 100644 --- a/tests/e2e/test_memory_capacity.py +++ b/tests/e2e/test_memory_capacity.py @@ -19,9 +19,23 @@ import httpx import pytest +from powercontext.builtin.artifacts.memory import ( + CapabilityNotSupportedError, + MemoryCapacityExceededError, + MemoryEntryInput, +) from powercontext.builtin.persistence.sqlite import SQLiteConfig +from powercontext.builtin.runtime import ( + BuiltinConfig, + GetMemoryEntryRequest, + RetireMemoryEntryRequest, + open_builtin_runtime, +) +from powercontext.builtin.runtime import RememberMemoryRequest as RuntimeRememberMemoryRequest from powercontext.builtin.runtime.config import RuntimeConfig +from powercontext.builtin.scope import ScopeDraft from powercontext.client import PowerContextClient, ServerResponseError +from powercontext.errors import ArtifactNotFoundError, RevisionConflictError from powercontext.http import GetMemoryCapacityRequest, RememberMemoryRequest from powercontext.server.authentication import StaticBearerAuthenticationProvider from powercontext.server.authz import PrincipalRef @@ -29,6 +43,62 @@ from powercontext.server.settings import AccessControlConfig, BearerAuthConfig, McpConfig, ServerSettings +@pytest.mark.parametrize("enabled", [False, True]) +def test_runtime_compaction_requires_enablement_and_recovers_capacity(tmp_path, enabled): + async def scenario(): + config = BuiltinConfig( + database=SQLiteConfig(url=f"sqlite+aiosqlite:///{tmp_path / 'runtime-capacity.db'}"), + runtime=RuntimeConfig( + memory_max_active_entries=1, + memory_max_manifest_entries=1, + memory_compaction_enabled=enabled, + memory_compaction_min_tombstone_revisions=0, + ), + ) + async with open_builtin_runtime(config) as runtime: + assert runtime.scopes is not None + scope = await runtime.scopes.create( + ScopeDraft(title="Capacity", summary="Runtime compaction", idempotency_key="capacity") + ) + memory = runtime.memory.for_scope(scope.scope_id) + with pytest.raises(ArtifactNotFoundError): + await memory.compact(dry_run=True) + assert (await memory.list()).memory_ref is None + await memory.remember( + RuntimeRememberMemoryRequest(entries=(MemoryEntryInput(kind="fact", text="Old fact"),)) + ) + entry = (await memory.list()).entries[0] + retired = await memory.retire(RetireMemoryEntryRequest(citation=entry.citation)) + request = RuntimeRememberMemoryRequest(entries=(MemoryEntryInput(kind="fact", text="New fact"),)) + with pytest.raises(MemoryCapacityExceededError): + await memory.remember(request) + before = await memory.capacity() + preview = await memory.compact(dry_run=True, limit=1, reason="Recover capacity") + assert preview.entry_ids == (entry.entry.entry_id,) + assert preview.memory.as_ref() == retired.memory_ref + assert await memory.capacity() == before + if not enabled: + with pytest.raises(CapabilityNotSupportedError, match="compaction"): + await memory.compact() + assert await memory.capacity() == before + return + with pytest.raises(RevisionConflictError): + await memory.compact(expected_revision=entry.memory_ref.revision) + assert await memory.capacity() == before + result = await memory.compact(expected_revision=preview.memory.revision, limit=1, reason="Recover capacity") + assert result.entry_ids == preview.entry_ids + assert result.reclaimed_bytes == preview.reclaimed_bytes + assert result.memory.revision == retired.memory_ref.revision + 1 + assert (await memory.capacity()).manifest_entry_count == 0 + assert await memory.get(GetMemoryEntryRequest(citation=entry.citation)) == entry + changes = await memory.changes(since_revision=retired.memory_ref.revision) + assert changes.revisions[0].changes[0].op == "compact" + await memory.remember(request) + assert (await memory.capacity()).manifest_entry_count == 1 + + asyncio.run(scenario()) + + def test_capacity_and_refusal_through_server_and_client(tmp_path): async def scenario(): app = create_server_app( diff --git a/tests/test_mcp.py b/tests/test_mcp.py index 2e5af73b59..738e999af7 100644 --- a/tests/test_mcp.py +++ b/tests/test_mcp.py @@ -91,6 +91,7 @@ async def inspect_components() -> tuple[list[str], int, int]: "create_work_contract", "finalize_handoff", "get_artifact_candidate", + "get_memory_capacity", "get_memory_entry", "get_topic_memory", "get_scope", @@ -118,7 +119,7 @@ async def inspect_components() -> tuple[list[str], int, int]: assert prompt_count == 0 -def test_mcp_topic_memory_tools_are_read_only_and_flush_is_excluded() -> None: +def test_mcp_memory_reads_are_read_only_and_flush_is_excluded() -> None: async def inspect_annotations() -> dict[str, Any]: async with Client(create_mcp_server(create_app())) as client: return {tool.name: tool.annotations for tool in await client.list_tools()} @@ -126,11 +127,12 @@ async def inspect_annotations() -> dict[str, Any]: tools = run_async(inspect_annotations) assert "flush_topic_memory" not in tools - for name in ("search_topic_memory", "get_topic_memory"): + for name in ("search_topic_memory", "get_topic_memory", "get_memory_capacity"): annotations = tools[name] assert annotations is not None assert annotations.readOnlyHint is True assert annotations.destructiveHint is False + assert annotations.idempotentHint is True assert annotations.openWorldHint is False