From c482b1213bfb9569800183dd03132fe1998bd807 Mon Sep 17 00:00:00 2001 From: KXH Date: Fri, 31 Jul 2026 15:57:21 +0800 Subject: [PATCH] docs: publish bilingual project guides --- CONTRIBUTING.md | 143 ++++++++-- CONTRIBUTING.zh-CN.md | 129 +++++++++ DOCUMENT_INDEX.md | 16 +- README.md | 307 +++++++++++++++++----- README.zh-CN.md | 231 ++++++++++++++++ docs/development-log/daily/2026-07-31.md | 8 + tests/unit/runtime/test_release_assets.py | 49 +++- 7 files changed, 794 insertions(+), 89 deletions(-) create mode 100644 CONTRIBUTING.zh-CN.md create mode 100644 README.zh-CN.md diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 992482c..472e2e7 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,30 +1,141 @@ # Contributing to OperCerta -Thank you for contributing to OperCerta. +**English** | [简体中文](CONTRIBUTING.zh-CN.md) -## Before Starting +Thank you for helping improve OperCerta. Contributions should preserve its +core property: LLM output may assist investigation and explanation, but +deterministic policy, human approval, and database constraints control +high-risk business writes. +## Before You Start + +- Search existing issues and pull requests before opening a duplicate. +- Open an issue before changing the Agent state model, approval boundary, + persistence model, public API, or dependency architecture. - Keep each change focused on one problem. -- Do not commit credentials, private keys, customer data, or confidential company information. -- Open an issue before making significant architectural changes. +- Use only synthetic or anonymized examples. +- Never commit credentials, tokens, private endpoints, customer records, or + confidential company material. + +## Development Environment + +The recommended environment is Linux or WSL2 with Docker Compose v2, Python +3.12 managed by `uv`, and Node.js 24. + +```bash +git clone https://github.com/KXHXK/opercerta.git +cd opercerta + +uv sync --frozen --all-groups + +cd web +npm ci +``` + +For a local runtime, copy `.env.compose.example` to `.env.compose`, replace its +placeholders with local-only values, and follow the [Quick Start](README.md#quick-start). ## Development Workflow -1. Create a branch from `main`. -2. Make the smallest change needed. -3. Add or update tests when behavior changes. -4. Run the relevant tests and formatting checks. -5. Review the final diff before opening a pull request. +1. Create a branch from the latest `main`. +2. Reproduce the problem or add a failing test. +3. Implement the smallest coherent change. +4. Run the relevant focused tests. +5. Run the required quality gates for the affected area. +6. Review `git diff` for unrelated changes and sensitive data. +7. Open a pull request with the problem, implementation, validation, and known limits. + +Do not weaken or delete a safety assertion merely to make a test pass. When a +contract must change, explain the business reason and update implementation, +tests, documentation, and migration/recovery behavior together. + +## Agent and Business Safety Rules + +- Keep user inputs and model outputs behind strict typed schemas. +- Add tools to the explicit allowlist; do not enable arbitrary tool execution. +- Keep business quantities, permissions, and state transitions deterministic. +- Preserve human approval for controlled writes. +- Bind approvals to the relevant evidence, rule, fact, and plan hashes. +- Re-fetch authoritative facts after approval and before execution. +- Make write tools idempotent and verify their database postconditions. +- Fail closed on provider, parsing, policy, approval, or dependency errors. +- Preserve LangGraph restart behavior and do not treat checkpoints as the business source of truth. +- Do not log secrets, full prompts, hidden reasoning, SQL parameters, or sensitive evidence. + +## Quality Gates + +### Python and backend -## Pull Requests +```bash +uv run ruff check . +uv run ruff format --check . +uv run mypy src +uv run pytest -q +uv run python scripts/run_opercerta_evaluation.py +uv run python scripts/run_agent_evaluation.py +uv run python scripts/verify_repository_safety.py +``` + +Database integration tests require a compatible PostgreSQL/pgvector instance. +Use an isolated test database and never point the suite at business or personal +data. + +### Frontend + +```bash +cd web +npm run test:run +npm run build +``` + +### Compose behavior + +With a fresh local Compose project: + +```bash +docker compose up --build -d --wait +python3 scripts/verify_agent_compose.py +docker compose restart api mcp +python3 scripts/verify_agent_compose.py --recovery-only +``` + +Do not run the business verifier against a database whose state must be +preserved: it intentionally creates synthetic operations and work orders. + +## Documentation + +- Keep `README.md` and `README.zh-CN.md` structurally equivalent when public + project behavior changes. +- Keep `CONTRIBUTING.md` and `CONTRIBUTING.zh-CN.md` equivalent when the + contribution process changes. +- Register every added, moved, or removed Markdown file in `DOCUMENT_INDEX.md` + in the same commit. +- Distinguish observed results from assumptions, and do not present fixed + synthetic evaluations as production accuracy or SLA evidence. + +## Pull Request Checklist A pull request should include: -- A short description of the problem. -- A summary of the implementation. -- The validation performed. -- Any known limitations or follow-up work. +- the problem and intended behavior; +- the implementation and important trade-offs; +- exact validation commands and results; +- database or API compatibility impact; +- recovery, idempotency, approval, and security impact when relevant; +- known limitations and follow-up work. + +Before requesting review, confirm: + +- [ ] the change is scoped and the branch is based on current `main`; +- [ ] behavior changes have tests; +- [ ] secrets and private data are absent; +- [ ] relevant Python/frontend/Compose gates pass; +- [ ] public English and Chinese documentation remain synchronized; +- [ ] `DOCUMENT_INDEX.md` is current. -## Security and Privacy +## Reporting Security Issues -Use synthetic or anonymized examples in documentation and tests. Never publish real access tokens, internal endpoints, customer records, or confidential operational data. +Do not publish exploitable details, credentials, or sensitive data in a public +issue. Contact the repository owner privately with a minimal reproduction, +affected versions, and impact. A dedicated security policy and private +reporting channel are planned but are not yet configured. diff --git a/CONTRIBUTING.zh-CN.md b/CONTRIBUTING.zh-CN.md new file mode 100644 index 0000000..c431ac7 --- /dev/null +++ b/CONTRIBUTING.zh-CN.md @@ -0,0 +1,129 @@ +# 参与 OperCerta 开发 + +[English](CONTRIBUTING.md) | **简体中文** + +感谢你帮助改进 OperCerta。所有贡献都应保留项目的核心属性:LLM 输出可以辅助 +调查和解释,但高风险业务写入必须由确定性规则、人工审批和数据库约束控制。 + +## 开始之前 + +- 新建 Issue 前先搜索已有 Issue 和 Pull Request,避免重复。 +- 修改 Agent 状态模型、审批边界、持久化模型、公开 API 或依赖架构前,先建立 Issue 讨论。 +- 每次变更只解决一个明确问题。 +- 示例只使用合成或匿名数据。 +- 禁止提交凭据、token、私有地址、客户记录或原单位机密材料。 + +## 开发环境 + +推荐使用 Linux 或 WSL2、Docker Compose v2、由 `uv` 管理的 Python 3.12, +以及 Node.js 24。 + +```bash +git clone https://github.com/KXHXK/opercerta.git +cd opercerta + +uv sync --frozen --all-groups + +cd web +npm ci +``` + +本地运行时需要把 `.env.compose.example` 复制为 `.env.compose`,将占位符替换为 +仅本地使用的值,然后按照[快速启动](README.zh-CN.md#快速启动)操作。 + +## 开发流程 + +1. 从最新 `main` 创建分支。 +2. 复现问题或先增加失败测试。 +3. 实现最小且完整的修改。 +4. 运行受影响范围的定向测试。 +5. 运行对应区域的必需质量门禁。 +6. 检查 `git diff`,排除无关修改和敏感数据。 +7. 创建 Pull Request,说明问题、实现、验证和已知边界。 + +不得仅为让测试通过而削弱或删除安全断言。如果契约确实需要修改,应说明业务 +原因,并同步更新实现、测试、文档以及迁移/恢复行为。 + +## Agent 与业务安全规则 + +- 用户输入和模型输出必须经过严格的类型化 Schema。 +- 工具必须加入显式白名单,禁止任意工具执行。 +- 业务数量、权限和状态转换必须保持确定性。 +- 受控写入必须保留人工审批。 +- 审批必须绑定相关证据、规则、事实和计划哈希。 +- 审批后、执行前必须重新读取权威事实。 +- 写工具必须幂等,并验证数据库后置条件。 +- provider、解析、规则、审批或依赖异常时必须 fail closed。 +- 保持 LangGraph 重启恢复语义,不得把 checkpoint 当成业务事实源。 +- 不得记录密钥、完整 Prompt、隐藏推理、SQL 参数或敏感证据。 + +## 质量门禁 + +### Python 与后端 + +```bash +uv run ruff check . +uv run ruff format --check . +uv run mypy src +uv run pytest -q +uv run python scripts/run_opercerta_evaluation.py +uv run python scripts/run_agent_evaluation.py +uv run python scripts/verify_repository_safety.py +``` + +数据库集成测试需要兼容的 PostgreSQL/pgvector。必须使用隔离测试数据库,禁止把 +测试指向业务或个人数据。 + +### 前端 + +```bash +cd web +npm run test:run +npm run build +``` + +### Compose 行为 + +在全新的本地 Compose 项目中执行: + +```bash +docker compose up --build -d --wait +python3 scripts/verify_agent_compose.py +docker compose restart api mcp +python3 scripts/verify_agent_compose.py --recovery-only +``` + +业务验证脚本会创建合成 operation 和工单,不得对需要保留状态的数据库运行。 + +## 文档规则 + +- 公开项目行为变化时,保持 `README.md` 与 `README.zh-CN.md` 结构和事实一致。 +- 贡献流程变化时,保持 `CONTRIBUTING.md` 与 `CONTRIBUTING.zh-CN.md` 一致。 +- 新增、移动或删除 Markdown 文件时,必须在同一提交更新 `DOCUMENT_INDEX.md`。 +- 区分实测结果与假设,不得把固定合成评测表述为生产准确率或 SLA 证据。 + +## Pull Request 检查表 + +Pull Request 应包含: + +- 问题与预期行为; +- 实现方式和重要取舍; +- 准确的验证命令与结果; +- 数据库或 API 兼容性影响; +- 与恢复、幂等、审批和安全相关的影响; +- 已知限制和后续工作。 + +请求 Review 前确认: + +- [ ] 变更范围明确,分支基于最新 `main`; +- [ ] 行为变化具有测试; +- [ ] 不包含密钥和私有数据; +- [ ] 相关 Python、前端或 Compose 门禁通过; +- [ ] 公开中英文文档保持同步; +- [ ] `DOCUMENT_INDEX.md` 已更新。 + +## 报告安全问题 + +不要在公开 Issue 中发布可利用细节、凭据或敏感数据。请私下联系仓库所有者, +提供最小复现、受影响版本和影响范围。项目计划增加独立安全策略和私密报告渠道, +但当前尚未配置。 diff --git a/DOCUMENT_INDEX.md b/DOCUMENT_INDEX.md index 67a24cd..b9f486f 100644 --- a/DOCUMENT_INDEX.md +++ b/DOCUMENT_INDEX.md @@ -1,16 +1,16 @@ # OperCerta 文档总索引 -本索引完整登记 OperCerta 当前根工作树中的 116 份 Markdown 文档,并保留旧电脑 6 个 `.worktrees/` 的 456 条历史登记。当前根工作树与每个历史 worktree 使用独立六列表格和独立序号;路径均相对于仓库根目录;日期表示文档首次建立日期。历史 worktree 表用于追溯分支资料,不表示对应物理目录仍存在。Git 元数据、依赖目录、虚拟环境和工具缓存不属于项目文档登记范围。 +本索引完整登记 OperCerta 当前根工作树中的 118 份 Markdown 文档,并保留旧电脑 6 个 `.worktrees/` 的 456 条历史登记。当前根工作树与每个历史 worktree 使用独立六列表格和独立序号;路径均相对于仓库根目录;日期表示文档首次建立日期。历史 worktree 表用于追溯分支资料,不表示对应物理目录仍存在。Git 元数据、依赖目录、虚拟环境和工具缓存不属于项目文档登记范围。 Typora 显示:首次运行 `powershell -ExecutionPolicy Bypass -File scripts/install_typora_index_theme.ps1`,重启 Typora 后选择 `主题 → OperCerta Index`。该主题让正文使用 96% 窗口宽度,并统一设置下列全部六列表格的列宽、自动换行和字号。 ## 项目核心学习导航 -先按 A1–A9 完成一轮“阅读 → 找到代码 → 手动验证 → 自己复述”。后面的 116 份当前文档与 456 条历史 worktree 登记是排查问题和深入学习时使用的资料库,不需要从头到尾顺序阅读。 +先按 A1–A9 完成一轮“阅读 → 找到代码 → 手动验证 → 自己复述”。后面的 118 份当前文档与 456 条历史 worktree 登记是排查问题和深入学习时使用的资料库,不需要从头到尾顺序阅读。 | 阶段 | 核心主题 | 优先阅读 | 代码与配置入口 | 必做实践 | 掌握标准 | | ---: | --- | --- | --- | --- | --- | -| A1 | 项目地图与当前边界 | [README](README.md)、[当前状态](docs/development-log/current-state.md)、[实施交接](IMPLEMENTATION_HANDOFF.md) | [Compose](compose.yaml)、[API 运行入口](src/opercerta/runtime/api.py)、[MCP 运行入口](src/opercerta/runtime/mcp.py) | 不看稿,用 60 秒说清楚三个业务场景、系统解决的问题、当前已完成能力和仍关闭的生产门禁。 | 能区分“本地验证通过、静态公网展示、真实生产上线”三种状态,不夸大成果。 | +| A1 | 项目地图与当前边界 | [README English](README.md)、[README 简体中文](README.zh-CN.md)、[当前状态](docs/development-log/current-state.md)、[实施交接](IMPLEMENTATION_HANDOFF.md) | [Compose](compose.yaml)、[API 运行入口](src/opercerta/runtime/api.py)、[MCP 运行入口](src/opercerta/runtime/mcp.py) | 不看稿,用 60 秒说清楚三个业务场景、系统解决的问题、当前已完成能力和仍关闭的生产门禁。 | 能区分“本地验证通过、静态公网展示、真实生产上线”三种状态,不夸大成果。 | | A2 | 业务动机与三业务闭环 | [OperCerta 详细设计](docs/specs/2026-07-14-opercerta-design.md)、[三业务发布设计](docs/superpowers/specs/2026-07-20-opercerta-three-business-release-design.md) | [场景注册表](src/opercerta/application/scenario_registry.py)、[库存图](src/opercerta/workflow/replenishment_graph.py)、[设备图](src/opercerta/workflow/equipment_maintenance_graph.py)、[任务恢复图](src/opercerta/workflow/task_recovery_graph.py) | 分别画出库存短缺、设备异常、任务阻塞的“异常信号 → 调查 → 审批 → 写入 → 验证”流程。 | 能解释为什么传统工单需要 Agent 辅助,以及哪些规则必须由确定性代码控制。 | | A3 | Agent 总体架构与循环 | [Agent 核心架构设计](docs/superpowers/specs/2026-07-21-opercerta-agent-core-architecture-design.md)、[单根 Agent Loop 设计](docs/superpowers/specs/2026-07-26-single-root-agent-loop-and-case-workspace-design.md) | [主 Agent 图](src/opercerta/workflow/agent_controlled_action_graph.py)、[Harness](src/opercerta/agent/harness.py)、[Prompt 注册表](src/opercerta/agent/prompt_registry.py) | 对照一次 Agent Trace,逐步标出感知、语义理解、规划、工具、记忆、执行反馈和回环位置。 | 能解释 LangGraph 为什么是状态编排内核,而不是一个普通业务节点;能说明何时循环、何时中断等待审批。 | | A4 | LLM、Prompt 与 LangChain | [核心技术手册](docs/learning/opercerta-core-technical-guide.md)、[真实模型证据](docs/release-evidence/real-model-representative-validation.md) | [LangChain 模型适配器](src/opercerta/infrastructure/langchain_model_gateway.py)、[模型端口](src/opercerta/infrastructure/model_gateway.py)、[Tool Loop Prompt](src/opercerta/prompts/tool-loop-v1.md) | 对比 Mock 与 Kimi 模式的输入、结构化输出、超时和失败收口;从 Trace 找出一次模型决策。 | 能说清 LLM 负责语义与规划、不直接越权写库;Prompt、Harness、Provider Adapter 各自解决什么问题。 | @@ -20,15 +20,15 @@ Typora 显示:首次运行 `powershell -ExecutionPolicy Bypass -File scripts/i | A8 | PostgreSQL、Redis、Docker 与可观测性 | [三业务发布证据](docs/release-evidence/three-business-release.md)、[Docker 证据](docs/release-evidence/docker-linux-runtime.md)、[可观测性证据](docs/release-evidence/observability-security-regression.md) | [Compose](compose.yaml)、[Dockerfile](Dockerfile)、[Redis 缓存](src/opercerta/infrastructure/cache.py)、[数据库迁移](migrations)、[Tracing](src/opercerta/observability/tracing.py) | 执行健康检查、查看容器状态、重启 API/MCP,并确认 checkpoint、业务事实和工单没有丢失或重复。 | 能解释容器与 Compose 的区别、Redis 为什么不是权威存储、PostgreSQL/pgvector 的双重职责及日志如何安全关联。 | | A9 | 故障复盘与面试输出 | [面试讲解](docs/learning/opercerta-interview-guide.md)、[工程案例集](docs/development-log/interview-casebook.md)、[最新开发日志](docs/development-log/daily/2026-07-31.md) | 选择前述任一真实故障对应的代码、日志和修复提交。 | 分别完成 30 秒、3 分钟和 10 分钟讲解;独立复述一个“现象 → 根因 → 修复 → 验证 → 取舍”案例。 | 不看文档也能讲清业务价值、核心链路、技术取舍、可靠性证据和未完成边界,并能接受追问。 | -## 根工作树(116 份) +## 根工作树(118 份) 显示说明:本表及后续各 worktree 表均保持“序号、文件名、路径、用途、状态、日期”六列完整字段。 | 序号 | 文件名 | 路径 | 用途(详细) | 状态 | 日期 | | ---: | --- | --- | --- | --- | --- | -| 1 | `README.md` | `README.md` | 项目总入口,说明三业务场景、核心技术栈、运行方式、演示边界、学习入口和发布门禁,供开发者、审阅者与面试官快速了解 OperCerta。 | 已同步最终 main、667/60/9-of-9、Real Kimi 代表范围、静态 production、Showcase 预发布与产品边界 | 2026-07-30 | +| 1 | `README.md` | `README.md` | 英文项目总入口,说明业务背景、三业务功能、Agent 闭环、架构、快速启动、使用流程、模型模式、验证结果、可靠性边界和路线图,并提供中文切换。 | 已重构为开源项目导向英文默认页;移除求职、组合项目和过时设计入口 | 2026-07-30 | | 2 | `IMPLEMENTATION_HANDOFF.md` | `IMPLEMENTATION_HANDOFF.md` | 跨对话和上下文压缩后的实施交接文件,记录当前分支、已验证事实、未完成事项、下一步动作及禁止越过的发布边界。 | 已同步 PR #18、最终 main Compose、静态 production、Showcase 预发布与下一掌握阶段 | 2026-07-30 | -| 3 | `DOCUMENT_INDEX.md` | `DOCUMENT_INDEX.md` | OperCerta 全部项目文档的唯一总登记表,用于按文件名、路径、用途、状态和日期统一检索、复查与交接。 | 当前根工作树 116 份文档已完整登记;另保留 6 个旧 worktree 的 456 条历史记录 | 2026-07-15 | +| 3 | `DOCUMENT_INDEX.md` | `DOCUMENT_INDEX.md` | OperCerta 全部项目文档的唯一总登记表,用于按文件名、路径、用途、状态和日期统一检索、复查与交接。 | 当前根工作树 118 份文档已完整登记;另保留 6 个旧 worktree 的 456 条历史记录 | 2026-07-15 | | 4 | `2026-07-14-agent-project-naming-design.md` | `docs/specs/2026-07-14-agent-project-naming-design.md` | 定义 OperCerta、ForenTrail、SiteVerum、Federune 四个项目的命名原则、语义边界与品牌一致性,防止项目职责和名称漂移。 | 已冻结为命名基线 | 2026-07-14 | | 5 | `ai-agent-portfolio-overall-design.md` | `docs/specs/ai-agent-portfolio-overall-design.md` | 规定四个 AI Agent 项目的整体定位、差异化业务范围、技术能力组合、实施顺序和共同约束,是项目组合的最高层设计依据。 | 已冻结为总体设计基线;文件名已统一为英文路径 | 2026-07-14 | | 6 | `2026-07-14-agent-portfolio-design.md` | `docs/specs/2026-07-14-agent-portfolio-design.md` | 设计四项目如何组合成求职作品集,包括能力覆盖、展示顺序、共享基础设施边界和避免重复建设的原则。 | 已冻结为组合设计基线 | 2026-07-14 | @@ -140,8 +140,10 @@ Typora 显示:首次运行 `powershell -ExecutionPolicy Bypass -File scripts/i | 112 | `tool-loop-v1.md` | `src/opercerta/prompts/tool-loop-v1.md` | Tool Loop 主 Prompt,规定模型在 Observation 后选择继续调用只读工具、请求审批或结束,并约束 JSON 工具协议。 | v1 已用于单根 Agent Loop;真实 Kimi 兼容边界另有事件记录 | 2026-07-26 | | 113 | `verifier-v1.md` | `src/opercerta/prompts/verifier-v1.md` | Verifier 角色版本化 Prompt,用于批准后重新核对事实、识别漂移并决定执行、重审批或安全终止。 | v1 已实施并有事实漂移回归测试 | 2026-07-21 | | 114 | `2026-07-30.md` | `docs/development-log/daily/2026-07-30.md` | 记录新电脑环境最终收口、WSL 登录 shell Node PATH 的 TDD 修复、冻结依赖国内镜像恢复、DrvFS 权限边界、PR/main 门禁,以及发布材料防漂移与静态安全头修订。 | PR #17/main 五项 CI、667/60/9-of-9、Netlify 静态 production 安全头与产物一致性均已验证 | 2026-07-30 | -| 115 | `CONTRIBUTING.md` | `CONTRIBUTING.md` | 规定 OperCerta 的分支、最小变更、测试、PR 说明、安全和隐私贡献流程,确保外部协作不引入凭据、客户数据或未经评审的架构扩张。 | 已由 main 的 PR #16 引入;在环境分支同步登记以保证合并态索引一致 | 2026-07-30 | +| 115 | `CONTRIBUTING.md` | `CONTRIBUTING.md` | 英文贡献指南,规定开发环境、TDD 流程、Agent/业务安全规则、Python/前端/Compose 门禁、双语文档同步、PR 检查表和安全问题报告方式。 | 已扩展为完整英文贡献入口,并提供简体中文切换 | 2026-07-30 | | 116 | `2026-07-31.md` | `docs/development-log/daily/2026-07-31.md` | 记录 OperCerta 最终 main/CI 复核、Netlify 静态发布与回滚点、Showcase 预发布、自动化工程门禁结论、产品边界和下一掌握任务。 | 求职静态展示与自动化工程已收口;个人掌握待实演,产品 production gate 保持 CLOSED | 2026-07-31 | +| 117 | `README.zh-CN.md` | `README.zh-CN.md` | 简体中文项目总入口,与英文 README 对齐说明业务背景、功能、Agent 闭环、技术架构、快速启动、用法、测试结果、开发状态和生产边界。 | 已建立独立中文版;通过顶部链接与英文默认页互相切换 | 2026-07-31 | +| 118 | `CONTRIBUTING.zh-CN.md` | `CONTRIBUTING.zh-CN.md` | 简体中文贡献指南,与英文版对齐说明开发流程、Agent 安全约束、质量门禁、双语文档维护、PR 检查表和安全报告要求。 | 已建立独立中文版;通过顶部链接与英文版互相切换 | 2026-07-31 | ## 历史 Worktree:agent-core-architecture(82 条记录) diff --git a/README.md b/README.md index 90734e6..ab55a8e 100644 --- a/README.md +++ b/README.md @@ -1,63 +1,244 @@ -# OperCerta | Auditable Operations Agent - -OperCerta is a controlled AI operations agent for inventory exceptions, -equipment alerts, and operational work orders. It combines FastAPI, -LangGraph, FastMCP, PostgreSQL, approval checkpoints, idempotent tool calls, -and audit-focused observability in a reproducible reference implementation. - -- [Project showcase](https://opercerta-kxh.netlify.app) -- [Portfolio overview](https://kxh-agent-portfolio.netlify.app) -- [Showcase pre-release](https://github.com/KXHXK/opercerta/releases/tag/v0.1.0-showcase.1) -- Status: engineering showcase; the public site is read-only and the - production release gate remains closed. - -## 中文说明 - -OperCerta 是面向库存异常、设备告警和运营工单的智能运营处置 Agent 独立作品仓库。 - -> 当前状态:库存补货、设备维修、作业异常恢复三条 FastAPI + 单根 LangGraph + 最小 LangChain + FastMCP + PostgreSQL/pgvector 闭环、演示 JWT/RBAC、Agent Trace、本地 React 控制台、Redis 只读证据缓存和真实 FastEmbed RAG 已有自动化证据。少量 Moonshot AI `kimi-k2.6` 代表验证覆盖三业务只读、库存批准写入和无效 provider fail-closed,未回退 Mock 冒充成功。[PR #18](https://github.com/KXHXK/opercerta/pull/18) 合入最终静态发布证据;最新 [main CI](https://github.com/KXHXK/opercerta/actions/runs/30541088053) 在 `298fc59` 上通过 667 条后端测试、19 个前端测试文件/60 条用例、9/9 冻结 Agent 评测和真实 Compose 构建、三业务数据库副作用与 API/MCP 重启恢复。[零成本静态项目专题](https://opercerta-kxh.netlify.app)已晋级 deploy `6a6b17cf496c38056f737264`,安全头和资源 SHA-256 已经公网复验;[Showcase 预发布 v0.1.0-showcase.1](https://github.com/KXHXK/opercerta/releases/tag/v0.1.0-showcase.1)精确指向该已验证提交。公开页面不提供后端写入口;生产身份、交互 HTTPS 后端、自动部署和公开 API 尚未完成,产品生产发布门禁:`CLOSED`。 - -## 当前已验证范围 - -- 严格非法输入与 JSON-only 恢复快照; -- 确定性恢复决策矩阵; -- PostgreSQL Schema、Alembic 升降级和审批原子竞态; -- 授权后幂等工单写入与并发安全重放; -- 独立 `langgraph` Schema checkpointer; -- LangGraph interrupt、审批绑定、批准后事实重取、拒绝终止和 A/B 重启恢复; -- 真实 MCP 工单幂等写入、写后读验证、预写工单安全重放和审批过期扫描; -- FastAPI 操作创建、业务事实查询、绑定审批、固定安全错误映射和生产 lifespan 启动恢复。 -- 冻结依赖、`0002` 迁移升降级、审批竞态与 A/B 重启重复、真实 FastMCP + FastAPI 双服务进程和 PostgreSQL 终态事实。 -- WSL2 Ubuntu Compose 的非 root 应用镜像、PostgreSQL/Redis、API/MCP 健康检查、三业务审批/拒绝/唯一工单、数据库断言和 API/MCP 重启恢复。 -- 版本化 42 条固定评测(库存 30、设备 6、作业 6)与 2×2 缓存/工具模式矩阵;小样本只证明调用/命中行为,不作为生产 SLA。 -- 服务端 UUIDv4 request_id、异常后上下文清理、安全 JSON 日志、应用级低基数 Prometheus 指标、SSE 实际回放计数,以及默认关闭的 `/metrics`。 -- Public GitHub remote、只读且固定 Action SHA 的四个 PR 快速门禁,以及 `main` 上实际通过的 Compose 业务 smoke、API/MCP 重启恢复和无条件清理。 -- Netlify 公开静态专题、真实部署资源指纹和证据图片响应验证;该站点不连接 API、数据库或 MCP。 -- 新 Plan-and-Execute Agent 的 Real Kimi 首轮三业务报告为 failed;修复模型/MCP timeout 耦合、thinking/tool-call 兼容和内部结构化提交边界后,三业务只读、库存批准写入和无效 provider fail-closed 代表路径通过。首轮失败报告继续保留为修复前证据;少量通过不解释为准确率、SLA 或成本指标。 -- 冻结 Agent 轨迹评测 9/9,覆盖非法 schema、提示注入、未知工具、对象漂移、RAG 隔离、批准后事实漂移、审批竞态、幂等写入与关键重启;这不是生产准确率。 -- 最新 main 求职展示门禁:后端 667 条测试、前端 19 个测试文件/60 条用例、三业务契约评测、冻结 Agent 评测 9/9、Ruff、mypy、仓库安全和 Compose 重启恢复全部通过;公开专题与本机构建 JS 的 SHA-256 一致。固定用例只证明已声明的合成契约,不是生产准确率或 SLA。 - -新 Agent 核心的修复前失败边界见 [Agent 核心架构交付证据](docs/release-evidence/agent-core-architecture.md),修复后的单根路径与真实 Kimi 代表通过项见 [单根 Agent Loop 与 Case 工作台证据](docs/release-evidence/single-root-agent-loop-case-workspace.md)。旧阶段报告保留为时间顺序证据,不能覆盖后续结果。中文学习入口为 [核心技术手册](docs/learning/opercerta-core-technical-guide.md)、[手动实验手册](docs/learning/opercerta-manual-experiment-guide.md)和[面试讲解](docs/learning/opercerta-interview-guide.md)。这些不是生产 IAM、交互 HTTPS 后端或公开 API 完成声明。 - -## 下一实施边界 - -下一阶段仍只实施 OperCerta。单根 Agent 纠偏、main Compose、静态专题发布与 `v0.1.0-showcase.1` Showcase 预发布均已通过;下一步只做用户手动业务演示、源码讲解与口述掌握检查,并录制 3–5 分钟演示、定稿简历话术。是否建设公网可写 HTTPS 后端仍需单独选择托管环境并审批成本与安全治理。生产 IAM、限流/防滥用、备份、高可用、自动部署和产品级正式 Release 仍待完成。产品生产发布门禁为 `CLOSED`,关闭前不把静态展示误报为生产系统。 - -三业务收口规格与八项 TDD 主计划见 [设计](docs/superpowers/specs/2026-07-20-opercerta-three-business-release-design.md)和[计划](docs/superpowers/plans/2026-07-20-opercerta-three-business-release.md)。历史库存切片设计仍作为可靠性内核演进记录保留。 - -## 实施依据 - -按以下顺序阅读并实施: - -1. [AI Agent 四项目命名设计规格](docs/specs/2026-07-14-agent-project-naming-design.md) -2. [AI Agent 四项目总体设计规格](docs/specs/ai-agent-portfolio-overall-design.md) -3. [AI Agent 四项目作品集组合设计](docs/specs/2026-07-14-agent-portfolio-design.md) -4. [OperCerta 详细设计](docs/specs/2026-07-14-opercerta-design.md) - -新对话的启动约束和交接清单见 [IMPLEMENTATION_HANDOFF.md](IMPLEMENTATION_HANDOFF.md)。 - -## 仓库边界 - -- 本仓库包含从零实现的可靠性内核代码、测试、设计、计划、开发日志、本地证据和公开静态专题;尚不包含可公开写入的完整生产应用。 -- 代码、接口、数据和展示材料均从零实现,只使用公开或合成数据,不复制或依赖任何原单位资产。 -- 性能、准确率、成本和稳定性数字只能引用可复现评测的实测结果。 +# OperCerta + +**English** | [简体中文](README.zh-CN.md) + +[![OperCerta CI](https://github.com/KXHXK/opercerta/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/KXHXK/opercerta/actions/workflows/ci.yml) +![Python](https://img.shields.io/badge/Python-3.12-3776AB?logo=python&logoColor=white) +![React](https://img.shields.io/badge/React-19-61DAFB?logo=react&logoColor=111) +![Docker Compose](https://img.shields.io/badge/Docker_Compose-supported-2496ED?logo=docker&logoColor=white) + +OperCerta is a controlled, auditable operations agent for inventory shortages, +equipment incidents, and blocked operational tasks. It combines bounded LLM +reasoning with deterministic policy, human approval, durable workflow state, +and idempotent business writes. + +[Read-only project page](https://opercerta-kxh.netlify.app) · +[Release `v0.1.0-showcase.1`](https://github.com/KXHXK/opercerta/releases/tag/v0.1.0-showcase.1) + +> **Development status:** the complete three-business workflow runs locally in +> a single-node Docker Compose environment. The public project page is static +> and does not expose the API or database. **Not production-ready:** production +> identity, public ingress, rate limiting, backups, high availability, and +> automated deployment are not implemented. + +## Why OperCerta + +Operational systems often detect an exception before an operator knows how to +resolve it. The evidence may be spread across inventory, equipment, task, and +procedure systems; it may also change while approval is pending. An unrestricted +agent that writes directly to those systems would create unacceptable safety and +audit risks. + +OperCerta separates responsibilities: + +- deterministic detectors discover bounded operational signals; +- an LLM-assisted agent gathers evidence and explains a proposed action; +- policy code decides whether approval is required and validates parameters; +- a human approves a binding of facts, rules, and the proposed plan; +- the system fetches fresh evidence before execution; +- PostgreSQL transactions and unique constraints make retries safe. + +The result is an agent that can assist with investigation and planning without +becoming the authority for high-risk writes. + +## Supported Workflows + +| Workflow | Trigger | Agent investigation | Controlled outcome | +| --- | --- | --- | --- | +| Inventory replenishment | Available stock falls below the reorder point | Read inventory and policy evidence, calculate a bounded recommendation, retrieve the relevant procedure | Create one replenishment work order after approval and fresh-fact validation | +| Equipment maintenance | Equipment is offline or reports an alert | Read equipment status and maintenance policy, retrieve an isolation/repair procedure | Create one maintenance work order, or stop safely when facts no longer match | +| Task recovery | An operational task remains blocked | Read task state and recovery policy, retrieve a recovery procedure | Create one recovery work order, or escalate when automatic recovery is not allowed | + +## Agent Loop + +```mermaid +flowchart LR + UI["React console"] --> API["FastAPI boundary"] + API --> SIGNAL["Deterministic signal scan"] + SIGNAL --> GOAL["Typed goal encoding"] + GOAL --> GRAPH["LangGraph plan-and-execute loop"] + GRAPH --> LLM["LLM reasoning"] + LLM --> POLICY["Tool policy and harness"] + POLICY --> MCP["FastMCP read tools"] + MCP --> FACTS["Business facts and pgvector procedures"] + FACTS --> GRAPH + GRAPH --> HITL["Human approval interrupt"] + HITL --> FRESH["Fresh evidence and verifier"] + FRESH --> WRITE["Controlled idempotent write"] + WRITE --> DB["PostgreSQL and checkpoint state"] + DB --> TRACE["Agent trace, audit, and feedback"] + TRACE --> UI +``` + +The model is not the source of truth for quantities, permissions, or state +transitions. LangGraph owns the durable execution flow; MCP exposes a small +allowlist of typed tools; deterministic code and the database enforce the write +boundary. + +## Architecture + +| Area | Implementation | Responsibility | +| --- | --- | --- | +| Web console | React 19, TypeScript, Vite | Signal inbox, case workspace, approvals, results, trace, and audit views | +| API boundary | FastAPI, Pydantic | Authentication, strict request validation, RBAC, stable errors, health endpoints, SSE audit replay | +| Agent runtime | LangGraph, minimal LangChain Tool Calling | Bounded planning, tool observation loop, interrupt/resume, revalidation, and recovery | +| Tool protocol | FastMCP | Typed inventory, equipment, task, policy, knowledge, and work-order tools | +| Persistence | PostgreSQL 18, pgvector, Alembic | Business truth, approval locks, unique work orders, checkpoints, trace, and procedure retrieval | +| Cache | Redis | Read-only evidence caching; approval-time validation bypasses the cache | +| Model adapter | OpenAI-compatible API | Mock mode by default; explicit timeout and fail-closed behavior in real mode | +| Runtime | Docker Compose | Reproducible PostgreSQL, Redis, MCP, bootstrap, and API services | +| Delivery | GitHub Actions | Repository safety, Python quality, backend tests, frontend tests, and main-branch Compose recovery smoke | + +## Quick Start + +### Prerequisites + +- Linux or WSL2 +- Docker Engine with Docker Compose v2 +- Node.js 24 and npm 11 for the local web console +- `uv` 0.11 and Python 3.12 for source-level tests + +### 1. Configure local-only credentials + +```bash +cp .env.compose.example .env.compose +``` + +Replace every `CHANGE_ME` placeholder in `.env.compose`. Use the same strong +database password in `POSTGRES_PASSWORD` and `OPERCERTA_DATABASE_URL`, generate +a separate signing key, and set the Mock-only model values to a non-secret name +such as `mock` and `not-used-in-mock-mode`. The file is ignored by Git. Mock +model mode is enabled by default and does not require a real API key. + +### 2. Start the backend stack + +```bash +OPERCERTA_HF_HUB_OFFLINE=false docker compose up --build -d --wait +curl http://127.0.0.1:8080/health/ready +``` + +The first run downloads the embedding model. Later runs can set +`OPERCERTA_HF_HUB_OFFLINE=true` when the FastEmbed cache is already populated. +A ready response reports `database`, `checkpoint`, and `mcp` as `ready`. + +### 3. Start the web console + +```bash +cd web +npm ci +npm run dev +``` + +Open . Vite proxies `/api` to the local FastAPI +service, so no browser-side CORS configuration is required. + +## Use the Application + +1. Select the `operator` demo account and scan operational signals. +2. Open an inventory, equipment, or task case and start the Agent investigation. +3. Inspect the typed goal, tool plan, MCP observations, procedure citations, and Agent Trace. +4. Switch to `approver` and approve or reject the bound proposal. +5. Switch to `auditor` to inspect fresh-fact verification, the resulting work order, and the audit sequence. +6. Repeat or restart services to observe idempotency and durable recovery. + +The demo JWT issuer is local-only and is not a production identity system. + +## Model Modes + +- **Mock mode** is deterministic, credential-free, and used for repeatable + contract, safety, and recovery tests. +- **Real mode** uses an OpenAI-compatible endpoint and fails closed when the + provider, output contract, or tool loop is invalid. Secrets stay in ignored + local environment files. + +Representative Moonshot/Kimi K2.6 validation passed for three read-only +business paths, an approved inventory write, and invalid-provider fail-closed. +This limited sample verifies provider compatibility; it is not a model accuracy, +latency, cost, or SLA claim. + +## Verification + +| Gate | Current verified result | +| --- | ---: | +| Backend suite | 667 tests passed | +| Frontend suite | 19 test files, 60 tests passed | +| Three-business fixed contracts | 42/42 passed | +| Frozen Agent safety and recovery evaluation | 9/9 passed | +| Main Compose smoke | Build, business database effects, API/MCP restart, recovery, and cleanup passed | + +Run the local gates: + +```bash +uv sync --frozen --all-groups +uv run ruff check . +uv run ruff format --check . +uv run mypy src +uv run pytest -q +uv run python scripts/run_opercerta_evaluation.py +uv run python scripts/run_agent_evaluation.py + +cd web +npm ci +npm run test:run +npm run build +``` + +With a fresh Compose database, `python3 scripts/verify_agent_compose.py` also +asserts Agent trajectories and the resulting PostgreSQL facts. Fixed synthetic +cases verify declared contracts; they do not represent production traffic or an +independent accuracy benchmark. + +## Reliability and Safety Properties + +- strict input schemas and stable safe error envelopes; +- tool allowlists, typed arguments, bounded retries, and explicit timeouts; +- human approval bound to evidence, rule, fact, and plan hashes; +- approval-time cache bypass and fresh-fact revalidation; +- PostgreSQL row locking for approval races; +- deterministic idempotency keys, unique constraints, and write-after-read verification; +- durable LangGraph checkpoints and business-table-led restart recovery; +- request IDs, trace context, safe structured logs, and low-cardinality metrics; +- synthetic data only; no customer records or confidential company material. + +## Repository Layout + +```text +src/opercerta/ API, Agent runtime, policies, persistence, MCP, and observability +web/ React console and static project page +tests/ Unit, integration, database, API, Agent, and runtime tests +data/ Versioned synthetic evaluation cases and procedure knowledge +migrations/ Alembic database migrations +scripts/ Bootstrap, evaluation, safety, and Compose verification tools +docs/ Technical guides, development records, and release evidence +``` + +## Documentation + +- [Core technical guide](docs/learning/opercerta-core-technical-guide.md) +- [Manual experiment guide](docs/learning/opercerta-manual-experiment-guide.md) +- [Current implementation state](docs/development-log/current-state.md) +- [Single-root Agent loop evidence](docs/release-evidence/single-root-agent-loop-case-workspace.md) +- [GitHub Actions evidence](docs/release-evidence/github-actions-ci.md) + +## Development Status and Roadmap + +Completed locally: + +- three bounded business workflows and the shared Agent loop; +- approval binding, revalidation, idempotent writes, and restart recovery; +- real PostgreSQL/pgvector, Redis, FastMCP, FastEmbed retrieval, and React console; +- fixed contract/evaluation suites and main-branch Compose recovery evidence; +- read-only public project page and a reproducible pre-release. + +Open work before a production deployment: + +- production identity and authorization lifecycle; +- public HTTPS API ingress, exact-origin CORS, rate limiting, and abuse controls; +- managed secrets, backups, restore drills, and high-availability coordination; +- automated deployment, migration orchestration, and operational alerting; +- broader independent model and business-quality evaluation. + +## Contributing + +See [CONTRIBUTING.md](CONTRIBUTING.md) for development setup, quality gates, +pull-request expectations, and Agent safety rules. diff --git a/README.zh-CN.md b/README.zh-CN.md new file mode 100644 index 0000000..f4fa529 --- /dev/null +++ b/README.zh-CN.md @@ -0,0 +1,231 @@ +# OperCerta + +[English](README.md) | **简体中文** + +[![OperCerta CI](https://github.com/KXHXK/opercerta/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/KXHXK/opercerta/actions/workflows/ci.yml) +![Python](https://img.shields.io/badge/Python-3.12-3776AB?logo=python&logoColor=white) +![React](https://img.shields.io/badge/React-19-61DAFB?logo=react&logoColor=111) +![Docker Compose](https://img.shields.io/badge/Docker_Compose-supported-2496ED?logo=docker&logoColor=white) + +OperCerta 是一个面向库存短缺、设备异常和作业阻塞的受控、可审计运营处置 +Agent。它把有限的大模型推理与确定性规则、人工审批、持久化工作流状态和 +幂等业务写入组合成完整闭环。 + +[只读项目页面](https://opercerta-kxh.netlify.app) · +[版本 `v0.1.0-showcase.1`](https://github.com/KXHXK/opercerta/releases/tag/v0.1.0-showcase.1) + +> **开发状态:** 三业务完整流程可以在本地单节点 Docker Compose 环境中运行。 +> 公开项目页面为纯静态页面,不连接 API 或数据库。项目**尚未达到生产就绪**: +> 生产身份、公网入口、限流、备份、高可用和自动部署仍未实现。 + +## 项目背景 + +运营系统通常可以先检测到异常,但操作人员还需要从库存、设备、任务和操作规程 +等多个系统收集证据,才能决定如何处置;审批等待期间,业务事实还可能变化。 +如果让开放式 Agent 直接写业务系统,会产生不可接受的安全和审计风险。 + +OperCerta 对职责进行了拆分: + +- 确定性检测器发现边界明确的运营异常信号; +- LLM 辅助的 Agent 收集证据并解释建议动作; +- 规则代码判断是否需要审批并校验业务参数; +- 人工审批绑定事实、规则和计划快照; +- 执行前重新获取最新事实; +- PostgreSQL 事务与唯一约束保证重试不会重复写入。 + +因此,Agent 可以辅助调查和规划,但不会成为高风险业务写入的最终权威。 + +## 支持的业务流程 + +| 业务 | 触发条件 | Agent 调查 | 受控结果 | +| --- | --- | --- | --- | +| 库存补货 | 可用库存低于补货点 | 读取库存与规则证据、计算有边界的建议数量、检索相关操作规程 | 审批并完成最新事实复核后,只创建一张补货工单 | +| 设备维修 | 设备离线或产生告警 | 读取设备状态与维修规则、检索隔离和维修规程 | 创建一张维修工单;事实不再匹配时安全终止 | +| 作业恢复 | 运营任务持续阻塞 | 读取任务状态与恢复规则、检索恢复规程 | 创建一张恢复工单;不允许自动恢复时升级处理 | + +## Agent 闭环 + +```mermaid +flowchart LR + UI["React 控制台"] --> API["FastAPI 边界"] + API --> SIGNAL["确定性异常扫描"] + SIGNAL --> GOAL["类型化目标编码"] + GOAL --> GRAPH["LangGraph 规划与执行循环"] + GRAPH --> LLM["LLM 推理"] + LLM --> POLICY["工具策略与 Harness"] + POLICY --> MCP["FastMCP 只读工具"] + MCP --> FACTS["业务事实与 pgvector 规程"] + FACTS --> GRAPH + GRAPH --> HITL["人工审批中断"] + HITL --> FRESH["最新事实与 Verifier"] + FRESH --> WRITE["受控幂等写入"] + WRITE --> DB["PostgreSQL 与 checkpoint"] + DB --> TRACE["Agent Trace、审计与反馈"] + TRACE --> UI +``` + +模型不是业务数量、权限或状态转换的事实来源。LangGraph 负责持久化执行流程, +MCP 只开放少量类型化白名单工具,确定性代码和数据库共同守住写入边界。 + +## 技术架构 + +| 区域 | 实现 | 职责 | +| --- | --- | --- | +| Web 控制台 | React 19、TypeScript、Vite | 异常收件箱、Case 工作区、审批、结果、Trace 和审计展示 | +| API 边界 | FastAPI、Pydantic | 身份校验、严格输入、RBAC、稳定错误、健康检查、SSE 审计回放 | +| Agent 运行时 | LangGraph、最小 LangChain Tool Calling | 有界规划、工具观察循环、中断恢复、重新取证和重启恢复 | +| 工具协议 | FastMCP | 类型化库存、设备、任务、规则、知识和工单工具 | +| 持久化 | PostgreSQL 18、pgvector、Alembic | 业务事实、审批锁、唯一工单、checkpoint、Trace 和规程检索 | +| 缓存 | Redis | 只读证据缓存;审批后复核绕过缓存 | +| 模型适配 | OpenAI-compatible API | 默认 Mock;真实模式具有显式超时和 fail-closed 行为 | +| 运行环境 | Docker Compose | 可复现的 PostgreSQL、Redis、MCP、bootstrap 和 API 服务 | +| 持续集成 | GitHub Actions | 仓库安全、Python 质量、后端、前端和 main 分支 Compose 恢复门禁 | + +## 快速启动 + +### 环境要求 + +- Linux 或 WSL2 +- Docker Engine 与 Docker Compose v2 +- 本地控制台需要 Node.js 24 和 npm 11 +- 源码测试需要 `uv` 0.11 和 Python 3.12 + +### 1. 配置仅本地使用的凭据 + +```bash +cp .env.compose.example .env.compose +``` + +替换 `.env.compose` 中的所有 `CHANGE_ME` 占位符:`POSTGRES_PASSWORD` 和 +`OPERCERTA_DATABASE_URL` 必须使用同一个高强度数据库密码,JWT 使用独立签名密钥; +Mock 模式的模型名和 key 可以分别使用 `mock` 与 `not-used-in-mock-mode` 等非敏感值。 +该文件已被 Git 忽略,默认 Mock 模式不需要真实模型 API key。 + +### 2. 启动后端服务 + +```bash +OPERCERTA_HF_HUB_OFFLINE=false docker compose up --build -d --wait +curl http://127.0.0.1:8080/health/ready +``` + +首次运行会下载 embedding 模型。FastEmbed 缓存准备完成后,后续可以设置 +`OPERCERTA_HF_HUB_OFFLINE=true`。就绪响应中的 `database`、`checkpoint` 和 +`mcp` 应全部为 `ready`。 + +### 3. 启动 Web 控制台 + +```bash +cd web +npm ci +npm run dev +``` + +打开 。Vite 会把 `/api` 代理到本机 FastAPI, +因此不需要额外配置浏览器跨域。 + +## 功能用法 + +1. 选择 `operator` 演示账号并扫描业务异常。 +2. 打开库存、设备或作业 Case,启动 Agent 调查。 +3. 查看类型化 Goal、工具计划、MCP Observation、规程引用和 Agent Trace。 +4. 切换到 `approver`,批准或拒绝已绑定的处置建议。 +5. 切换到 `auditor`,查看最新事实复核、工单结果和审计序列。 +6. 重复请求或重启服务,观察幂等写入与持久化恢复。 + +演示 JWT 只用于本地流程,不是生产身份系统。 + +## 模型模式 + +- **Mock 模式**确定、无需凭据,用于可复现的契约、安全和恢复测试。 +- **Real 模式**连接 OpenAI-compatible endpoint;provider、输出契约或工具循环 + 不合法时会 fail closed。密钥只保存在被忽略的本地环境文件中。 + +Moonshot/Kimi K2.6 的少量代表验证已覆盖三业务只读、库存批准写入和无效 +provider fail-closed。该小样本只证明 provider 兼容性,不代表模型准确率、 +延迟、成本或 SLA。 + +## 测试结果 + +| 门禁 | 当前已验证结果 | +| --- | ---: | +| 后端测试 | 667 条通过 | +| 前端测试 | 19 个测试文件、60 条用例通过 | +| 三业务固定契约 | 42/42 通过 | +| 冻结 Agent 安全与恢复评测 | 9/9 通过 | +| main Compose smoke | 构建、业务数据库副作用、API/MCP 重启、恢复和清理通过 | + +运行本地门禁: + +```bash +uv sync --frozen --all-groups +uv run ruff check . +uv run ruff format --check . +uv run mypy src +uv run pytest -q +uv run python scripts/run_opercerta_evaluation.py +uv run python scripts/run_agent_evaluation.py + +cd web +npm ci +npm run test:run +npm run build +``` + +在全新 Compose 数据库上执行 `python3 scripts/verify_agent_compose.py`,还会断言 +Agent 轨迹和 PostgreSQL 最终事实。固定合成用例只验证已声明契约,不代表生产 +流量或独立准确率评测。 + +## 可靠性与安全属性 + +- 严格输入 Schema 和稳定的安全错误 envelope; +- 工具白名单、类型化参数、有限重试和显式 timeout; +- 审批绑定证据、规则、事实和计划哈希; +- 审批后绕过缓存并重新读取最新事实; +- PostgreSQL 行锁解决审批竞态; +- 确定性幂等键、唯一约束和写后读验证; +- 持久化 LangGraph checkpoint 与业务表主导的重启恢复; +- request ID、trace context、安全结构化日志和低基数指标; +- 只使用合成数据,不包含客户记录或原单位机密材料。 + +## 仓库结构 + +```text +src/opercerta/ API、Agent、规则、持久化、MCP 和可观测性 +web/ React 控制台与静态项目页面 +tests/ 单元、集成、数据库、API、Agent 和运行时测试 +data/ 版本化合成评测用例与规程知识 +migrations/ Alembic 数据库迁移 +scripts/ 启动、评测、安全和 Compose 验证工具 +docs/ 技术指南、开发记录和发布证据 +``` + +## 项目文档 + +- [核心技术手册](docs/learning/opercerta-core-technical-guide.md) +- [手动实验手册](docs/learning/opercerta-manual-experiment-guide.md) +- [当前实施状态](docs/development-log/current-state.md) +- [单根 Agent Loop 实施证据](docs/release-evidence/single-root-agent-loop-case-workspace.md) +- [GitHub Actions 证据](docs/release-evidence/github-actions-ci.md) + +## 开发状态与路线图 + +本地已经完成: + +- 三条受控业务流程和共享 Agent Loop; +- 审批绑定、重新取证、幂等写入和重启恢复; +- 真实 PostgreSQL/pgvector、Redis、FastMCP、FastEmbed 检索和 React 控制台; +- 固定契约/评测以及 main 分支 Compose 恢复证据; +- 只读公开项目页面和可复现预发布版本。 + +生产部署前仍需完成: + +- 生产身份和授权生命周期; +- 公网 HTTPS API、精确 CORS、限流和防滥用; +- 托管密钥、备份、恢复演练和高可用协调; +- 自动部署、迁移编排和运营告警; +- 更广泛的独立模型和业务质量评测。 + +## 参与贡献 + +开发环境、质量门禁、Pull Request 要求和 Agent 安全规则见 +[CONTRIBUTING.zh-CN.md](CONTRIBUTING.zh-CN.md)。 diff --git a/docs/development-log/daily/2026-07-31.md b/docs/development-log/daily/2026-07-31.md index 1f8f584..d58f00b 100644 --- a/docs/development-log/daily/2026-07-31.md +++ b/docs/development-log/daily/2026-07-31.md @@ -24,3 +24,11 @@ 2. 独立演示一条完整业务闭环与一次 MCP/进程重启恢复,并能解释数据库后置条件。 3. 录制 3–5 分钟演示,使用面试讲解中的简历表述定稿简历。 4. 上述求职交付完成前继续只处理 OperCerta 收口,不扩展产品生产范围。 + +## 开源项目入口重构 + +- 将根 `README.md` 重构为英文默认项目入口,只介绍 OperCerta 的业务背景、三业务功能、Agent 闭环、技术架构、快速启动、使用方法、验证结果、可靠性边界和开发路线。 +- 新建 `README.zh-CN.md` 作为结构与事实对齐的独立简体中文版;GitHub 页面通过顶部链接切换语言,不再在一个页面同时堆叠中英文正文。 +- 扩展 `CONTRIBUTING.md`,补齐开发环境、TDD、Agent 安全边界、质量门禁、PR 检查表和安全报告;新增等价的 `CONTRIBUTING.zh-CN.md`。 +- README 不再链接面向个人复盘的讲解材料、组合项目设计或过时的原始设计入口;历史文档继续留在仓库和 `DOCUMENT_INDEX.md` 中供追溯,不做破坏性删除。 +- 新增文档防漂移测试,约束中英文互链、关键技术栈、快速启动、真实模型代表验证边界以及不得重新引入明显求职导向内容。 diff --git a/tests/unit/runtime/test_release_assets.py b/tests/unit/runtime/test_release_assets.py index cff47aa..d1af6fd 100644 --- a/tests/unit/runtime/test_release_assets.py +++ b/tests/unit/runtime/test_release_assets.py @@ -107,16 +107,50 @@ def test_learning_pack_covers_three_business_manual_failure_and_interview_explan def test_release_documents_keep_verified_boundaries_truthful() -> None: readme = (ROOT / "README.md").read_text(encoding="utf-8") + readme_zh = (ROOT / "README.zh-CN.md").read_text(encoding="utf-8") state = (ROOT / "docs" / "development-log" / "current-state.md").read_text(encoding="utf-8") assert "Private GitHub" not in readme assert "Private GitHub" not in state - assert "生产发布门禁" in readme - assert "CLOSED" in readme + assert "Not production-ready" in readme + assert "尚未达到生产就绪" in readme_zh assert "真实模型代表性" in state assert "尚未" in state +def test_public_entry_documents_are_bilingual_and_project_focused() -> None: + readme = (ROOT / "README.md").read_text(encoding="utf-8") + readme_zh = (ROOT / "README.zh-CN.md").read_text(encoding="utf-8") + contributing = (ROOT / "CONTRIBUTING.md").read_text(encoding="utf-8") + contributing_zh = (ROOT / "CONTRIBUTING.zh-CN.md").read_text(encoding="utf-8") + + assert "[简体中文](README.zh-CN.md)" in readme + assert "[English](README.md)" in readme_zh + assert "[简体中文](CONTRIBUTING.zh-CN.md)" in contributing + assert "[English](CONTRIBUTING.md)" in contributing_zh + assert readme.count("\n## ") == readme_zh.count("\n## ") + assert contributing.count("\n## ") == contributing_zh.count("\n## ") + assert "## 中文说明" not in readme + assert "## Why OperCerta" not in readme_zh + + for content, heading in ((readme, "Quick Start"), (readme_zh, "快速启动")): + for phrase in ("FastAPI", "LangGraph", "FastMCP", "PostgreSQL", "Redis", heading): + assert phrase in content + + for forbidden in ( + "interview guide", + "resume wording", + "portfolio overview", + "面试讲解", + "简历", + "求职", + "四项目", + "作品集", + ): + assert forbidden.lower() not in readme.lower() + assert forbidden.lower() not in readme_zh.lower() + + def test_agent_delivery_documents_cover_architecture_learning_and_truthful_evidence() -> None: handbook = (ROOT / "docs" / "learning" / "opercerta-core-technical-guide.md").read_text( encoding="utf-8" @@ -172,6 +206,7 @@ def test_agent_delivery_documents_cover_architecture_learning_and_truthful_evide def test_current_demo_and_learning_docs_match_the_single_root_agent_release() -> None: readme = (ROOT / "README.md").read_text(encoding="utf-8") + readme_zh = (ROOT / "README.zh-CN.md").read_text(encoding="utf-8") demo = (ROOT / "docs" / "demo-script.md").read_text(encoding="utf-8") manual = (ROOT / "docs" / "learning" / "opercerta-manual-experiment-guide.md").read_text( encoding="utf-8" @@ -183,10 +218,18 @@ def test_current_demo_and_learning_docs_match_the_single_root_agent_release() -> for content in (demo, manual): assert "扫描业务异常" in content assert "启动 Agent 调查" in content - for content in (readme, demo, interview): + for content in (demo, interview): assert "三业务只读、库存批准写入和无效 provider fail-closed" in content assert "新 Agent 核心的 Real Kimi Tool Calling 代表 query 为 failed" not in content + normalized_readme = " ".join(readme.split()) + normalized_readme_zh = " ".join(readme_zh.split()) + assert ( + "three read-only business paths, an approved inventory write, and " + "invalid-provider fail-closed" in normalized_readme + ) + assert "三业务只读、库存批准写入和无效 provider fail-closed" in normalized_readme_zh + assert "667 条后端测试" in interview assert "v0.1.0-showcase.1" in readme assert "v0.1.0-showcase.1" in interview