From 39050692ce30cd1abe8a3766368cbfefb193fcbc Mon Sep 17 00:00:00 2001 From: Nicholas Santi Date: Sun, 4 Oct 2026 12:26:46 +0000 Subject: [PATCH 1/2] feat(agents): set up Chess-specific Claude Code and Herdr team --- .agents/team/LICENSE-MIT | 21 + .agents/team/README.md | 119 ++++ .agents/team/agents.py | 627 +++++++++++++++++++ .agents/team/fleet.toml | 50 ++ .agents/team/herdr-plugin.toml | 24 + .agents/team/roles/developer.md | 16 + .agents/team/roles/devops-engineer.md | 19 + .agents/team/roles/firmware-engineer.md | 31 + .agents/team/roles/hardware-engineer.md | 36 ++ .agents/team/roles/lead.md | 28 + .agents/team/roles/manufacturing-engineer.md | 22 + .agents/team/roles/mechanical-engineer.md | 33 + .agents/team/roles/pm.md | 6 + .agents/team/roles/pr-maker.md | 12 + .agents/team/roles/pushback.md | 12 + .agents/team/roles/qa.md | 19 + .agents/team/roles/reviewer.md | 16 + .agents/team/roles/test-engineer.md | 19 + .agents/team/roles/upgrade-reviewer.md | 7 + .agents/team/setup.py | 69 ++ .agents/team/team.md | 104 +++ .agents/team/tests/test_herdr_integration.py | 149 +++++ .agents/team/tests/test_team.py | 585 +++++++++++++++++ .agents/team/usage.py | 364 +++++++++++ .claude/settings.json | 10 + .devcontainer/Dockerfile | 22 +- .devcontainer/README.md | 17 +- .devcontainer/devcontainer.json | 7 +- .devcontainer/post-create.sh | 18 +- .github/workflows/ci.yml | 11 +- .github/workflows/pr.yml | 11 +- .gitignore | 4 + AGENTS.md | 51 ++ CLAUDE.md | 1 + README.md | 7 + justfile | 51 +- 36 files changed, 2585 insertions(+), 13 deletions(-) create mode 100644 .agents/team/LICENSE-MIT create mode 100644 .agents/team/README.md create mode 100755 .agents/team/agents.py create mode 100644 .agents/team/fleet.toml create mode 100644 .agents/team/herdr-plugin.toml create mode 100644 .agents/team/roles/developer.md create mode 100644 .agents/team/roles/devops-engineer.md create mode 100644 .agents/team/roles/firmware-engineer.md create mode 100644 .agents/team/roles/hardware-engineer.md create mode 100644 .agents/team/roles/lead.md create mode 100644 .agents/team/roles/manufacturing-engineer.md create mode 100644 .agents/team/roles/mechanical-engineer.md create mode 100644 .agents/team/roles/pm.md create mode 100644 .agents/team/roles/pr-maker.md create mode 100644 .agents/team/roles/pushback.md create mode 100644 .agents/team/roles/qa.md create mode 100644 .agents/team/roles/reviewer.md create mode 100644 .agents/team/roles/test-engineer.md create mode 100644 .agents/team/roles/upgrade-reviewer.md create mode 100755 .agents/team/setup.py create mode 100644 .agents/team/team.md create mode 100644 .agents/team/tests/test_herdr_integration.py create mode 100644 .agents/team/tests/test_team.py create mode 100755 .agents/team/usage.py create mode 100644 .claude/settings.json create mode 100644 AGENTS.md create mode 100644 CLAUDE.md diff --git a/.agents/team/LICENSE-MIT b/.agents/team/LICENSE-MIT new file mode 100644 index 00000000..e67ac463 --- /dev/null +++ b/.agents/team/LICENSE-MIT @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Nicholas Santi + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/.agents/team/README.md b/.agents/team/README.md new file mode 100644 index 00000000..56f3c0af --- /dev/null +++ b/.agents/team/README.md @@ -0,0 +1,119 @@ +# Chess agent team + +Optional Herdr runtime alongside the existing [Pi setup](../../.pi/README.md). +All fourteen configured roles use Claude Code by default. The eight regular roles +are lead, portable Rust developer, firmware engineer, hardware engineer, mechanical +engineer, QA, reviewer and pushback. Manufacturing engineer, test engineer, DevOps +engineer, PM, upgrade reviewer and PR maker start on demand, with at most two alongside +the full default roster (ten active agents maximum). + +Opus handles electrical/mechanical/DFM reasoning, lead decisions and review; Sonnet +handles bounded software/tooling/test work. See [team.md](team.md) for the ownership +matrix and interface handoffs. There is no student role, generic scenario role or +mandatory approval chain. Start only agents relevant to the task; roles start idle. +Starting the fleet may consume subscription/API usage. + +## Setup + +Rebuild/reopen the devcontainer after pulling these changes. The image pins +Claude Code 2.1.287 and checksum-verified Herdr 0.9.3. Post-create installs +Codex 0.159.3, Herdr's provider hooks, Reviewr v0.39.0 and the Chess menu plugin. +`just agents-setup` repeats runtime integration setup (network required for Reviewr). +Reviewr is optional: a marketplace outage warns without blocking the container or team; +retry setup later to install the panel. +Setup never signs you in, starts agents, makes model calls, or pre-trusts the +checkout. Run `claude` once and accept workspace trust yourself on first use. +If startup is blocked or times out, the launcher keeps that pane for inspection instead +of destroying the dialog. Attach to the session, answer prompts yourself, then +`just agents-stop ROLE` and `just agents ROLE` to complete a clean startup/brief. + +Login inside the container: + +```sh +claude auth login +# Optional alternate harness: +codex login +just agents-doctor +just agents-list +just agents lead developer --dry-run +just agents lead developer +# PCB/enclosure work without unrelated software roles: +just agents lead hardware-engineer mechanical-engineer +# Pi adapter/runtime work: +just agents lead firmware-engineer +# Or start the complete default team: +just agents +``` + +Claude/Codex credentials and Herdr configuration use separate named Docker +volumes, not host credential bind mounts. Login survives a container rebuild, +not volume deletion. Host OAuth can instead be forwarded with +`CLAUDE_CODE_OAUTH_TOKEN`; never commit it. Pi's existing volumes/settings remain +unchanged. `ANTHROPIC_API_KEY` is also forwarded for Pi: unset it before container +creation if you want Claude subscription billing rather than API billing. +Doctor warns about API credentials; it does not make a model request. + +## Lifecycle + +```sh +just agents --no-attach # start/reuse without opening the UI +herdr --session chess # attach; Ctrl+B Q detaches, agents keep running +just agents reviewer --session chess # add only a missing role +just agents pr-maker --no-attach # only with a delivery assignment +just agents-stop pr-maker # release an on-demand slot +just agents --fresh --no-attach # new conversations, retain task checkpoint +just agents-reset --no-attach # new conversations, clear task checkpoint +just agents-stop # close this team's workspace only +just agents-usage --hours 24 # local recorded tokens, not plan allowance +just agents-check # lint, format and offline regressions +just agents-test # just the offline regression suite +HERDR_TEST_BIN="$(command -v herdr)" just agents-test # isolated native API smoke tests +``` + +Use `--session NAME` on launcher commands for a separate named session. Do not +run duplicate sessions for the same task. Names permit ASCII letters, digits, +underscores and hyphens, starting with a letter/digit. `list`/`--dry-run` work +without installed providers or login. Mutations require a container. Workspace labels +include a stable checkout-path hash so another clone/worktree's team cannot be mistaken +for this one's. Herdr agent names remain server-scoped: use different sessions for +concurrent checkouts. Explicit named-session commands ignore inherited foreign socket, +workspace and machine settings; plugin actions retain their validated invoking context. +If you already launched the earlier generic `chess-team` roster, stop it in Herdr before +starting this checkout-bound roster; old roles are not automatically migrated/evicted. + +`fleet.toml` owns model aliases, effort, provider adapter, role map and cap. +To use Codex for a work kind set `harness = "codex"`, `model = "-"`, and a +supported effort such as `medium`, then log in to Codex. Restart affected roles +to apply changes; `up` deliberately preserves existing conversations. +Per-role briefs and private reports/checkpoints live in `target/agents/`. +The launcher locks each session during mutations, enforces the cap across +incremental starts, rejects foreign role collisions and cleans definitively failed +new panes. Blocked/timed-out startup panes are kept for operator intervention, not +silently accepted or retried. +Menu actions are bound to their invoking session/socket/workspace. + +Provider permission checks remain enabled. If you knowingly want unattended +permission bypass, pass `--dangerous` to `agents`/`agents-reset` explicitly. +A container is not a complete sandbox: agents still see credentials, mounted +source, the network and the in-container Docker daemon. The menu never enables +bypass, and restarting without this flag restores ordinary permission behavior. + +`just check`, `just quality` and the normal pre-commit hook include `agents-check`. +CI always runs offline tests; native smoke tests are opt-in, use temporary Git repositories +and isolated Herdr servers, and never launch a provider or make model requests. + +Follow [team.md](team.md) and [AGENTS.md](../../AGENTS.md): one writer per file, +one Git owner, one expensive-check owner, and no publication before approval +when the user requests unstaged review. No automatic task dispatch or merging. + +## Provenance + +Launcher and local token reporting are adapted from +[`nicksan222/study`](https://github.com/nicksan222/study/tree/500b4c563c571db21e9a2f86d740506fbf816bed/.agents/team) +at `500b4c563c571db21e9a2f86d740506fbf816bed`; setup hook normalization follows +its devcontainer setup. The source MIT notice is retained in `LICENSE-MIT`. +The roster/briefs are Chess-specific: native PCB electronics, Blender mechanics, +Pi Linux firmware, DFM/prototype evidence, Linux/SPICE/mesh tests and Yocto/toolchains. +Study's learner/scenario roles, desktop-specific skills and third-party prompt plugins +are not imported. Portable persistence work follows the actual owning crate, not a +presumed Study database stack. diff --git a/.agents/team/agents.py b/.agents/team/agents.py new file mode 100755 index 00000000..9b966875 --- /dev/null +++ b/.agents/team/agents.py @@ -0,0 +1,627 @@ +#!/usr/bin/env python3 +"""Manage Chess's project-wide Herdr team (adapted from nicksan222/study). + +Lifecycle: validate configuration and selectors, check provider login, lock the named +session, reuse or create the owned workspace, start only missing roles, then release +the lock before attaching. Plugin actions additionally prove their socket and workspace. +""" + +import argparse +import fcntl +import hashlib +import json +import os +import shutil +import subprocess +import sys +import time +import tomllib +from pathlib import Path + +TEAM = Path(__file__).resolve().parent +REPO = TEAM.parent.parent +# The effort levels each harness accepts ('-' keeps its default). +EFFORTS = { + "claude": {"low", "medium", "high", "xhigh", "max"}, + "codex": {"minimal", "low", "medium", "high", "xhigh"}, +} + + +class LauncherError(Exception): + def __init__(self, msg, code=2): + super().__init__(msg) + self.code = code + + +class StartupPending(LauncherError): + """Keep a provider pane available for operator startup/trust intervention.""" + + +def transport_environment(plugin=False): + """Explicit sessions must not inherit another pane's socket/machine context.""" + environment = os.environ.copy() + if not plugin: + keep = {"HERDR_CONFIG_PATH", "HERDR_BIN_PATH", "HERDR_LOG"} + for key in list(environment): + if key.startswith("HERDR_") and key not in keep: + environment.pop(key) + return environment + + +def call(arguments, capture=False, check=True, **options): + """Run one argv command without involving a shell.""" + run_options = {"text": True, **options} + if capture: + run_options.update(stdout=subprocess.PIPE, stderr=subprocess.PIPE) + try: + result = subprocess.run(arguments, check=False, **run_options) + except OSError as error: + raise LauncherError(f"cannot run {arguments[0]}: {error}", 1) from error + if check and result.returncode: + if capture: + sys.stderr.write(result.stderr or result.stdout) + raise LauncherError(f"command failed: {arguments[0]}", result.returncode) + return result + + +def protocol_error_code(result): + """Herdr 0.9.3 writes API failures to stderr, successes to stdout.""" + for stream in (result.stderr, result.stdout): + try: + payload = json.loads(stream) + except json.JSONDecodeError: + continue + if isinstance(payload, dict) and isinstance(payload.get("error"), dict): + return payload["error"].get("code") + return None + + +class Fleet: + """Own configuration, current transport, private state, and its mutation lock.""" + + def __init__(self, options): + with (TEAM / "fleet.toml").open("rb") as config_file: + config = tomllib.load(config_file) + self.options = options + self.session = config["session"] if options.session is None else options.session + # Herdr exposes checkout metadata only for managed worktrees. Bind ordinary + # checkout workspaces too, without guessing ownership from a generic label. + checkout_key = hashlib.sha256(str(REPO.resolve()).encode()).hexdigest()[:12] + self.workspace = f"{config['workspace']}-{checkout_key}" + self.maximum = config["max_agents"] + self.reviewr = config["reviewr"] + self.defaults = config["default_agents"] + self.kinds = config["kinds"] + self.roles = config["roles"] + self.herdr = os.getenv("HERDR_BIN_PATH", "herdr") + self.plugin = options.plugin + self.lockfile = None + if self.plugin: + self.plugin_context() + self.state = REPO / "target" / "agents" / self.session + self.validate() + + def validate(self): + """Reject invalid configuration and selectors before any Herdr mutation.""" + valid = set("abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789_-") + valid_session = ( + isinstance(self.session, str) + and bool(self.session) + and len(self.session) <= 64 + and self.session[0].isalnum() + and set(self.session) <= valid + ) + if not valid_session: + raise LauncherError("invalid session name") + if type(self.maximum) is not int or self.maximum < 1: + raise LauncherError("max_agents must be positive") + if ( + not isinstance(self.defaults, list) + or not self.defaults + or not all(isinstance(role, str) for role in self.defaults) + ): + raise LauncherError("default_agents must be a nonempty list of role names") + if len(self.defaults) > self.maximum: + raise LauncherError("default_agents exceeds max_agents") + if len(set(self.defaults)) != len(self.defaults): + raise LauncherError("default_agents contains a duplicate role") + if ( + not isinstance(self.roles, dict) + or not self.roles + or not all( + isinstance(role, str) and isinstance(work, str) + for role, work in self.roles.items() + ) + ): + raise LauncherError("roles must map role names to work kinds") + if not isinstance(self.kinds, dict): + raise LauncherError("kinds must map work kinds to provider settings") + unknown_defaults = set(self.defaults) - set(self.roles) + if unknown_defaults: + raise LauncherError(f"unknown default agent: {min(unknown_defaults)}") + for role, work in self.roles.items(): + valid_role = role[0:1].islower() and all( + character.islower() or character.isdigit() or character == "-" + for character in role + ) + if not valid_role: + raise LauncherError(f"invalid role: {role}") + if not (TEAM / "roles" / f"{role}.md").is_file(): + raise LauncherError(f"missing role: {role}") + settings = self.kinds.get(work, {}) + if not isinstance(settings, dict) or settings.get("harness") not in EFFORTS: + raise LauncherError(f"invalid work settings: {work}") + model = settings.get("model") + if not isinstance(model, str) or not model: + raise LauncherError(f"invalid model for work kind: {work}") + effort = settings.get("effort") + if effort != "-" and effort not in EFFORTS[settings["harness"]]: + raise LauncherError(f"invalid effort for work kind {work}: {effort}") + unknown = set(self.options.roles) - set(self.roles) - {"team"} + if unknown: + raise LauncherError( + f"unknown agent: {min(unknown)} (see list for configured roles)" + ) + if self.options.command == "reset" and self.options.roles: + raise LauncherError( + "reset always replaces the whole team; omit role selectors" + ) + if self.options.fresh and self.options.roles: + raise LauncherError("--fresh restarts the whole team; omit agent selectors") + if self.plugin and self.options.command not in {"up", "reset", "down"}: + raise LauncherError("plugin mode supports only up, reset or down") + + def plugin_context(self): + """Bind a menu action to the session socket and workspace that invoked it.""" + expected = "stop" if self.options.command == "down" else self.options.command + action = os.getenv("HERDR_PLUGIN_ACTION_ID", "").rsplit(".", 1)[-1] + if os.getenv("HERDR_PLUGIN_ID") != "chess.team": + raise LauncherError("plugin action is not chess.team") + if action != expected: + raise LauncherError( + f"plugin action {action} cannot run {self.options.command}" + ) + socket = os.getenv("HERDR_SOCKET_PATH") + if not socket or not os.getenv("HERDR_WORKSPACE_ID"): + raise LauncherError("plugin action has no workspace transport context") + status = json.loads(call([self.herdr, "status", "--json"], capture=True).stdout) + if status.get("server", {}).get("socket") != socket: + raise LauncherError("plugin socket does not match the active Herdr session") + self.session = status.get("server", {}).get("session") or status.get( + "client", {} + ).get("session") + if not self.session: + raise LauncherError("Herdr did not report the plugin session name") + + def herdr_command(self, *arguments, capture=True, check=True): + """Use the selected transport; keep protocol JSON out of human-facing output.""" + command = [self.herdr] + if not self.plugin: + command.extend(["--session", self.session]) + command.extend(arguments) + return call( + command, + capture=capture, + check=check, + env=transport_environment(self.plugin), + ) + + def herdr_json(self, *arguments): + """Call Herdr and decode its structured response.""" + response = self.herdr_command(*arguments, capture=True) + return json.loads(response.stdout) + + def selected_roles(self): + names = set(self.options.roles or self.defaults) + if "team" in names: + names.update(self.defaults) + return [(role, work) for role, work in self.roles.items() if role in names] + + def validate_container(self): + if not (Path("/.dockerenv").exists() or Path("/run/.containerenv").exists()): + raise LauncherError("run inside the devcontainer (never on the host)") + + def doctor(self): + """Check tools and provider login without making a model request.""" + self.validate_container() + for tool in [self.herdr, "python3", "just", "cargo"]: + if not shutil.which(tool): + raise LauncherError(f"missing {tool}: rebuild the devcontainer") + call([self.herdr, "--version"], env=transport_environment(self.plugin)) + seen = set() + for _, work in self.selected_roles(): + provider = self.kinds[work]["harness"] + if provider in seen: + continue + seen.add(provider) + if not shutil.which(provider): + raise LauncherError(f"missing {provider}: rebuild the devcontainer") + call([provider, "--version"]) + if provider == "claude": + result = call(["claude", "auth", "status"], capture=True, check=False) + if result.returncode: + raise LauncherError( + "Claude is not signed in: run claude auth login" + ) + auth = json.loads(result.stdout) + if not auth.get("loggedIn"): + raise LauncherError( + "Claude is not signed in: run claude auth login" + ) + if os.getenv("ANTHROPIC_API_KEY") or os.getenv("ANTHROPIC_AUTH_TOKEN"): + print( + "Warning: API credentials are set; Claude may use API billing instead of your subscription." + ) + print("Claude login found (no model request made).") + else: + call(["codex", "login", "status"]) + print("Container ready (no model request made).") + + def lock(self): + """Take the private, per-session mutation lock without waiting.""" + previous_umask = os.umask(0o077) + try: + (self.state / "reports").mkdir(parents=True, exist_ok=True) + self.lockfile = (self.state / "launcher.lock").open("a+") + finally: + os.umask(previous_umask) + try: + fcntl.flock(self.lockfile, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError as error: + raise LauncherError( + f"another launcher is changing session {self.session}" + ) from error + + def unlock(self): + if self.lockfile: + fcntl.flock(self.lockfile, fcntl.LOCK_UN) + self.lockfile.close() + self.lockfile = None + + def server_running(self): + """Recognize Herdr's first status line while allowing its metadata lines.""" + response = self.herdr_command("status", "server", capture=True, check=False) + return ( + response.returncode == 0 + and "status: running" in response.stdout.splitlines() + ) + + def ensure_server(self): + if self.server_running(): + return + environment = transport_environment(self.plugin) + # Remove nested-agent markers while preserving subscription OAuth credentials. + markers = { + "CLAUDECODE", + "CLAUDE_CODE_ENTRYPOINT", + "CLAUDE_CODE_CHILD_SESSION", + "CLAUDE_CODE_EFFORT_LEVEL", + } + for key in list(environment): + if key in markers: + environment.pop(key) + subprocess.Popen( + [self.herdr, "--session", self.session, "server"], + env=environment, + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + start_new_session=True, + close_fds=True, + ) + for _ in range(50): + if self.server_running(): + return + time.sleep(0.1) + raise LauncherError("Herdr server did not start") + + def workspace_id(self): + items = self.herdr_json("workspace", "list")["result"]["workspaces"] + found = [item for item in items if item.get("label") == self.workspace] + if len(found) > 1: + raise LauncherError( + "duplicate team workspaces; resolve them in Herdr first" + ) + if not found: + return "" + checkout = (found[0].get("worktree") or {}).get("checkout_path") + if checkout and Path(checkout).resolve() != REPO.resolve(): + raise LauncherError( + "team workspace checkout does not match this repository; " + "use a different --session (no workspace was changed)" + ) + return found[0]["workspace_id"] + + def guard_workspace(self, workspace_id): + if self.plugin and ( + not workspace_id or workspace_id != os.getenv("HERDR_WORKSPACE_ID") + ): + raise LauncherError( + "plugin action did not originate in the Chess team workspace" + ) + + def require_ready(self, result, role, pane): + """Preserve a live startup pane if the provider needs operator intervention.""" + if not result.returncode: + return + code = protocol_error_code(result) + if code in {"agent_not_ready", "agent_blocked", "timeout"}: + raise StartupPending( + f"{role} startup needs inspection ({code}); kept pane {pane}. " + f"Attach with herdr --session {self.session}, answer any trust/permission " + f"prompt yourself, then stop/restart this role to finish its brief." + ) + sys.stderr.write(result.stderr or result.stdout) + raise LauncherError(f"failed to start {role}", result.returncode) + + def start_role(self, role, work, pane): + """Write one private role brief, then start its provider in a new pane.""" + settings = self.kinds[work] + prompt = self.state / f"{role}.md" + try: + team_brief = (TEAM / "team.md").read_text() + role_brief = (TEAM / "roles" / f"{role}.md").read_text() + handoff = ( + f"\nHerdr session: {self.session}. " + f"Use herdr --session {self.session} for every message.\n" + f"Handoff: {self.state}/task.md (lead owns it; keep it brief).\n" + f"Reports: {self.state}/reports (one -.md per report).\n" + ) + prompt.write_text(team_brief + role_brief + handoff) + prompt.chmod(0o600) + except OSError as error: + raise LauncherError(str(error), 1) from error + provider = settings["harness"] + if provider == "claude": + provider_arguments = [ + "--append-system-prompt-file", + str(prompt), + "--disallowedTools", + "Agent,Workflow", + ] + else: + provider_arguments = [] + if self.options.dangerous: + flag = ( + "--dangerously-skip-permissions" + if provider == "claude" + else "--dangerously-bypass-approvals-and-sandbox" + ) + provider_arguments.append(flag) + if settings["model"] != "-": + provider_arguments += ["--model", settings["model"]] + if settings["effort"] != "-": + if provider == "claude": + provider_arguments.extend(["--effort", settings["effort"]]) + else: + effort = f'model_reasoning_effort="{settings["effort"]}"' + provider_arguments.extend(["-c", effort]) + result = self.herdr_command( + "agent", + "start", + role, + "--kind", + provider, + "--pane", + pane, + "--timeout", + "60000", + "--", + *provider_arguments, + check=False, + ) + self.require_ready(result, role, pane) + if provider == "codex": + result = self.herdr_command( + "agent", "prompt", role, prompt.read_text(), check=False + ) + self.require_ready(result, role, pane) + print(f"Started {role}: {provider} {settings['model']} ({settings['effort']}).") + + def up(self, reset=False): + """Start missing roles, or replace every conversation for reset/fresh.""" + selected_roles = self.selected_roles() + if not selected_roles: + raise LauncherError("no agents selected") + if len(selected_roles) > self.maximum: + raise LauncherError(f"request exceeds max_agents={self.maximum}") + if self.options.dry_run: + self.roster(selected_roles) + return + self.doctor() + self.lock() + self.ensure_server() + workspace_id = self.workspace_id() + agents = self.herdr_json("agent", "list")["result"]["agents"] + requested_names = {role for role, _ in selected_roles} + for agent in agents: + belongs_elsewhere = agent.get("workspace_id") != workspace_id + if agent.get("name") in requested_names and belongs_elsewhere: + raise LauncherError(f"{agent['name']} belongs to another workspace") + owned = { + agent["name"] + for agent in agents + if agent.get("workspace_id") == workspace_id + } + self.guard_workspace(workspace_id) + if self.options.fresh or reset: + if workspace_id: + self.herdr_command("workspace", "close", workspace_id) + if reset: + (self.state / "task.md").unlink(missing_ok=True) + workspace_id = "" + owned = set() + # Count the union so several incremental starts cannot bypass the team cap. + if len(owned | requested_names) > self.maximum: + raise LauncherError(f"team would exceed MAX_AGENTS={self.maximum}") + self.herdr_command("plugin", "link", str(TEAM)) + first = False + for role, work in selected_roles: + if role in owned: + print(f"{role} already present; keeping its conversation.") + continue + if not workspace_id: + response = self.herdr_json( + "workspace", + "create", + "--cwd", + str(REPO), + "--label", + self.workspace, + "--no-focus", + ) + created = response["result"] + workspace_id = created["workspace"]["workspace_id"] + pane = created["root_pane"]["pane_id"] + self.herdr_command("tab", "rename", created["tab"]["tab_id"], role) + first = True + else: + response = self.herdr_json( + "tab", + "create", + "--workspace", + workspace_id, + "--cwd", + str(REPO), + "--label", + role, + "--no-focus", + ) + created = response["result"] + pane = created["root_pane"]["pane_id"] + try: + self.start_role(role, work, pane) + except StartupPending: + raise + except LauncherError: + self.herdr_command("pane", "close", pane, check=False) + print(f"failed to start {role}; removed its new pane", file=sys.stderr) + raise + if role == "lead": + self.herdr_command("agent", "focus", "lead") + review_panel = self.herdr_command( + "plugin", + "action", + "invoke", + "open", + "--plugin", + self.reviewr, + check=False, + ) + if review_panel.returncode: + print("reviewr unavailable; run just agents-setup.") + self.herdr_command("workspace", "focus", workspace_id) + if "lead" in requested_names: + self.herdr_command("agent", "focus", "lead") + if first: + if reset: + print("New conversations started with no prior handoff.") + else: + print( + "New conversations: the lead can recover the task from its handoff." + ) + # Attaching can last for hours; never retain the mutation lock in the UI. + self.unlock() + if not self.options.no_attach: + os.execvpe( + self.herdr, + [self.herdr, "--session", self.session], + transport_environment(self.plugin), + ) + + def down(self): + """Close owned panes or the complete owned workspace.""" + self.validate_container() + self.lock() + if not self.server_running(): + print("No team is running.") + return + workspace_id = self.workspace_id() + if not workspace_id: + print("No team workspace is open.") + return + self.guard_workspace(workspace_id) + if not self.options.roles: + self.herdr_command("workspace", "close", workspace_id) + print(f"Closed {self.workspace} in {self.session}.") + return + agents = self.herdr_json("agent", "list")["result"]["agents"] + for role, _ in self.selected_roles(): + for agent in agents: + if ( + agent.get("name") == role + and agent.get("workspace_id") == workspace_id + ): + self.herdr_command("pane", "close", agent["pane_id"]) + + def roster(self, items=None): + print( + f"Session: {self.session} | workspace: {self.workspace} | maximum agents: {self.maximum}" + ) + width = max(map(len, self.roles)) + 2 + for role, work in items or self.roles.items(): + settings = self.kinds[work] + print( + f"{role:<{width}} {settings['harness']:<8} {settings['model']:<24} {settings['effort']}" + ) + + +def parse(): + parser = argparse.ArgumentParser() + subcommands = parser.add_subparsers(dest="command", required=True) + for name in ("up", "reset", "down", "doctor", "list"): + command_parser = subcommands.add_parser(name) + command_parser.add_argument("--session") + command_parser.add_argument( + "--plugin", action="store_true", help=argparse.SUPPRESS + ) + command_parser.add_argument("roles", nargs="*") + command_parser.set_defaults( + fresh=False, no_attach=False, dry_run=False, dangerous=False + ) + if name in {"up", "reset"}: + command_parser.add_argument("--no-attach", action="store_true") + command_parser.add_argument("--dry-run", action="store_true") + command_parser.add_argument( + "--dangerous", + action="store_true", + help="bypass provider permission checks inside the container (opt-in)", + ) + if name == "up": + command_parser.add_argument("--fresh", action="store_true") + return parser.parse_args() + + +def main(): + options = parse() + fleet = Fleet(options) + try: + if options.command == "up": + fleet.up() + elif options.command == "reset": + fleet.up(reset=True) + elif options.command == "down": + fleet.down() + elif options.command == "doctor": + fleet.doctor() + else: + fleet.roster() + finally: + fleet.unlock() + + +if __name__ == "__main__": + try: + main() + except LauncherError as error: + print(error, file=sys.stderr) + raise SystemExit(error.code) + except ( + OSError, + KeyError, + TypeError, + tomllib.TOMLDecodeError, + json.JSONDecodeError, + ) as error: + print(f"invalid launcher data: {error}", file=sys.stderr) + raise SystemExit(2) diff --git a/.agents/team/fleet.toml b/.agents/team/fleet.toml new file mode 100644 index 00000000..58b27da9 --- /dev/null +++ b/.agents/team/fleet.toml @@ -0,0 +1,50 @@ +# Eight regular roles; up to two task-specific specialists at once. +# Availability is not a requirement to assign every role to every task. +session = "chess" +workspace = "chess-team" +max_agents = 10 +reviewr = "persiyanov.reviewr" +default_agents = [ + "lead", "developer", "firmware-engineer", "hardware-engineer", + "mechanical-engineer", "qa", "reviewer", "pushback", +] + +# CLI aliases avoid assuming access to a particular model release. +# Set harness = "codex", model = "-" to use Codex's configured default. +[kinds.build] +harness = "claude" +model = "sonnet" +effort = "medium" + +[kinds.reasoning] +harness = "claude" +model = "opus" +effort = "high" + +[kinds.review] +harness = "claude" +model = "opus" +effort = "medium" + +# Electrical/mechanical decisions require explicit reasoning and cited evidence. +[kinds.engineering] +harness = "claude" +model = "opus" +effort = "high" + +[roles] +lead = "reasoning" +developer = "build" +firmware-engineer = "build" +hardware-engineer = "engineering" +mechanical-engineer = "engineering" +qa = "build" +reviewer = "review" +pushback = "review" +# Start only for a concrete task; these do not consume slots until started. +manufacturing-engineer = "engineering" +test-engineer = "build" +devops-engineer = "build" +pm = "build" +upgrade-reviewer = "review" +pr-maker = "build" diff --git a/.agents/team/herdr-plugin.toml b/.agents/team/herdr-plugin.toml new file mode 100644 index 00000000..fbfdc663 --- /dev/null +++ b/.agents/team/herdr-plugin.toml @@ -0,0 +1,24 @@ +id = "chess.team" +name = "Chess team" +version = "0.1.0" +min_herdr_version = "0.9.3" +description = "Manage Chess's project-wide agent team in the devcontainer" +platforms = ["linux"] + +[[actions]] +id = "up" +title = "Start missing Chess agents" +contexts = ["workspace"] +command = ["python3", "agents.py", "up", "--plugin", "--no-attach"] + +[[actions]] +id = "reset" +title = "Reset all agent context" +contexts = ["workspace"] +command = ["python3", "agents.py", "reset", "--plugin", "--no-attach"] + +[[actions]] +id = "stop" +title = "Stop all Chess agents" +contexts = ["workspace"] +command = ["python3", "agents.py", "down", "--plugin"] diff --git a/.agents/team/roles/developer.md b/.agents/team/roles/developer.md new file mode 100644 index 00000000..ecd871bd --- /dev/null +++ b/.agents/team/roles/developer.md @@ -0,0 +1,16 @@ +# Developer — portable Rust logic + +Read `crates/README.md` and the owning crate README before implementing an assigned slice +in chess rules/state/history, menu navigation, shared core, persistence or logging. +Keep these crates independent of physical board/OS adapters. Product-specific menu and +runtime/device behavior belong in `apps/firmware`, owned by firmware engineer. + +Read owning code and callers, coordinate API changes with firmware engineer, and keep +changes small and type-safe. Do not duplicate chess/runtime logic in tests or introduce +a new hardware abstraction merely to mirror PCB generation. Stored formats/errors and +cross-crate behavior need focused regression coverage and upgrade review when relevant. + +Only edit assigned files; do not take over PCB/CAD/shared electrical contracts. Run +focused owning-crate checks, report paths/results/interface changes, and wait for review +before the next slice. Fix findings within the slice. Ask lead before broadening scope; +do not own Git or repeat final workspace checks unassigned. diff --git a/.agents/team/roles/devops-engineer.md b/.agents/team/roles/devops-engineer.md new file mode 100644 index 00000000..9a0ff2da --- /dev/null +++ b/.agents/team/roles/devops-engineer.md @@ -0,0 +1,19 @@ +# DevOps engineer — toolchains, CI and Yocto (on demand) + +Read `.devcontainer/README.md`, `.github/README.md`, `docs/development.md`, +`apps/firmware/README.md` and relevant Yocto configuration before assigned changes. +Own bounded devcontainer/toolchain, Just/hook, CI artifact/cache, Yocto image/recipe, +system integration packaging or agent-runtime tasks. Coordinate runtime/package/service +requirements with firmware engineer; this is not permission to rewrite application code. + +Preserve pinned/checksum-verified tools, frozen lockfiles, normal commit gates and required +CI checks. Keep credentials out of image layers, configuration, logs and artifacts. +Separate offline tooling tests from authenticated/provider/model requests. Do not install +third-party prompt plugins or silently switch a governed Pi workflow to another runner. +Coordinate expensive Docker/QEMU/Yocto jobs with the lead; avoid duplicate builds. + +Check changed configuration syntax and focused tooling tests first. AArch64 binary +linkage, BitBake parsing/dry-run and complete image builds are separate evidence; none +proves physical Pi behavior. Report exact commands, versions, image/artifact identity, +network/auth limitations and upgrade implications. Never publish images/releases, flash +hardware, weaken branch checks or change repository settings without authorization. diff --git a/.agents/team/roles/firmware-engineer.md b/.agents/team/roles/firmware-engineer.md new file mode 100644 index 00000000..cc22e52e --- /dev/null +++ b/.agents/team/roles/firmware-engineer.md @@ -0,0 +1,31 @@ +# Firmware engineer — Raspberry Pi Linux runtime and adapters + +## Read before matching work + +- `apps/firmware/README.md`, `docs/host.md`, relevant module and crate READMEs. +- `hardware/shared/README.md` and current wiring/Hall-bank contracts for device work. + +## Scope and boundaries + +Own assigned `apps/firmware/src/` runtime, typed events, Linux GPIO/I2C/SPI adapters, +button debounce, buffered SSD1306 display, Hall polling/board reconciliation, LED +brightness limits, NetworkManager provisioning and systemd/watchdog integration. +This is a Linux process on a Pi Zero 2 W, not bare-metal microcontroller firmware. +Use the production event loop/harness for E2E tests; Tokio channels stay inside events. +Keep portable chess/menu logic in its crates and OS/device behavior in app adapters. + +Verify current implementations before claiming adapters are wired into startup. Pi pin +identities are hand-maintained, not generated from PCB output; changes require explicit +coordination with hardware engineer. Coordinate portable APIs with developer and Yocto +packages, kernel/device configuration and services with DevOps engineer. Do not add a +second PCB-to-Rust generator, shell Wi-Fi control or a duplicated test runtime. + +## Evidence and stopping point + +Run assigned focused Rust tests and `just firmware-binary` when the dependency graph or +native linkage changes. Linux E2E uses Docker/QEMU and can be expensive: obtain check +ownership from the lead and share results. Scripted fixtures, simulator PNGs, Linux +VM tests, AArch64 linkage, Yocto metadata and physical Pi tests establish different +things; label each accurately. Report changed interfaces, test results and untested +physical assumptions. Never flash a Pi, change credentials/network state or drive real +GPIO/power without explicit authorization. Stop at the assigned slice and report to lead. diff --git a/.agents/team/roles/hardware-engineer.md b/.agents/team/roles/hardware-engineer.md new file mode 100644 index 00000000..b2c40396 --- /dev/null +++ b/.agents/team/roles/hardware-engineer.md @@ -0,0 +1,36 @@ +# Hardware engineer — electronics and PCB + +## Read before matching work + +- `hardware/shared/README.md`, `hardware/pcb/README.md`. +- `docs/hardware.md`, `docs/power.md`, `docs/host.md` for circuit/power/acquisition changes. +- Exact approved-part datasheets from `hardware/shared/components.py`; verify claims in + those sources and current code. Historical prose can be stale. + +## Scope and boundaries + +Own assigned electrical design: approved components and typed pins, native `pcbnew` +footprints/connectivity, Hall-bank address straps and square mapping, I2C pull-ups and +bus loading, SPI/LED level shifting and chain order, OLED/buttons, power protection, +current/thermal margins, track/via/copper rules, decoupling and test access. +The Pi Zero 2 W is the sole processor; do not invent a microcontroller or sensor IRQ. +Hall acquisition is polled across eight TCA9554 banks. Check exact parts/mappings in +shared code rather than copying numbers from documentation. + +Author `hardware/shared/{components,electronics,hall_banks,wiring}` and +`hardware/pcb/definition/` only as assigned. Native board pads/nets are the electrical +source, not a parallel schematic model; schematic/netlist/BOM are derived output. +Coordinate shared dimensions with mechanical engineer and hand-maintained Pi pin +identities/acquisition behavior with firmware engineer. Do not edit their files or +shared contracts without explicit file ownership from the lead. + +## Evidence and stopping point + +Run assigned shared checks and PCB `check`/`review` recipes; coordinate heavy generation +with QA. Trace changed nets/pins and inspect ERC/DRC plus focused SPICE outcomes, including +startup, approved/full-white load, fault cases and logic thresholds where relevant. +Report datasheet passages, assumptions, calculations, source paths, changed interfaces, +commands and remaining bench measurements. A schematic that passes checks is not proof +of power safety, Hall/magnet margin or a working physical prototype. Never fabricate +measurements, bypass `definition/evidence/`, release fabrication or order boards. Stop +for missing datasheet/physical evidence or cross-domain decisions; send one report to lead. diff --git a/.agents/team/roles/lead.md b/.agents/team/roles/lead.md new file mode 100644 index 00000000..9d0c4a32 --- /dev/null +++ b/.agents/team/roles/lead.md @@ -0,0 +1,28 @@ +# Lead + +Read `AGENTS.md`, `.agents/team/team.md`, `docs/development.md` and the affected package +READMEs before planning. Own the requested outcome, user communication, integration and +Git until an explicit authorized PR-maker handoff. Keep the session checkpoint current. + +Route portable crate work to developer, Pi runtime/adapters to firmware engineer, +circuit/PCB/power to hardware engineer, and case/fit/optical stack to mechanical engineer. +For cross-domain changes, agree the interface with both owners, name one writer per +shared contract/file and verify both sides. Do not assume docs or generated outputs are +current authoring sources. Preserve the one-Pi architecture and physical-evidence gates. + +Start only useful colleagues. Use QA for independent behavior/evidence verification, +reviewer for correctness and pushback for consequential assumptions. Manufacturing, tests, +DevOps, product scope, upgrade review and PR delivery are on-demand roles with two spare +slots, not mandatory stages. Stop an idle optional role only after preserving its handoff; +never evict a working agent automatically. No nested teams or duplicate sessions. + +Assign one small slice at a time, explicit paths, acceptance/evidence and stopping point. +Review it before the next slice. One owner runs expensive PCB/CAD generation, Linux E2E +and Yocto tasks; integrate and run the combined appropriate checks once. Do not turn +image builds or fabrication release into the default fast validation loop. Distinguish +software/simulation/render evidence from actual physical measurements. + +Respect existing work and authorization. Unstaged review means no staging, commits, +pushes or PR yet. Never bypass hooks, physical measurements, trust/permission dialogs or +publish/order/flash hardware on behalf of a vague implementation request. Report concise +results, verification, unresolved engineering decisions and meaningful limitations. diff --git a/.agents/team/roles/manufacturing-engineer.md b/.agents/team/roles/manufacturing-engineer.md new file mode 100644 index 00000000..c695c021 --- /dev/null +++ b/.agents/team/roles/manufacturing-engineer.md @@ -0,0 +1,22 @@ +# Manufacturing engineer — DFM, sourcing and prototype evidence (on demand) + +Read `hardware/pcb/README.md`, `hardware/cad/README.md`, `docs/fabrication.md`, +`docs/assembly.md`, `docs/power.md` and `hardware/pcb/definition/evidence/` before work. +Check current component/dimension/rule sources; generated BOM and layout are evidence +for that build, not authoring inputs or proof of physical correctness. + +Own an assigned DFM/assembly review or a bounded source/documentation change: exact MPNs, +packages/footprint compatibility, supply availability, substitutions, board/print-service +capabilities, edge/copper/drill/slot requirements, panel and connector access, fasteners, +assembly sequence, inspectability, test points, rework and one-square prototype plans. +Substitutions require hardware engineer approval of electrical/footprint compatibility; +mechanical changes require mechanical engineer agreement. Do not copy supplier prices or +availability without a dated source. Do not promise a process/tolerance not confirmed by +the manufacturer. Read-only unless the lead assigns explicit paths to edit. + +For prototype work, specify measurable acceptance, instruments, operating/fault limits +and missing evidence. Only record actual operator-supplied or authorized bench results, +with provenance. PCB `review` is not `release`: missing Hall/magnet evidence must keep +fabrication export blocked. Never fabricate measurements, bypass the release gate, +order parts/boards, send files to a vendor or claim production readiness. Report DFM +findings, citations, revision/artifact identity and remaining physical validation to lead. diff --git a/.agents/team/roles/mechanical-engineer.md b/.agents/team/roles/mechanical-engineer.md new file mode 100644 index 00000000..6127cdd8 --- /dev/null +++ b/.agents/team/roles/mechanical-engineer.md @@ -0,0 +1,33 @@ +# Mechanical engineer — enclosure, fit and optical stack + +## Read before matching work + +- `hardware/cad/README.md`, `hardware/shared/README.md`. +- `docs/cad.md`, `docs/assembly.md` and the owning `hardware/cad/projects/` README. + +## Scope and boundaries + +Own assigned case/tile-plate Python generators, printable geometry, fasteners, board +supports, Pi/control clearances, connector access, tolerance stack, FDM printability, +assembly/service access and light diffusion. Shared dimensions and coordinates belong +in `hardware/shared/dimensions.py`; CAD-only modeling/presentation stays in CAD. +Coordinates centre on the playing area, and printable models use assembly coordinates. +Each printable object has one owning generator; assembly imports it instead of redefining +it. The PCB proxy is a render stand-in, not the electrical/placement source of truth. + +Coordinate populated PCB heights/keepouts with hardware engineer and manufacturing +constraints with manufacturing engineer. Hall-to-magnet distance, LED-to-plate spacing +and material/finish choices need evidence, not attractive renders. Do not change the +shared stack, PCB footprint positions or connectors without an agreed interface handoff. +Do not invent an alternate per-square enclosure or assume a desktop printer can fit the +current two large parts. Presentation materials do not specify purchased materials. + +## Evidence and stopping point + +Run assigned dimension/fast tests first, then coordinate `hardware/cad` generation with +QA. Check manifold/positive-volume meshes, scale, bounding boxes, assembly alignment, +clearance/tolerance calculations and generated views. Never hand-edit `.blend`/PNG +outputs or silently weaken validation. Report source paths, dimensions changed, commands, +artifacts and required real print/fit/optical measurements. Printable geometry is not +proof of fit, diffusion or Hall detection on a manufactured board. Stop for missing +manufacturing inputs or cross-domain conflicts and report to the lead. diff --git a/.agents/team/roles/pm.md b/.agents/team/roles/pm.md new file mode 100644 index 00000000..25bd09c4 --- /dev/null +++ b/.agents/team/roles/pm.md @@ -0,0 +1,6 @@ +# PM + +Turn the request into concrete acceptance criteria and small, ordered outcomes. Check +scope against Chess's physical-board goals and current prototype status. Distinguish +requirements from assumptions. Send concise decisions/questions to the lead; do not +implement, control Git or impose a heavyweight approval process. diff --git a/.agents/team/roles/pr-maker.md b/.agents/team/roles/pr-maker.md new file mode 100644 index 00000000..f8820985 --- /dev/null +++ b/.agents/team/roles/pr-maker.md @@ -0,0 +1,12 @@ +# PR maker + +Wait for the lead's verified delivery handoff and exclusive Git ownership. Confirm remote, +base/head, owned paths, final diff and evidence. Local-only or unstaged instructions win: +do not stage, commit, push or create a PR until explicitly approved. Otherwise stage only +assigned paths, inspect the staged diff, commit through the normal hook, push the confirmed +feature branch and open/update one PR with `gh`. Check for an existing PR before retrying. +Never include unrelated work, bypass hooks, force-push, merge or change repository settings. +Explain behavior, significant decisions, tests and limitations. Agent/tooling-only changes +need no visual demo; hardware screenshots are review artifacts, not proof of fabrication. +Read back URL/head/body and distinguish pending CI from passing. Return URL, commit and +verification/blockers to the lead, then idle. Do not repeat valid checks unassigned. diff --git a/.agents/team/roles/pushback.md b/.agents/team/roles/pushback.md new file mode 100644 index 00000000..9598a585 --- /dev/null +++ b/.agents/team/roles/pushback.md @@ -0,0 +1,12 @@ +# Pushback — architecture and engineering assumptions + +Read the assigned consequential decision and current evidence. Challenge unsupported +physical/power/thermal/manufacturing assumptions, new processors/protocols, duplicated +contracts/generators, test-only runtime paths, unnecessary abstractions and scope growth. +For electrical or mechanical limits, cite current code/datasheets and ask the relevant +engineer for missing evidence rather than pretending to know the bench result. + +Propose the smallest credible alternative and explain the trade-off, reversibility and +remaining uncertainty. Distinguish real safety/contract risks from preferences. Send one +bounded read-only result to lead; do not veto by taste, edit implementation, expand scope, +poll peers or impose another mandatory approval chain. diff --git a/.agents/team/roles/qa.md b/.agents/team/roles/qa.md new file mode 100644 index 00000000..42476fc3 --- /dev/null +++ b/.agents/team/roles/qa.md @@ -0,0 +1,19 @@ +# QA — independent verification + +Read the affected package README and `.agents/team/team.md` before verifying the assigned +acceptance criteria on final code/inputs. Review the developer/engineer/test-engineer +handoff; confirm its revision/diff before reusing results. QA verifies behavior independently, +while test engineer authors fixtures/harnesses; request missing test states through lead. +Read-only implementation scope unless lead explicitly assigns a test/documentation path. + +Choose evidence appropriate to risk: portable Rust tests, production runtime E2E/button/ +display behavior, shared contract/dimension checks, native ERC/DRC/parity and SPICE, CAD +mesh/assembly views, Linux VM, AArch64 linkage or Yocto metadata. Do not duplicate every +check. Obtain ownership before heavy generation, Docker/QEMU or image tasks; report exact +commands, inputs, artifacts/results and remaining limitations. Validate negative/startup/ +fault cases and peer interface contracts, not just a successful screenshot. + +Never substitute simulation/render/check output for bench measurements or printed fit. +Physical release evidence, Pi flashing, vendor uploads and electrical experiments require +explicit authorization and appropriate engineering guidance. Never reset user data or +relax a gate for a demo. Send one evidence-based report to lead; reviewers reuse it. diff --git a/.agents/team/roles/reviewer.md b/.agents/team/roles/reviewer.md new file mode 100644 index 00000000..ae0b6188 --- /dev/null +++ b/.agents/team/roles/reviewer.md @@ -0,0 +1,16 @@ +# Reviewer — independent correctness and contract review + +Read the assigned slice, current owning sources/callers and package README. Review +read-only: bugs, errors/startup/fault behavior, missing tests, persisted compatibility +and cross-domain regressions. Give file/line evidence and severity, not speculative cleanup. + +For hardware, trace shared exact-part/pin/dimension inputs to native PCB connectivity or +owning CAD generator, not a parallel graph or manually edited output. Check explicit +firmware pin handoff, sensor/LED mapping, power assumptions, stack/keepouts and evidence +gates where affected. Use engineering peer reports/datasheets for domain-specific limits; +flag unknown physical behavior instead of approving it from a render or clean DRC. +For runtime, check production event/harness boundaries and Linux versus simulated behavior. + +Reuse QA/check evidence, request missing relevant specialist review through lead, and +recheck only affected findings after fixes. No implementation edits, Git operations, +unassigned full-suite repeats or automatic release approval. Send one report to lead. diff --git a/.agents/team/roles/test-engineer.md b/.agents/team/roles/test-engineer.md new file mode 100644 index 00000000..6985c08b --- /dev/null +++ b/.agents/team/roles/test-engineer.md @@ -0,0 +1,19 @@ +# Test engineer — regression harnesses and reproducible fixtures (on demand) + +Read the owning package README and test recipes, plus `apps/firmware/README.md` for +runtime/Linux tests, `hardware/pcb/README.md` for SPICE, or `hardware/cad/README.md` for +mesh/dimension checks. Build assigned automated tests and fixtures; QA independently +verifies the resulting behavior. Do not create a second implementation to test itself. + +Use the production firmware runtime/harness and maintained GPIO/display/Linux simulation +interfaces, shared hardware contracts, native PCB validation and real generator outputs +as appropriate. Cover negative paths, startup/failure states, contract boundaries and +reported regressions, not snapshot volume or mirrored implementation details. Obtain +ownership of test paths and heavy check execution from lead. Do not mutate another +agent's code or the shared checkout to prove a test fails; use an isolated copy. + +Record fixture inputs, code/diff identity, deterministic reproduction, observed results +and limitations. A simulated pin/button/display/SPICE outcome is not a measured circuit, +Pi integration or manufactured fit. Never invent physical/model responses, overwrite +user data, flash hardware or bypass physical-evidence gates. Send one report to lead, +then idle; QA and PR maker can reuse current evidence without rerunning it unassigned. diff --git a/.agents/team/roles/upgrade-reviewer.md b/.agents/team/roles/upgrade-reviewer.md new file mode 100644 index 00000000..6706fec7 --- /dev/null +++ b/.agents/team/roles/upgrade-reviewer.md @@ -0,0 +1,7 @@ +# Upgrade reviewer + +Read-only check of the assigned change against the base: persisted data, serialized +values, firmware pin maps, shared physical/electrical contracts, toolchain and contributor +setup. Identify what an existing builder or contributor must migrate/regenerate. Check firmware/electrical mapping handoffs and old generated outputs against current +contracts; a contract upgrade does not prove a new physical board revision works. Report +concrete breakage and missing upgrade guidance once, then release an on-demand slot. diff --git a/.agents/team/setup.py b/.agents/team/setup.py new file mode 100755 index 00000000..3b3993ee --- /dev/null +++ b/.agents/team/setup.py @@ -0,0 +1,69 @@ +#!/usr/bin/env python3 +"""Install runtime Herdr integrations; never trust a workspace or start a model.""" + +import json +import subprocess +from pathlib import Path + + +def portable_claude_hooks(home): + """Normalize installer commands and deduplicate hooks on repeated setup.""" + path = home / ".claude/settings.json" + if not path.exists(): + return + settings = json.loads(path.read_text()) + installed = { + f'bash "{home}/.claude/hooks/herdr-agent-state.sh" session', + f"bash '{home}/.claude/hooks/herdr-agent-state.sh' session", + } + for event, groups in settings.get("hooks", {}).items(): + unique = [] + for group in groups: + for hook in group.get("hooks", []): + if hook.get("command") in installed: + hook["command"] = ( + 'bash "$HOME/.claude/hooks/herdr-agent-state.sh" session' + ) + if group not in unique: + unique.append(group) + settings["hooks"][event] = unique + path.write_text(json.dumps(settings, indent=2) + "\n") + + +def main(): + if not (Path("/.dockerenv").exists() or Path("/run/.containerenv").exists()): + raise SystemExit("run setup inside the devcontainer") + home = Path.home() + for directory in (".claude", ".codex", ".config/herdr"): + (home / directory).mkdir(parents=True, exist_ok=True) + for provider in ("claude", "codex"): + subprocess.run(["herdr", "integration", "install", provider], check=True) + portable_claude_hooks(home) + reviewr = subprocess.run( + [ + "herdr", + "plugin", + "install", + "persiyanov/herdr-reviewr", + "--ref", + "v0.39.0", + "--yes", + ], + check=False, + ) + if reviewr.returncode: + print( + "warning: optional Reviewr installation failed; retry just agents-setup when GitHub is reachable" + ) + subprocess.run( + ["herdr", "plugin", "link", str(Path(__file__).resolve().parent)], check=True + ) + for tool in ("claude", "codex", "herdr"): + subprocess.run([tool, "--version"], check=True) + print( + "Herdr configured. Run claude auth login (or codex login), then just agents-doctor." + ) + + +if __name__ == "__main__": + main() diff --git a/.agents/team/team.md b/.agents/team/team.md new file mode 100644 index 00000000..2e5f2da2 --- /dev/null +++ b/.agents/team/team.md @@ -0,0 +1,104 @@ +# Chess engineering team + +Read `AGENTS.md`, your role brief and the owning package documentation. Work in the +devcontainer. This is a physical chessboard driven by one Pi Linux process, with native +KiCad electrical design and Python-generated Blender mechanics—not a desktop study app. + +## Roster and ownership + +The default roster has eight roles. Availability is not a requirement to use every role: +small tasks should start only the needed agents. At most ten agents may run together, +leaving two concurrent slots for the six on-demand roles. The lead chooses those roles +for concrete outcomes; do not auto-evict workers, create per-domain/nested teams or raise +the cap without operator approval. Start missing roles with `just agents --no-attach`. + +| Role | Responsibility | Normal authoring scope (only assigned files) | +| --- | --- | --- | +| lead | Requirements, interface decisions, integration and Git | Checkpoint, integration fixes | +| developer | Portable chess/menu/core/persistence/logger logic | `crates/` and callers agreed with firmware | +| firmware-engineer | Pi runtime, typed events, Linux device/network adapters | `apps/firmware/src/` and assigned tests | +| hardware-engineer | Circuit/power/pins, native PCB/connectivity/routing | `hardware/pcb/definition/`, shared electrical contracts | +| mechanical-engineer | Enclosure/tile plate, tolerance/optical/assembly stack | `hardware/cad/`, shared dimensions | +| qa | Independent behavior/evidence verification | Reports; assigned validation execution | +| reviewer | Read-only correctness/contracts/regression review | Reports | +| pushback | Read-only challenge to architecture and physical assumptions | Reports | +| manufacturing-engineer (on demand) | DFM, sourcing, assembly/prototype/release evidence | Reports; explicitly assigned source/docs | +| test-engineer (on demand) | Test harnesses, negative cases, reproducible fixtures | Assigned test paths | +| devops-engineer (on demand) | Devcontainer, CI, toolchain, Yocto packaging | `.devcontainer/`, `.github/`, tooling, Yocto | +| pm (on demand) | Product acceptance and scope decisions | Reports | +| upgrade-reviewer (on demand) | Persisted data, contract and contributor upgrade risks | Reports | +| pr-maker (on demand) | Authorized Git/PR delivery using existing evidence | Explicitly handed-off Git paths | + +These are responsibility boundaries, not blanket write permission. The lead assigns the +question/outcome, allowed files, acceptance check and stopping point. One writer per +file, one Git owner and one expensive-check owner. Reviewers/QA/pushback do not modify +implementation by default. Unassigned agents stay idle and send no status prompts. + +## Interface changes + +- Electronics → firmware: share net/pin identities, voltage/polarity, timing, Hall-bank + addresses/mapping, LED order and brightness/power limits. Pi pin declarations are + hand-maintained; coordinate both sides rather than inventing PCB-to-Rust generation. +- PCB → mechanics: share populated heights, keepouts, supports, holes, connector access + and magnet/sensor/LED stack. Only agreed cross-domain measurements belong in + `hardware/shared/dimensions.py`; tool behavior stays with its domain. +- Design → manufacturing: confirm exact approved parts, footprint/process compatibility, + print/board tolerance and assembly/prototype acceptance. A render or clean DRC does + not authorize a part substitution or fabrication. +- Runtime → image: coordinate required devices/kernel/packages/services between firmware + and DevOps. A host/VM test, AArch64 link and Yocto image validate different boundaries. + +For such a change, the lead identifies the affected peer, agrees the interface, assigns +one writer for each shared file, and includes the peer's evidence in review. Do not pass +broad ownership of `hardware/shared` to two engineers at once. Verify current code and +exact datasheets: historical documentation may contain obsolete counts or claims. + +## Execution and evidence + +Work in small slices: implement, focused test, report, read-only review, fix, then the +next slice. No fixed approval chain; use the perspectives relevant to the changed risk. +Never hand-edit generated hardware output, weaken checks or reproduce a failure by +resetting someone else's work. Generated PCB/CAD sets have one assigned producer. + +Run focused package checks before combined validation. The lead grants one runner for +PCB `review`, CAD generation, Docker/QEMU E2E and Yocto tasks, then reuses the results. +Full PCB `release`, Yocto image builds, physical flashing, vendor uploads and purchasing +are separate authorized actions, not routine QA. Missing physical evidence remains a +blocker; never replace it with synthetic measurements or relaxed validation. + +Every report labels its evidence: source/datasheet calculation, software unit test, +SPICE, mesh/render, Linux simulation, AArch64 linkage, Yocto metadata/image or actual +operator/bench observation. Record code/diff identity, exact inputs/commands, result and +untested limits. None of the automated categories proves physical electrical safety, +Hall/magnet margin, printed fit or a working manufactured board. + +Write reports to the `Reports:` path in your brief as `-.md`. Send the file +as one quoted argument; never paste arbitrary text into a shell command: + +```sh +session=chess # replace with the session in your brief +report=target/agents/"$session"/reports/hardware-engineer-result.md +herdr --session "$session" agent prompt lead "$(cat -- "$report")" +``` + +Only lead talks to the user and updates `task.md`: objective, starting branch/preexisting +changes, interface decisions, owned files/check runner, current step, evidence and next +action (under 60 lines). Inspect the current diff before recovering a checkpoint or +reusing test-engineer fixtures. Reports/checkpoints are gitignored under +`target/agents/`. Old evidence is a pointer to revalidate, not a current fact. + +Herdr's sidebar owns status. Do not poll or wake models for progress. Send blockers once +with a concrete decision needed. Lead may inspect/re-brief a stalled pane. Do not answer +provider permission/trust dialogs on the user's behalf or silently enable bypass. + +## Delivery + +Lead verifies the combined result and hands PR maker exclusive Git ownership only when +publication is authorized. Handoff includes objective, owned paths, base/head/remote, +diff/commit identity, check evidence, interface/review findings and physical limitations. +Local-only/unstaged instructions forbid staging, committing, pushing and opening a PR +until the user explicitly approves. Never auto-merge or sweep unrelated changes. + +`fleet.toml` owns models/effort: Opus for electrical/mechanical/DFM reasoning, lead and +review; Sonnet for bounded software/tooling/test tasks. `just agents-usage` reports local +recorded Claude tokens, not remaining allowance; Claude Code `/usage` is authoritative. diff --git a/.agents/team/tests/test_herdr_integration.py b/.agents/team/tests/test_herdr_integration.py new file mode 100644 index 00000000..1f0f90c0 --- /dev/null +++ b/.agents/team/tests/test_herdr_integration.py @@ -0,0 +1,149 @@ +"""Optional native Herdr smoke tests; never start a provider or make model calls. + +HERDR_TEST_BIN=/absolute/path/to/herdr just agents-test +The normal offline suite skips these tests when that variable is not set. +""" + +import json +import os +import subprocess +import tempfile +import time +import unittest +from pathlib import Path +from unittest.mock import patch + +from test_team import agents, options + + +@unittest.skipUnless( + os.environ.get("HERDR_TEST_BIN"), "set HERDR_TEST_BIN for native smoke tests" +) +class HerdrIntegrationTests(unittest.TestCase): + def setUp(self): + self.binary = str(Path(os.environ["HERDR_TEST_BIN"]).resolve()) + self.temp = tempfile.TemporaryDirectory(prefix="chess-herdr-test-") + self.addCleanup(self.temp.cleanup) + self.home = Path(self.temp.name) + self.repo = self.home / "repo" + self.repo.mkdir() + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + config = self.home / "config.toml" + config.write_text("onboarding = false\n") + self.environment = { + key: value + for key, value in os.environ.items() + if not key.startswith("HERDR_") + } + self.environment.update(HOME=str(self.home), HERDR_CONFIG_PATH=str(config)) + self.session = "chess-smoke-test" + self.server = subprocess.Popen( + [self.binary, "--session", self.session, "server"], + env=self.environment, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + self.addCleanup(self.stop_server) + for _ in range(50): + if self.command("status", "server").returncode == 0: + break + if self.server.poll() is not None: + self.fail("isolated Herdr server exited before readiness") + time.sleep(0.1) + else: + self.fail("isolated Herdr server did not become ready") + for patcher in ( + patch.dict(os.environ, self.environment, clear=True), + patch.dict(os.environ, {"HERDR_BIN_PATH": self.binary}), + patch.object(agents, "REPO", self.repo), + ): + patcher.start() + self.addCleanup(patcher.stop) + self.fleet = agents.Fleet(options(command="down", session=self.session)) + self.addCleanup(self.fleet.unlock) + + def command(self, *arguments): + return subprocess.run( + [self.binary, "--session", self.session, *arguments], + env=self.environment, + text=True, + capture_output=True, + timeout=15, + check=False, + ) + + def stop_server(self): + if self.server.poll() is None: + self.command("server", "stop") + try: + self.server.wait(timeout=5) + except subprocess.TimeoutExpired: + self.server.terminate() + self.server.wait(timeout=5) + + def workspace(self, label): + result = self.command( + "workspace", + "create", + "--cwd", + str(self.repo), + "--label", + label, + "--no-focus", + ) + self.assertEqual(result.returncode, 0, result.stderr) + return json.loads(result.stdout)["result"]["workspace"]["workspace_id"] + + def remaining_workspaces(self): + result = self.command("workspace", "list") + self.assertEqual(result.returncode, 0, result.stderr) + return { + item["workspace_id"] + for item in json.loads(result.stdout)["result"]["workspaces"] + } + + def test_checkout_binding_transport_and_owned_only_stop(self): + owned = self.workspace(self.fleet.workspace) + foreign = self.workspace("chess-team") + with patch.dict( + os.environ, + {"HERDR_SOCKET_PATH": "/foreign.sock", "HERDR_SESSION": "foreign"}, + ): + self.assertEqual(self.fleet.workspace_id(), owned) + self.fleet.down() + self.assertEqual(self.remaining_workspaces(), {foreign}) + + def test_native_error_response_is_read_from_stderr(self): + # An absent pane fails before process launch, even if Claude is installed. + result = self.command( + "agent", + "start", + "never-launched", + "--kind", + "claude", + "--pane", + "nonexistent", + ) + self.assertNotEqual(result.returncode, 0) + self.assertEqual(result.stdout, "") + self.assertEqual(agents.protocol_error_code(result), "agent_pane_not_found") + + def test_plugin_action_uses_verified_socket_session_and_workspace(self): + owned = self.workspace(self.fleet.workspace) + foreign = self.workspace("other-team") + status = json.loads(self.command("status", "--json").stdout) + context = { + "HERDR_PLUGIN_ID": "chess.team", + "HERDR_PLUGIN_ACTION_ID": "chess.team.stop", + "HERDR_WORKSPACE_ID": owned, + "HERDR_SESSION": self.session, + "HERDR_SOCKET_PATH": status["server"]["socket"], + } + with patch.dict(os.environ, context): + plugin = agents.Fleet(options(command="down", session=None, plugin=True)) + try: + self.assertEqual(plugin.session, self.session) + plugin.down() + finally: + plugin.unlock() + self.assertEqual(self.remaining_workspaces(), {foreign}) diff --git a/.agents/team/tests/test_team.py b/.agents/team/tests/test_team.py new file mode 100644 index 00000000..d413fd9f --- /dev/null +++ b/.agents/team/tests/test_team.py @@ -0,0 +1,585 @@ +"""Offline regressions: no provider process, model request or real session mutation.""" + +import argparse +import contextlib +import importlib.util +import io +import json +import os +import sys +import tempfile +import unittest +from datetime import UTC, datetime +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import Mock, patch + +TEAM = Path(__file__).resolve().parents[1] + + +def load(name): + spec = importlib.util.spec_from_file_location(name, TEAM / f"{name}.py") + module = importlib.util.module_from_spec(spec) + sys.modules[name] = module + spec.loader.exec_module(module) + return module + + +agents = load("agents") +usage = load("usage") +setup = load("setup") + + +def options(**overrides): + values = { + "command": "up", + "roles": [], + "session": None, + "plugin": False, + "fresh": False, + "no_attach": True, + "dry_run": False, + "dangerous": False, + } + return argparse.Namespace(**(values | overrides)) + + +class LauncherTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.fleet = agents.Fleet(options()) + self.fleet.state = Path(self.temp.name) / "state" + self.addCleanup(self.fleet.unlock) + self.start_result = SimpleNamespace(returncode=0, stdout="", stderr="") + + def fake_up(self, existing=(), workspace="ws"): + fleet = self.fleet + fleet.doctor = Mock() + fleet.ensure_server = Mock() + fleet.workspace_id = Mock(return_value=workspace) + fleet.herdr_json = Mock(return_value={"result": {"agents": list(existing)}}) + fleet.herdr_command = Mock(return_value=SimpleNamespace(returncode=0)) + fleet.start_role = Mock() + return fleet + + def test_default_roster_and_role_files(self): + self.assertEqual( + {role for role, _ in self.fleet.selected_roles()}, + { + "lead", + "developer", + "firmware-engineer", + "hardware-engineer", + "mechanical-engineer", + "qa", + "reviewer", + "pushback", + }, + ) + self.assertEqual( + set(self.fleet.roles) - set(self.fleet.defaults), + { + "manufacturing-engineer", + "test-engineer", + "devops-engineer", + "pm", + "upgrade-reviewer", + "pr-maker", + }, + ) + self.assertEqual(self.fleet.maximum - len(self.fleet.defaults), 2) + self.assertEqual( + {path.stem for path in (TEAM / "roles").glob("*.md")}, + set(self.fleet.roles), + ) + self.assertTrue( + all(work["harness"] == "claude" for work in self.fleet.kinds.values()) + ) + + def test_invalid_session_and_unknown_role(self): + for session in ("../escape", "", "-bad", "x" * 65): + with self.subTest(session=session), self.assertRaises(agents.LauncherError): + agents.Fleet(options(session=session)) + with self.assertRaisesRegex(agents.LauncherError, "unknown agent"): + agents.Fleet(options(roles=["missing"])) + + def test_fresh_and_reset_reject_partial_selectors(self): + for override in ({"fresh": True}, {"command": "reset"}): + with ( + self.subTest(override=override), + self.assertRaises(agents.LauncherError), + ): + agents.Fleet(options(roles=["lead"], **override)) + + def test_dry_run_never_runs_preflight_or_creates_state(self): + self.fleet.options.dry_run = True + self.fleet.doctor = Mock(side_effect=AssertionError("preflight called")) + with contextlib.redirect_stdout(io.StringIO()): + self.fleet.up() + self.assertFalse(self.fleet.state.exists()) + + def test_requested_cap_checked_before_preflight(self): + self.fleet.options.roles = list(self.fleet.roles) + self.fleet.doctor = Mock() + with self.assertRaisesRegex(agents.LauncherError, "max_agents"): + self.fleet.up() + self.fleet.doctor.assert_not_called() + + def test_incremental_starts_cannot_exceed_cap(self): + existing = [ + {"name": name, "workspace_id": "ws"} + for name in self.fleet.defaults + ["test-engineer", "devops-engineer"] + ] + fleet = self.fake_up(existing) + fleet.options.roles = ["pr-maker"] + with self.assertRaisesRegex(agents.LauncherError, "MAX_AGENTS"): + fleet.up() + fleet.start_role.assert_not_called() + fleet.herdr_command.assert_not_called() + + def test_foreign_role_conflict_does_not_mutate(self): + fleet = self.fake_up([{"name": "lead", "workspace_id": "other"}]) + with self.assertRaisesRegex(agents.LauncherError, "another workspace"): + fleet.up() + fleet.herdr_command.assert_not_called() + + def test_existing_roles_reused_and_lock_released(self): + existing = [ + {"name": name, "workspace_id": "ws"} for name in self.fleet.defaults + ] + fleet = self.fake_up(existing) + with contextlib.redirect_stdout(io.StringIO()): + fleet.up() + fleet.start_role.assert_not_called() + self.assertIsNone(fleet.lockfile) + self.assertEqual(fleet.state.stat().st_mode & 0o777, 0o700) + + def test_mutation_lock_is_exclusive(self): + self.fleet.lock() + other = agents.Fleet(options()) + other.state = self.fleet.state + self.addCleanup(other.unlock) + with self.assertRaisesRegex(agents.LauncherError, "another launcher"): + other.lock() + + def test_claude_argv_and_private_brief(self): + self.fleet.lock() + self.fleet.herdr_command = Mock(return_value=self.start_result) + with contextlib.redirect_stdout(io.StringIO()): + self.fleet.start_role("developer", "build", "pane") + argv = self.fleet.herdr_command.call_args.args + self.assertIn("--append-system-prompt-file", argv) + self.assertIn("sonnet", argv) + self.assertNotIn("--dangerously-skip-permissions", argv) + prompt = self.fleet.state / "developer.md" + self.assertEqual(prompt.stat().st_mode & 0o777, 0o600) + self.assertIn("Herdr session: chess.", prompt.read_text()) + + def test_codex_blocked_brief_keeps_startup_pending(self): + self.fleet.lock() + self.fleet.kinds["build"] = { + "harness": "codex", + "model": "-", + "effort": "medium", + } + blocked = SimpleNamespace( + returncode=1, stdout="", stderr='{"error":{"code":"agent_blocked"}}' + ) + self.fleet.herdr_command = Mock(side_effect=[self.start_result, blocked]) + with self.assertRaises(agents.StartupPending): + self.fleet.start_role("developer", "build", "new") + self.assertEqual( + self.fleet.herdr_command.call_args.args[:2], ("agent", "prompt") + ) + + def test_permission_bypass_is_explicit(self): + self.fleet.lock() + self.fleet.options.dangerous = True + self.fleet.herdr_command = Mock(return_value=self.start_result) + with contextlib.redirect_stdout(io.StringIO()): + self.fleet.start_role("developer", "build", "pane") + self.assertIn( + "--dangerously-skip-permissions", self.fleet.herdr_command.call_args.args + ) + + def test_codex_adapter_sends_brief_after_start(self): + self.fleet.lock() + self.fleet.kinds["build"] = { + "harness": "codex", + "model": "-", + "effort": "medium", + } + self.fleet.validate() + self.fleet.herdr_command = Mock(return_value=self.start_result) + with contextlib.redirect_stdout(io.StringIO()): + self.fleet.start_role("developer", "build", "pane") + start, prompt = self.fleet.herdr_command.call_args_list + self.assertIn("codex", start.args) + self.assertNotIn("--model", start.args) + self.assertNotIn("--dangerously-bypass-approvals-and-sandbox", start.args) + self.assertEqual(prompt.args[:3], ("agent", "prompt", "developer")) + + def test_plugin_workspace_guard(self): + self.fleet.plugin = True + with ( + patch.dict(os.environ, {"HERDR_WORKSPACE_ID": "other"}), + self.assertRaisesRegex(agents.LauncherError, "did not originate"), + ): + self.fleet.guard_workspace("ws") + + def test_wrong_plugin_rejected_before_transport(self): + with ( + patch.dict(os.environ, {}, clear=True), + patch.object(agents, "call") as call, + ): + with self.assertRaisesRegex(agents.LauncherError, "not chess.team"): + agents.Fleet(options(plugin=True)) + call.assert_not_called() + + def test_stop_selected_role_does_not_close_foreign_pane(self): + fleet = self.fake_up() + fleet.options.roles = ["developer"] + fleet.validate_container = Mock() + fleet.server_running = Mock(return_value=True) + fleet.herdr_json.return_value = { + "result": { + "agents": [ + {"name": "developer", "workspace_id": "ws", "pane_id": "ours"}, + {"name": "developer", "workspace_id": "other", "pane_id": "theirs"}, + ] + } + } + fleet.down() + fleet.herdr_command.assert_called_once_with("pane", "close", "ours") + + def test_failed_start_closes_only_new_pane(self): + fleet = self.fake_up() + fleet.options.roles = ["developer"] + fleet.herdr_json.side_effect = [ + {"result": {"agents": []}}, + {"result": {"root_pane": {"pane_id": "new"}}}, + ] + fleet.start_role.side_effect = agents.LauncherError("provider failed") + with ( + contextlib.redirect_stderr(io.StringIO()), + self.assertRaises(agents.LauncherError), + ): + fleet.up() + fleet.herdr_command.assert_called_with("pane", "close", "new", check=False) + + def test_doctor_only_uses_version_and_auth_status(self): + self.fleet.validate_container = Mock() + auth = SimpleNamespace(returncode=0, stdout='{"loggedIn":true}') + with ( + patch.object(agents.shutil, "which", return_value="/bin/tool"), + patch.object(agents, "call", return_value=auth) as call, + contextlib.redirect_stdout(io.StringIO()), + ): + self.fleet.doctor() + self.assertEqual( + [item.args[0] for item in call.call_args_list], + [ + ["herdr", "--version"], + ["claude", "--version"], + ["claude", "auth", "status"], + ], + ) + + def test_invalid_configuration_fails_before_mutation(self): + cases = [ + ("defaults", "lead", "nonempty list"), + ("defaults", ["lead", "lead"], "duplicate"), + ("defaults", ["unknown"], "unknown default"), + ("roles", {"lead": []}, "roles must map"), + ("kinds", [], "kinds must map"), + ("maximum", True, "positive"), + ] + for field, value, message in cases: + fleet = agents.Fleet(options()) + setattr(fleet, field, value) + with ( + self.subTest(field=field, value=value), + self.assertRaisesRegex(agents.LauncherError, message), + ): + fleet.validate() + + def test_invalid_provider_and_effort_are_rejected(self): + for settings in ( + {"harness": "unknown", "model": "-", "effort": "medium"}, + {"harness": "claude", "model": "", "effort": "medium"}, + {"harness": "claude", "model": "sonnet", "effort": "unsupported"}, + ): + self.fleet.kinds["build"] = settings + with ( + self.subTest(settings=settings), + self.assertRaises(agents.LauncherError), + ): + self.fleet.validate() + + def test_hardware_role_brief_contains_interface_and_evidence_contracts(self): + self.fleet.lock() + self.fleet.herdr_command = Mock(return_value=self.start_result) + with contextlib.redirect_stdout(io.StringIO()): + self.fleet.start_role("hardware-engineer", "engineering", "pane") + brief = (self.fleet.state / "hardware-engineer.md").read_text() + for requirement in ( + "typed pins", + "eight TCA9554", + "hand-maintained Pi pin", + "bench measurements", + "definition/evidence/", + ): + self.assertIn(requirement, brief) + argv = self.fleet.herdr_command.call_args.args + self.assertIn("opus", argv) + self.assertIn("high", argv) + + def test_os_launch_failure_is_reported_as_launcher_error(self): + with ( + patch.object( + agents.subprocess, + "run", + side_effect=FileNotFoundError("missing executable"), + ), + self.assertRaisesRegex(agents.LauncherError, "cannot run"), + ): + agents.call(["missing"]) + + def test_default_roster_cannot_exceed_cap(self): + self.fleet.maximum = len(self.fleet.defaults) - 1 + with self.assertRaisesRegex(agents.LauncherError, "default_agents exceeds"): + self.fleet.validate() + + def test_retired_roles_are_not_accepted(self): + for role in ("student", "scenario"): + with ( + self.subTest(role=role), + self.assertRaisesRegex(agents.LauncherError, "unknown agent"), + ): + agents.Fleet(options(roles=[role])) + + def test_specialists_use_deliberate_engineering_models(self): + for role in ( + "hardware-engineer", + "mechanical-engineer", + "manufacturing-engineer", + ): + settings = self.fleet.kinds[self.fleet.roles[role]] + self.assertEqual((settings["model"], settings["effort"]), ("opus", "high")) + + def test_workspace_is_bound_to_checkout_not_generic_label(self): + self.fleet.herdr_json = Mock( + return_value={ + "result": { + "workspaces": [ + {"workspace_id": "foreign", "label": "chess-team"}, + {"workspace_id": "ours", "label": self.fleet.workspace}, + ] + } + } + ) + self.assertEqual(self.fleet.workspace_id(), "ours") + with patch.object(agents, "REPO", Path(self.temp.name) / "another-checkout"): + other = agents.Fleet(options()) + self.assertNotEqual(other.workspace, self.fleet.workspace) + + def test_matching_managed_workspace_must_have_correct_checkout(self): + self.fleet.herdr_json = Mock( + return_value={ + "result": { + "workspaces": [ + { + "workspace_id": "ws", + "label": self.fleet.workspace, + "worktree": {"checkout_path": self.temp.name}, + }, + ] + } + } + ) + with self.assertRaisesRegex(agents.LauncherError, "checkout does not match"): + self.fleet.workspace_id() + + def test_duplicate_bound_workspaces_fail_closed(self): + self.fleet.herdr_json = Mock( + return_value={ + "result": { + "workspaces": [ + {"workspace_id": name, "label": self.fleet.workspace} + for name in ("a", "b") + ] + } + } + ) + with self.assertRaisesRegex(agents.LauncherError, "duplicate"): + self.fleet.workspace_id() + + def test_named_transport_ignores_inherited_socket_but_preserves_config(self): + environment = { + "HERDR_SOCKET_PATH": "/foreign.sock", + "HERDR_SESSION": "foreign", + "HERDR_WORKSPACE_ID": "foreign", + "HERDR_PLUGIN_ID": "foreign", + "HERDR_MACHINE_ID": "foreign", + "HERDR_CONFIG_PATH": "/custom/config.toml", + } + with patch.dict(os.environ, environment), patch.object(agents, "call") as call: + self.fleet.herdr_command("workspace", "list") + self.assertEqual(call.call_args.args[0][:3], ["herdr", "--session", "chess"]) + actual = call.call_args.kwargs["env"] + self.assertEqual(actual["HERDR_CONFIG_PATH"], "/custom/config.toml") + for key in environment.keys() - {"HERDR_CONFIG_PATH"}: + self.assertNotIn(key, actual) + + def test_plugin_transport_keeps_invoking_context(self): + self.fleet.plugin = True + with ( + patch.dict(os.environ, {"HERDR_SOCKET_PATH": "/plugin.sock"}), + patch.object(agents, "call") as call, + ): + self.fleet.herdr_command("workspace", "list") + self.assertNotIn("--session", call.call_args.args[0]) + self.assertEqual( + call.call_args.kwargs["env"]["HERDR_SOCKET_PATH"], "/plugin.sock" + ) + + def test_server_environment_preserves_config_and_removes_nested_markers(self): + self.fleet.server_running = Mock(side_effect=[False, True]) + environment = { + "HERDR_CONFIG_PATH": "/custom/config.toml", + "HERDR_SOCKET_PATH": "/foreign.sock", + "CLAUDECODE": "1", + "CLAUDE_CODE_OAUTH_TOKEN": "test-only-token", + } + with ( + patch.dict(os.environ, environment), + patch.object(agents.subprocess, "Popen") as popen, + ): + self.fleet.ensure_server() + actual = popen.call_args.kwargs["env"] + self.assertEqual(actual["HERDR_CONFIG_PATH"], environment["HERDR_CONFIG_PATH"]) + self.assertEqual(actual["CLAUDE_CODE_OAUTH_TOKEN"], "test-only-token") + self.assertNotIn("CLAUDECODE", actual) + self.assertNotIn("HERDR_SOCKET_PATH", actual) + + def test_provider_startup_dialog_and_timeout_are_preserved(self): + self.fleet.lock() + for code in ("agent_not_ready", "agent_blocked", "timeout"): + for stream in ("stderr", "stdout"): + result = SimpleNamespace(returncode=1, stderr="", stdout="") + setattr(result, stream, json.dumps({"error": {"code": code}})) + self.fleet.herdr_command = Mock(return_value=result) + with ( + self.subTest(code=code, stream=stream), + self.assertRaisesRegex(agents.StartupPending, "kept pane"), + ): + self.fleet.start_role("hardware-engineer", "engineering", "new") + + def test_protocol_error_parser_handles_non_json_stderr_and_invalid_payloads(self): + result = SimpleNamespace( + stderr="warning", stdout='{"error":{"code":"timeout"}}' + ) + self.assertEqual(agents.protocol_error_code(result), "timeout") + for payload in ("", "not JSON", "[]", '{"error":null}'): + result = SimpleNamespace(stderr=payload, stdout="") + self.assertIsNone(agents.protocol_error_code(result)) + + def test_up_does_not_close_a_startup_dialog(self): + fleet = self.fake_up() + fleet.options.roles = ["hardware-engineer"] + fleet.herdr_json.side_effect = [ + {"result": {"agents": []}}, + {"result": {"root_pane": {"pane_id": "new"}}}, + ] + fleet.start_role.side_effect = agents.StartupPending("answer trust prompt") + with self.assertRaises(agents.StartupPending): + fleet.up() + self.assertFalse( + any( + call.args[:2] == ("pane", "close") + for call in fleet.herdr_command.call_args_list + ) + ) + + +class SetupAndUsageTests(unittest.TestCase): + def test_optional_reviewr_failure_does_not_block_required_setup(self): + def result(argv, **_kwargs): + return SimpleNamespace( + returncode=1 if argv[1:3] == ["plugin", "install"] else 0 + ) + + with ( + tempfile.TemporaryDirectory() as directory, + patch.object(setup.Path, "home", return_value=Path(directory)), + patch.object(setup.subprocess, "run", side_effect=result) as run, + contextlib.redirect_stdout(io.StringIO()) as output, + ): + setup.main() + self.assertIn("optional Reviewr installation failed", output.getvalue()) + commands = [call.args[0] for call in run.call_args_list] + self.assertIn(["herdr", "integration", "install", "claude"], commands) + self.assertIn(["herdr", "integration", "install", "codex"], commands) + self.assertTrue(any(command[1:3] == ["plugin", "link"] for command in commands)) + self.assertIn(["claude", "--version"], commands) + + def test_hook_normalization_is_idempotent_preserves_custom_commands(self): + with tempfile.TemporaryDirectory() as directory: + home = Path(directory) + path = home / ".claude/settings.json" + path.parent.mkdir() + command = f'bash "{home}/.claude/hooks/herdr-agent-state.sh" session' + group = {"hooks": [{"type": "command", "command": command}]} + custom = {"hooks": [{"command": "echo custom && " + command}]} + path.write_text( + json.dumps( + {"unrelated": True, "hooks": {"Stop": [group, group, custom]}} + ) + ) + setup.portable_claude_hooks(home) + first = path.read_text() + setup.portable_claude_hooks(home) + self.assertEqual(path.read_text(), first) + settings = json.loads(first) + self.assertTrue(settings["unrelated"]) + self.assertEqual(len(settings["hooks"]["Stop"]), 2) + self.assertEqual(settings["hooks"]["Stop"][1], custom) + + def test_usage_deduplicates_streams_and_skips_malformed_records(self): + now = datetime(2026, 1, 1, tzinfo=UTC) + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "session.jsonl" + + def record(tokens): + return { + "type": "assistant", + "timestamp": now.isoformat(), + "sessionId": "s", + "message": { + "id": "m", + "model": "sonnet", + "usage": {"input_tokens": tokens}, + }, + } + + path.write_text( + "\n".join([json.dumps(record(2)), json.dumps(record(8)), "{truncated"]) + ) + report = usage.collect_report(Path(directory), 24, now) + self.assertEqual(report["messages"], 1) + self.assertEqual(report["sessions"], 1) + self.assertEqual(report["input_tokens"], 8) + self.assertEqual(report["malformed_records"], 1) + + def test_usage_missing_logs_and_invalid_hours(self): + with tempfile.TemporaryDirectory() as directory: + with self.assertRaises(FileNotFoundError): + usage.collect_report(Path(directory) / "missing", 24) + for hours in (0, -1, float("nan"), float("inf"), 1e300): + with self.subTest(hours=hours), self.assertRaises(ValueError): + usage.collect_report(Path(directory), hours) + + +if __name__ == "__main__": + unittest.main() diff --git a/.agents/team/usage.py b/.agents/team/usage.py new file mode 100755 index 00000000..d87d3902 --- /dev/null +++ b/.agents/team/usage.py @@ -0,0 +1,364 @@ +#!/usr/bin/env python3 +"""Report locally recorded Claude Code token usage. + +This intentionally reports only what is present in Claude Code's local JSONL +transcripts. It cannot say how much of a user's subscription allowance +remains. +""" + +from __future__ import annotations + +import argparse +import json +import math +import re +import sys +from collections.abc import Iterator +from dataclasses import dataclass, field +from datetime import UTC, datetime, timedelta +from pathlib import Path +from typing import Any + +TOKEN_FIELDS = ( + "input_tokens", + "output_tokens", + "cache_creation_input_tokens", + "cache_read_input_tokens", +) +_NON_ALPHANUMERIC = re.compile(r"[^A-Za-z0-9]") + + +def encoded_project_path(repo: Path) -> str: + """Return Claude Code's directory encoding for a repository path.""" + + return _NON_ALPHANUMERIC.sub("-", str(repo.resolve())) + + +def default_logs_path() -> Path: + """Return the local Claude project directory for this checkout.""" + + repo = Path(__file__).resolve().parents[2] + return Path.home() / ".claude" / "projects" / encoded_project_path(repo) + + +def _parse_timestamp(value: Any) -> datetime | None: + if not isinstance(value, str) or not value.strip(): + return None + text = value.strip() + if text.endswith("Z"): + text = text[:-1] + "+00:00" + try: + parsed = datetime.fromisoformat(text) + except ValueError: + return None + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=UTC) + return parsed.astimezone(UTC) + + +def _is_synthetic_model(model: str) -> bool: + # Synthetic records are internal and excluded from usage. + normalized = model.strip().lower() + return normalized in {"synthetic", "", ""} + + +def _token_value(value: Any) -> int | None: + if isinstance(value, bool): + return None + if isinstance(value, int): + return value if value >= 0 else None + if isinstance(value, float) and math.isfinite(value) and value.is_integer(): + return int(value) if value >= 0 else None + return None + + +@dataclass +class _UsageEntry: + """Maximum token usage for one deduplicated message/request identity. + + Streaming responses can write several snapshots for the same identity. + Token totals are maxima, while session IDs are accumulated for unique + model-level session counts. + """ + + model: str + session_ids: set[str] = field(default_factory=set) + totals: dict[str, int] = field( + default_factory=lambda: {field_name: 0 for field_name in TOKEN_FIELDS} + ) + + def update(self, values: dict[str, int], session_id: str | None) -> None: + # A replayed snapshot may contain lower totals than the final snapshot. + # A field-by-field max keeps the final value without double-counting. + for field_name, value in values.items(): + self.totals[field_name] = max(self.totals[field_name], value) + if session_id: + self.session_ids.add(session_id) + + +def _jsonl_files(logs_path: Path) -> Iterator[Path]: + """Yield all transcript files, including nested subagent logs.""" + + yield from sorted(path for path in logs_path.rglob("*.jsonl") if path.is_file()) + + +def _record_timestamp(record: dict[str, Any]) -> datetime | None: + timestamp = _parse_timestamp(record.get("timestamp")) + if timestamp is not None: + return timestamp + message = record.get("message") + if isinstance(message, dict): + return _parse_timestamp(message.get("timestamp")) + return None + + +def _session_id(record: dict[str, Any]) -> str | None: + value = record.get("sessionId", record.get("session_id")) + if isinstance(value, str) and value.strip(): + return value.strip() + return None + + +def _request_id(record: dict[str, Any], message: dict[str, Any]) -> str | None: + for value in ( + record.get("requestId"), + record.get("request_id"), + message.get("requestId"), + ): + if isinstance(value, str) and value.strip(): + return value.strip() + return None + + +def _parse_usage_record( + record: Any, +) -> tuple[tuple[str, ...], str, dict[str, int], str | None, bool] | None: + """Parse one assistant record into stable reporting fields. + + None means the record is irrelevant, such as a user or synthetic record. + ValueError means it looked like a real assistant usage record but was + incomplete. The final boolean flags a malformed token field; usable fields + are still retained. + """ + + if not isinstance(record, dict) or record.get("type") != "assistant": + return None + if record.get("isSynthetic") is True: + return None + message = record.get("message") + if not isinstance(message, dict): + raise ValueError("assistant record has no message object") + usage = message.get("usage") + if not isinstance(usage, dict): + raise ValueError("assistant record has no usage object") + model = message.get("model") + if not isinstance(model, str) or not model.strip(): + raise ValueError("assistant usage record has no model") + if _is_synthetic_model(model): + return None + + message_id = message.get("id") + if isinstance(message_id, str): + message_id = message_id.strip() or None + else: + message_id = None + request_id = _request_id(record, message) + if not message_id and not request_id: + raise ValueError("assistant usage record has no message id or request id") + key = (message_id or "", request_id or "") + + values: dict[str, int] = {} + invalid_field = False + for field_name in TOKEN_FIELDS: + if field_name not in usage: + continue + value = _token_value(usage[field_name]) + if value is None: + invalid_field = True + else: + values[field_name] = value + if not values: + raise ValueError("assistant usage record has no usable token fields") + return key, model.strip(), values, _session_id(record), invalid_field + + +def collect_report( + logs_path: Path, hours: float, now: datetime | None = None +) -> dict[str, Any]: + """Collect local recorded usage for one UTC time window. + + Records are filtered before deduplication. Each in-window identity then + contributes one message with the maximum observed value of each token + field, which avoids both replay double-counting and partial totals. + """ + + if not math.isfinite(hours) or hours <= 0: + raise ValueError("hours must be a positive finite number") + if not logs_path.is_dir(): + raise FileNotFoundError( + f"Claude project log directory does not exist: {logs_path}" + ) + + until = (now or datetime.now(UTC)).astimezone(UTC) + try: + since = until - timedelta(hours=hours) + except OverflowError as error: + raise ValueError("hours exceeds the supported timestamp range") from error + entries: dict[tuple[str, ...], _UsageEntry] = {} + malformed = 0 + + for path in _jsonl_files(logs_path): + # Each line is independent, so one truncated record cannot abort the + # remaining transcript scan. + try: + lines = path.open(encoding="utf-8", errors="replace") + except OSError: + malformed += 1 + continue + with lines: + for line in lines: + try: + record = json.loads(line) + except (json.JSONDecodeError, UnicodeDecodeError): + malformed += 1 + continue + if not isinstance(record, dict) or record.get("type") != "assistant": + continue + timestamp = _record_timestamp(record) + if timestamp is None: + malformed += 1 + continue + # _record_timestamp normalized offsets to UTC for this + # inclusive window check. + if timestamp < since or timestamp > until: + continue + try: + parsed = _parse_usage_record(record) + except ValueError: + malformed += 1 + continue + if parsed is None: + continue + key, model, values, session_id, has_invalid_field = parsed + if has_invalid_field: + malformed += 1 + entry = entries.get(key) + if entry is None: + entry = _UsageEntry(model=model) + entries[key] = entry + entry.update(values, session_id) + + # Keep session sets while aggregating so many messages in one session + # count once for the model and once for the global report. + by_model: dict[str, dict[str, Any]] = {} + model_sessions: dict[str, set[str]] = {} + all_sessions: set[str] = set() + totals = {field_name: 0 for field_name in TOKEN_FIELDS} + for entry in entries.values(): + model_report = by_model.setdefault( + entry.model, + {field_name: 0 for field_name in TOKEN_FIELDS} + | {"messages": 0, "sessions": 0}, + ) + model_report["messages"] += 1 + model_sessions.setdefault(entry.model, set()).update(entry.session_ids) + all_sessions.update(entry.session_ids) + for field_name, value in entry.totals.items(): + totals[field_name] += value + model_report[field_name] += value + for model, session_ids in model_sessions.items(): + by_model[model]["sessions"] = len(session_ids) + + report: dict[str, Any] = { + "scope": "local recorded usage", + "logs": str(logs_path), + "hours": hours, + "since_utc": since.isoformat().replace("+00:00", "Z"), + "until_utc": until.isoformat().replace("+00:00", "Z"), + "messages": len(entries), + "sessions": len(all_sessions), + "malformed_records": malformed, + "by_model": dict(sorted(by_model.items())), + } + report.update(totals) + if malformed: + report["warning"] = ( + f"Skipped {malformed} malformed or incomplete assistant record(s); " + "usage is local recorded usage only." + ) + return report + + +def _positive_finite_hours(value: str) -> float: + try: + parsed = float(value) + except ValueError as exc: + raise argparse.ArgumentTypeError( + "hours must be a positive finite number" + ) from exc + if not math.isfinite(parsed) or parsed <= 0: + raise argparse.ArgumentTypeError("hours must be a positive finite number") + return parsed + + +def _print_table(report: dict[str, Any]) -> None: + print(f"Local recorded usage for the last {report['hours']:g} UTC hour(s)") + print(f"Logs: {report['logs']}") + print(f"Messages: {report['messages']} Sessions: {report['sessions']}") + print( + "Totals: " + + " ".join(f"{field_name}={report[field_name]}" for field_name in TOKEN_FIELDS) + ) + if report["by_model"]: + print("\nModel usage:") + columns = ("model", "messages", "sessions", *TOKEN_FIELDS) + print(" ".join(columns)) + for model, values in report["by_model"].items(): + print( + " ".join( + [model, str(values["messages"]), str(values["sessions"])] + + [str(values[field_name]) for field_name in TOKEN_FIELDS] + ) + ) + else: + print("No matching assistant usage records.") + if report["malformed_records"]: + print( + f"Warning: {report['malformed_records']} malformed or incomplete record(s) skipped.", + file=sys.stderr, + ) + print("This is local recorded usage; it does not report remaining plan allowance.") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Report local Claude Code JSONL usage." + ) + parser.add_argument("--hours", type=_positive_finite_hours, default=24.0) + parser.add_argument("--json", action="store_true", dest="as_json") + parser.add_argument("--logs", type=Path, default=default_logs_path()) + args = parser.parse_args(argv) + + try: + report = collect_report(args.logs.expanduser(), args.hours) + except (FileNotFoundError, ValueError) as exc: + error = {"error": str(exc), "logs": str(args.logs.expanduser())} + if args.as_json: + print(json.dumps(error, sort_keys=True)) + else: + print(f"Error: {exc}", file=sys.stderr) + print( + "Pass --logs PATH to select a Claude project log directory.", + file=sys.stderr, + ) + return 2 + + if args.as_json: + print(json.dumps(report, sort_keys=True)) + else: + _print_table(report) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.claude/settings.json b/.claude/settings.json new file mode 100644 index 00000000..c7fb41ac --- /dev/null +++ b/.claude/settings.json @@ -0,0 +1,10 @@ +{ + "attribution": { + "commit": "", + "pr": "" + }, + "env": { + "CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION": "false", + "CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS": "0" + } +} diff --git a/.devcontainer/Dockerfile b/.devcontainer/Dockerfile index 71ebc891..c76b1f3f 100644 --- a/.devcontainer/Dockerfile +++ b/.devcontainer/Dockerfile @@ -2,6 +2,8 @@ ARG VARIANT="ubuntu-24.04" FROM mcr.microsoft.com/devcontainers/base:${VARIANT} +ARG CLAUDE_VERSION="2.1.287" +ARG HERDR_VERSION="0.9.3" ARG BLENDER_VERSION="4.5.13" ARG KICAD_PPA="ppa:kicad/kicad-9.0-releases" ARG BLENDER_SHA256="da4e69b06b75b9e642d106496c50e7e240218b411d2f6e18271c1d1d819cef91" @@ -93,7 +95,7 @@ RUN archive="blender-${BLENDER_VERSION}-linux-x64.tar.xz" \ && rm "/tmp/${archive}" ENV BLENDER_BIN=/opt/blender/blender \ - PATH="/home/vscode/.cargo/bin:${PATH}" + PATH="/home/vscode/.local/bin:/home/vscode/.cargo/bin:${PATH}" USER vscode RUN curl --fail --location --proto '=https' --tlsv1.2 --show-error \ @@ -102,4 +104,22 @@ RUN curl --fail --location --proto '=https' --tlsv1.2 --show-error \ && rustup component add clippy rustfmt rust-src \ && rustup target add aarch64-unknown-linux-gnu \ && cargo install just --locked --version 1.40.0 + +# Native Claude and checksum-verified Herdr; login happens only at runtime. +ENV DISABLE_AUTOUPDATER=1 +RUN curl --fail --location --retry 3 --show-error \ + --output /tmp/claude-install.sh https://claude.ai/install.sh \ + && bash /tmp/claude-install.sh "${CLAUDE_VERSION}" \ + && rm /tmp/claude-install.sh \ + && case "$(uname -m)" in \ + x86_64) arch=x86_64; checksum=18a8dc65f1c2fa485884344356dea1cfd911c6f06cf46fa78e193f4087f4dba7 ;; \ + aarch64) arch=aarch64; checksum=4de7aa3e25678812e92960de64f7c2aaa1bca1f0f80a3c5e559837e231e1f5c0 ;; \ + *) echo "Unsupported Herdr architecture" >&2; exit 1 ;; \ + esac \ + && mkdir -p /home/vscode/.local/bin \ + && curl --fail --location --retry 3 --show-error \ + --output /home/vscode/.local/bin/herdr \ + "https://github.com/herdrdev/herdr/releases/download/v${HERDR_VERSION}/herdr-linux-${arch}" \ + && printf '%s /home/vscode/.local/bin/herdr\n' "${checksum}" | sha256sum --check \ + && chmod +x /home/vscode/.local/bin/herdr USER root diff --git a/.devcontainer/README.md b/.devcontainer/README.md index 398d908f..e808e012 100644 --- a/.devcontainer/README.md +++ b/.devcontainer/README.md @@ -19,6 +19,9 @@ The image provides: - Node.js 22, Bun 1.4, and the [Pi coding agent](https://pi.dev/), installed during container creation; +- pinned Claude Code and Herdr, plus Codex, provider status hooks and the Reviewr + panel; the optional [agent team](../.agents/team/README.md) defaults to Claude + Code for all roles; - a Bun-managed `.pi` TypeScript project with pinned Pi API types, workspace IntelliSense, and `bun run --cwd .pi check` validation for project extensions; - stable Rust with `rustfmt`, Clippy, Just, and the AArch64 GNU target/linker; @@ -35,7 +38,14 @@ CI prebuilds this image, pushes it to GHCR, then runs Python, CAD, PCB, and Rust parallel `devcontainer exec` jobs against that digest. Subsequent prebuilds reuse the image layers when the Dockerfile is unchanged. -Container creation configures the repository pre-commit hook and installs Pi. +Container creation configures the repository pre-commit hook, installs Pi/Codex, +and configures Herdr integrations without starting models. Claude Code is installed +in the image. Separate `chess-claude`, `chess-codex` and `chess-herdr` volumes keep +provider credentials and runtime configuration across rebuilds. Run +`claude auth login` inside the container (or forward `CLAUDE_CODE_OAUTH_TOKEN`). +Accept workspace trust manually. Host API credentials may select API billing; +unset `ANTHROPIC_API_KEY` before opening the container for subscription-only use. +No host credential directories are mounted and no secrets are copied into the image. Credentials for every built-in Pi API-key provider are forwarded from matching host environment variables without writing secrets to the repository. Pi's `~/.pi/agent` directory uses the persistent `chess-pi-agent` Docker volume, so @@ -54,6 +64,11 @@ After create: ```sh bun run --cwd .pi check # type-check project Pi extensions pi # start the coding agent; use /login for OAuth +claude auth login # persistent Claude login; no automatic sign-in +just agents-doctor # tools/login check, no model request +just agents lead developer # portable Rust team (eight engineering roles with just agents) +just agents lead hardware-engineer mechanical-engineer # circuit/enclosure team +just agents-test # offline launcher/configuration tests just pcb # test, then generate PCB review output just cad # test, then generate CAD output just quality # lint and format-check every package diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json index def528c0..98f7cbcd 100644 --- a/.devcontainer/devcontainer.json +++ b/.devcontainer/devcontainer.json @@ -20,11 +20,15 @@ "updateRemoteUserUID": true, "mounts": [ "source=chess-pi-agent,target=/home/vscode/.pi/agent,type=volume", - "source=chess-pi-worktrees,target=/home/vscode/.local/share/pi-worktrees,type=volume" + "source=chess-pi-worktrees,target=/home/vscode/.local/share/pi-worktrees,type=volume", + "source=chess-claude,target=/home/vscode/.claude,type=volume", + "source=chess-codex,target=/home/vscode/.codex,type=volume", + "source=chess-herdr,target=/home/vscode/.config/herdr,type=volume" ], "remoteEnv": { "AI_GATEWAY_API_KEY": "${localEnv:AI_GATEWAY_API_KEY}", "ANTHROPIC_API_KEY": "${localEnv:ANTHROPIC_API_KEY}", + "CLAUDE_CODE_OAUTH_TOKEN": "${localEnv:CLAUDE_CODE_OAUTH_TOKEN}", "ANT_LING_API_KEY": "${localEnv:ANT_LING_API_KEY}", "AWS_ACCESS_KEY_ID": "${localEnv:AWS_ACCESS_KEY_ID}", "AWS_BEARER_TOKEN_BEDROCK": "${localEnv:AWS_BEARER_TOKEN_BEDROCK}", @@ -87,6 +91,7 @@ "customizations": { "vscode": { "extensions": [ + "anthropic.claude-code", "charliermarsh.ruff", "DavidAnson.vscode-markdownlint", "fill-labs.dependi", diff --git a/.devcontainer/post-create.sh b/.devcontainer/post-create.sh index 6ab7fa10..364763f5 100755 --- a/.devcontainer/post-create.sh +++ b/.devcontainer/post-create.sh @@ -3,9 +3,11 @@ set -euo pipefail git config --local core.hooksPath .githooks -# Docker creates new named volumes as root; Pi must be able to persist state. -sudo chown "$(id -u):$(id -g)" "${HOME}/.pi/agent" "${HOME}/.local/share/pi-worktrees" -npm install --global --ignore-scripts @earendil-works/pi-coding-agent +# Docker creates new named volumes as root; each agent runtime must persist state. +sudo chown "$(id -u):$(id -g)" \ + "${HOME}/.pi/agent" "${HOME}/.local/share/pi-worktrees" \ + "${HOME}/.claude" "${HOME}/.codex" "${HOME}/.config/herdr" +npm install --global --ignore-scripts @earendil-works/pi-coding-agent @openai/codex@0.159.3 bun install --cwd .pi --frozen-lockfile bun run --cwd .pi check @@ -14,12 +16,20 @@ if ! command -v python3 >/dev/null 2>&1; then exit 1 fi +python3 .agents/team/setup.py +python3 .agents/team/agents.py list + cat <<'EOF' -The container is ready. Hardware toolchains, Bun, and Pi are installed. +The container is ready. Hardware toolchains, Bun, Pi, Claude Code, Codex and Herdr are installed. bun run --cwd .pi check type-check project Pi extensions pi start the Pi coding agent + claude auth login sign in to Claude (persisted across rebuilds) + just agents-doctor check Herdr tools/login without model requests + just agents lead developer start a small Claude Code team + just agents start the eight-role engineering team + just agents lead hardware-engineer mechanical-engineer circuit/enclosure team just list repository capabilities just cad test, then generate CAD output just pcb test, then generate PCB review output diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 89826f24..51579715 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -26,6 +26,15 @@ jobs: - id: prebuild uses: ./.github/actions/prebuild-devcontainer + agent-team: + name: Agent team (offline) + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - uses: actions/checkout@v7 + - run: python3 -m unittest discover -s .agents/team/tests -v + pi-harness: name: Pi harness runs-on: ubuntu-latest @@ -60,7 +69,7 @@ jobs: firmware: name: Firmware if: startsWith(github.ref, 'refs/tags/v') - needs: [hardware, quality, pi-harness] + needs: [hardware, quality, pi-harness, agent-team] permissions: contents: write uses: ./.github/workflows/firmware.yml diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index e8a519eb..323c76d6 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -23,6 +23,15 @@ jobs: - id: prebuild uses: ./.github/actions/prebuild-devcontainer + agent-team: + name: Agent team (offline) + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - uses: actions/checkout@v7 + - run: python3 -m unittest discover -s .agents/team/tests -v + pi-harness: name: Pi harness runs-on: ubuntu-latest @@ -125,7 +134,7 @@ jobs: required-checks: name: Required checks if: always() - needs: [hardware, hardware-report, quality, pi-harness, firmware-check] + needs: [hardware, hardware-report, quality, pi-harness, agent-team, firmware-check] runs-on: ubuntu-latest steps: - name: Verify required jobs succeeded diff --git a/.gitignore b/.gitignore index 8991ff2d..743e4838 100644 --- a/.gitignore +++ b/.gitignore @@ -15,6 +15,10 @@ __pycache__/ *.py[cod] .cache/ + +# Agent credentials/configuration overrides stay local. +.claude/settings.local.json +/.codex/ /dist/firmware/ # Generated next to their Python SPICE test cases. diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 00000000..8bd36697 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,51 @@ +# Chess contributor agents + +Read `README.md`, `CONTRIBUTING.md`, and the relevant package README before editing. +Architecture is in `docs/architecture.md`; development commands are in +`docs/development.md`. Use the devcontainer for reproducible hardware checks. + +## Boundaries + +- `apps/firmware`: Raspberry Pi process, system integration, Yocto image. +- `crates/{chess,core,logger,menu,persistence}`: portable Rust logic and services. +- `hardware/shared`: authoritative physical, electrical and firmware contracts. +- `hardware/pcb`: KiCad generation, routing, SPICE and board validation. +- `hardware/cad`: Blender geometry and review renders. + +Change authoritative contracts before regenerating artifacts. Do not hand-edit generated +CAD/PCB output or weaken tests to make hardware pass. Automated checks do not prove a +physical board is safe or manufactured correctly. Do not flash hardware, order parts, +publish fabrication files, or claim physical validation without explicit authorization. + +## Validation + +Use focused package recipes first: `just --justfile /justfile check`. +`just quality` checks formatting/lints; `just precommit` is the commit gate; +`just check` validates and generates across domains. PCB review is `just pcb`, CAD is +`just cad`. Physical release (`just pcb-release`) and Yocto (`just firmware-check`, +`just firmware`) are separate, explicit gates. Commits run `.githooks/pre-commit`. +Agent tooling lint/format/tests: `just agents-check` (included in the commit gate). +Existing Pi tooling: `bun run --cwd .pi check`. + +## Working agreement + +Preserve unrelated changes. Never stage everything, stash/reset someone else's work, +bypass hooks, force-push or merge without authorization. An explicit request for local +or unstaged review wins over any normal PR workflow: do not stage, commit, push or open +that PR until the user approves. + +Only delegate when the user authorizes it. Pi remains supported alongside Claude Code; +Herdr is an optional terminal/team runtime, not a replacement for Pi's governed subagent +protocol. Do not silently move a failed Pi workflow to another runner. + +For authorized Herdr teams, read `.agents/team/team.md` and your assigned role brief. +`just agents-list` shows the roster; `just agents-doctor` checks tools/login without a +model request. `just agents lead developer` starts a small software team; +`just agents lead hardware-engineer mechanical-engineer` starts a hardware/fit team. +`just agents` starts eight default roles: lead, developer, firmware engineer, hardware +engineer, mechanical engineer, QA, reviewer and pushback. Manufacturing, test engineering, +DevOps, PM, upgrade review and PR delivery are on demand; the cap is ten active agents. +There is no student role or per-domain team. One writer per file, one Git owner and one +expensive-check owner. Idle agents wait for concrete assignments; never create nested +teams or exceed the cap. Cross-domain pin/power/stack/image changes require an explicit +interface handoff between the relevant engineers, not a duplicate source of truth. diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 00000000..43c994c2 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1 @@ +@AGENTS.md diff --git a/README.md b/README.md index 43338505..fd5ca9e3 100644 --- a/README.md +++ b/README.md @@ -135,6 +135,13 @@ just --justfile hardware/cad/justfile generate See [`docs/development.md`](docs/development.md) for the full workflow. +Coding assistants can use the existing [Pi setup](.pi/README.md) or the optional +[Claude Code / Herdr team](.agents/team/README.md). Rebuild the devcontainer, +run `claude auth login`, then `just agents-doctor` and `just agents lead developer`. +`just agents` starts the eight-role engineering team, including hardware, mechanical +and firmware engineers. Manufacturing, test and DevOps specialists are available on +demand; provider permissions remain on by default. + ## Project status > [!IMPORTANT] diff --git a/justfile b/justfile index 29ab093b..cca8b120 100644 --- a/justfile +++ b/justfile @@ -8,7 +8,7 @@ default: @just --list # Complete repository validation and generation. -check: _automation-format +check: _automation-format agents-check #!/usr/bin/env bash set -euo pipefail for package in {{ rust_packages }}; do just --justfile "$package/justfile" check; done @@ -18,7 +18,7 @@ check: _automation-format just --justfile hardware/pcb/justfile review # Commit gate without CAD renders or PCB fabrication output. -precommit: _automation-format +precommit: _automation-format agents-check #!/usr/bin/env bash set -euo pipefail for package in {{ rust_packages }}; do just --justfile "$package/justfile" check; done @@ -28,7 +28,7 @@ precommit: _automation-format just --justfile hardware/pcb/justfile review # Formatting, linting, checking, and documentation. -quality: _automation-format +quality: _automation-format agents-check #!/usr/bin/env bash set -euo pipefail for package in {{ rust_packages }}; do just --justfile "$package/justfile" quality; done @@ -46,7 +46,7 @@ _automation-format: done # All package tests, including hardware validation. -test: +test: agents-test #!/usr/bin/env bash set -euo pipefail for package in {{ rust_packages }}; do just --justfile "$package/justfile" test; done @@ -87,6 +87,49 @@ firmware-check: firmware: just --justfile apps/firmware/justfile image +# Configure runtime Herdr hooks and the pinned review panel (no model calls). +agents-setup: + python3 .agents/team/setup.py + +# Start missing team roles; accepts role selectors and launcher flags. +[positional-arguments] +agents *args: + python3 .agents/team/agents.py up "$@" + +# Stop only this team's workspace or selected roles. +[positional-arguments] +agents-stop *args: + python3 .agents/team/agents.py down "$@" + +# Start fresh conversations and clear the session handoff. +[positional-arguments] +agents-reset *args: + python3 .agents/team/agents.py reset "$@" + +# Check container tools and provider login without a model request. +[positional-arguments] +agents-doctor *args: + python3 .agents/team/agents.py doctor "$@" + +# Show all configured models, harnesses and roles without starting them. +agents-list: + python3 .agents/team/agents.py list + +# Report local recorded Claude tokens, not remaining subscription allowance. +[positional-arguments] +agents-usage *args: + python3 .agents/team/usage.py "$@" + +# Lint, format-check and test agent tooling without provider/model requests. +agents-check: + ruff check .agents/team + ruff format --check .agents/team + just agents-test + +# Offline regressions; optional native smoke tests use HERDR_TEST_BIN. +agents-test: + python3 -m unittest discover -s .agents/team/tests -v + # Remove package-local caches and transient output. clean: #!/usr/bin/env bash From e9ed02118b30bd5ded234481e2876964ec47ed31 Mon Sep 17 00:00:00 2001 From: Nicholas Santi Date: Sun, 4 Oct 2026 12:35:42 +0000 Subject: [PATCH 2/2] test(agents): stabilize host and native setup checks --- .agents/team/setup.py | 6 +++++- .agents/team/tests/test_herdr_integration.py | 4 +++- .agents/team/tests/test_team.py | 12 ++++++++++++ 3 files changed, 20 insertions(+), 2 deletions(-) diff --git a/.agents/team/setup.py b/.agents/team/setup.py index 3b3993ee..3313b7f9 100755 --- a/.agents/team/setup.py +++ b/.agents/team/setup.py @@ -30,9 +30,13 @@ def portable_claude_hooks(home): path.write_text(json.dumps(settings, indent=2) + "\n") -def main(): +def require_container(): if not (Path("/.dockerenv").exists() or Path("/run/.containerenv").exists()): raise SystemExit("run setup inside the devcontainer") + + +def main(): + require_container() home = Path.home() for directory in (".claude", ".codex", ".config/herdr"): (home / directory).mkdir(parents=True, exist_ok=True) diff --git a/.agents/team/tests/test_herdr_integration.py b/.agents/team/tests/test_herdr_integration.py index 1f0f90c0..7705ea0c 100644 --- a/.agents/team/tests/test_herdr_integration.py +++ b/.agents/team/tests/test_herdr_integration.py @@ -45,7 +45,9 @@ def setUp(self): ) self.addCleanup(self.stop_server) for _ in range(50): - if self.command("status", "server").returncode == 0: + # `status server` can exit zero while reporting a stopped server. + # Wait for a successful socket API read, not just a CLI exit code. + if self.command("workspace", "list").returncode == 0: break if self.server.poll() is not None: self.fail("isolated Herdr server exited before readiness") diff --git a/.agents/team/tests/test_team.py b/.agents/team/tests/test_team.py index d413fd9f..33894e8b 100644 --- a/.agents/team/tests/test_team.py +++ b/.agents/team/tests/test_team.py @@ -504,6 +504,17 @@ def test_up_does_not_close_a_startup_dialog(self): class SetupAndUsageTests(unittest.TestCase): + def test_setup_rejects_host_before_filesystem_or_tool_mutation(self): + with ( + patch.object(setup.Path, "exists", return_value=False), + patch.object(setup.Path, "mkdir") as mkdir, + patch.object(setup.subprocess, "run") as run, + self.assertRaisesRegex(SystemExit, "inside the devcontainer"), + ): + setup.main() + mkdir.assert_not_called() + run.assert_not_called() + def test_optional_reviewr_failure_does_not_block_required_setup(self): def result(argv, **_kwargs): return SimpleNamespace( @@ -513,6 +524,7 @@ def result(argv, **_kwargs): with ( tempfile.TemporaryDirectory() as directory, patch.object(setup.Path, "home", return_value=Path(directory)), + patch.object(setup, "require_container"), patch.object(setup.subprocess, "run", side_effect=result) as run, contextlib.redirect_stdout(io.StringIO()) as output, ):