From bf58930ae99fcbd3938334c2532c7cca277f9998 Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Tue, 30 Jun 2026 14:27:47 -0400 Subject: [PATCH 01/14] =?UTF-8?q?feat(ctx):=20capability=20grants=20?= =?UTF-8?q?=E2=80=94=20agents,=20MCP=20servers,=20harnesses=20+=20own-mode?= =?UTF-8?q?l=20gating?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend the ctx skill-source integration beyond skills to the full capability set, behind a fail-closed permission model. - DSL (additive, config tier): `grant ctx: skills, agents, mcps, harnesses` and `ctx may use my own model "/"`. - Runtime: CtxProvisionResult carries capability groups + harnessInstall + warnings; McpCtxAdapter threads permissions/own-model and parses the ctx.loop_adapter.v1 contract; the engine merges skills+agents and surfaces mcps/harnesses on the ctx event (never auto-installs); the cli passes grants. - Schema, AGENTS.md, docs/ctx-skill-source.md, docs/ctx-integration-guide.md, examples/ctx_capabilities.loop. - Back-compat: no grant => skills+agents, exactly as before. Tests: parser 54, runtime 101 (4 new), full build green. Co-Authored-By: Claude Opus 4.8 (1M context) --- .claude/skills/loopflow/SKILL.md | 9 + AGENTS.md | 29 +- docs/ctx-integration-guide.md | 388 +++++++++++++++++++++++++++ docs/ctx-skill-source.md | 35 ++- examples/ctx_capabilities.loop | 23 ++ packages/parser/src/parser.ts | 29 ++ packages/parser/src/types.ts | 23 ++ packages/parser/test/parser.test.js | 19 ++ packages/runtime/src/cli.ts | 4 + packages/runtime/src/ctx.ts | 58 +++- packages/runtime/src/engine.ts | 29 +- packages/runtime/src/types.ts | 36 ++- packages/runtime/test/engine.test.js | 41 ++- spec/loop-spec.schema.json | 15 ++ 14 files changed, 720 insertions(+), 18 deletions(-) create mode 100644 docs/ctx-integration-guide.md create mode 100644 examples/ctx_capabilities.loop diff --git a/.claude/skills/loopflow/SKILL.md b/.claude/skills/loopflow/SKILL.md index 99ab4e5..740eea0 100644 --- a/.claude/skills/loopflow/SKILL.md +++ b/.claude/skills/loopflow/SKILL.md @@ -60,6 +60,8 @@ schedule: run unattended on a cadence runner: which agent executes the loop target: operate on another directory/repo recommend skills with ctx ctx is this file's skill source — recommends + installs skills per loop goal +grant ctx: skills, agents, mcps, harnesses capability groups ctx may recommend (fail-closed; default skills+agents; mcps/harnesses are recommend-only) +ctx may use my own model "/" declares a user-owned model — unlocks harness recommendations (dry-run only) ``` Predicates: @@ -178,6 +180,13 @@ When ctx's tools are available (`ctx__loop_provision`, `ctx__recommend_bundle`): first keeps the `.loop` self-contained and reproducible; the second lets a headless `loop run` re-resolve the bundle from ctx. 3. Offer `top up skills from ctx` if the loop should pull more skills when a cycle fails. +4. **Beyond skills** — if the goal needs more than skills, add a `grant ctx: skills, agents, + mcps, harnesses` line for the groups that apply (fail-closed; omit it for skills-only). + `mcps` and `harnesses` are **recommend-only** — ctx surfaces them with an install command + the user runs; the loop never auto-installs them. Harnesses additionally need a + `ctx may use my own model "/"` line, and always come as a `--dry-run` + command. Pass the granted groups (and own-model) to `ctx__loop_provision` as `permissions` / + `own_llm` / `model_provider` / `model`. When ctx is **not** attached, skip this silently and author `use skills:` by hand as usual — the loop runs the same either way. diff --git a/AGENTS.md b/AGENTS.md index 59c7c89..395e67a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -99,6 +99,8 @@ rigor: vibe coding | structured ai-assisted | agentic engineering (the spectru mode: conductor | orchestrator (supervision posture: in-session/sync vs async/opens-a-PR) runs as: (an auditable principal for unattended runs) recommend skills with ctx (config tier: ctx is this file's skill source — recommends + installs skills per loop goal; see "Skill source: ctx" below) +grant ctx: skills, agents, mcps, harnesses (config tier: capability groups the file lets ctx recommend; fails closed, default skills+agents; mcps/harnesses are recommend-only) +ctx may use my own model "/" (config tier: declares a user-owned/local/API model — unlocks ctx harness recommendations, always dry-run) observe: (block) trace every cycle / meter tokens and cost / stop and warn if cost exceeds "$N" sandbox: (block) no network access / allow egress to "host" only / cap cpu at … memory at … time at … hooks: (loop body block) before each cycle | after act | on commit | on stop : "" passes|finds nothing (a failing hook blocks) @@ -186,8 +188,33 @@ loop "harden the stripe webhook handler": MCP server before the first plan, and `top up skills from ctx` after a failed cycle reflects. - **No ctx attached?** The lines are inert — the loop runs exactly as it would without them. +**Beyond skills — the full capability set.** By default ctx provisions only `skills` +(and the agents Loop loads the same way). A `grant ctx:` line widens what ctx may recommend to +any of `skills, agents, mcps, harnesses`, **failing closed** — only listed groups are returned: + +```loop +recommend skills with ctx +grant ctx: skills, agents, mcps, harnesses # capability grants (fail-closed) +ctx may use my own model "ollama/llama3.1" # unlocks harness recs (dry-run only) + +loop "stand up a local agent loop": + goal: an MCP agent loop running on local ollama with filesystem access + use skills recommended by ctx + done when "pytest tests/agent_loop" passes +``` + +- **skills / agents** install into `~/.claude/skills` (as before) and merge into the cycle's + skill set. +- **mcps** are **recommend-only**: ctx surfaces fitting MCP servers + a suggested + `ctx-mcp-install `; the runtime emits them on a `ctx` event, it never auto-registers one. +- **harnesses** (autogen, langfuse, …) recommend **only** when you declare a user-owned model + (`ctx may use my own model "…"`), and ship as an explicit `ctx-harness-install --dry-run` + command — never an automatic install. This is the one capability that pulls real software, so it + stays human-gated by design. + Setup: `claude mcp add ctx -- ctx-mcp-server` (needs `pip install claude-ctx`). See -`examples/ctx_skills.loop` and `docs/ctx-skill-source.md`. +`examples/ctx_skills.loop`, `examples/ctx_capabilities.loop`, and `docs/ctx-skill-source.md`. +Full customer-facing walkthrough (setup, own-model, the capability model): `docs/ctx-integration-guide.md`. ### `remember in` — cross-run memory diff --git a/docs/ctx-integration-guide.md b/docs/ctx-integration-guide.md new file mode 100644 index 0000000..86b1269 --- /dev/null +++ b/docs/ctx-integration-guide.md @@ -0,0 +1,388 @@ +# Loop × ctx — the self-equipping coding loop + +> Your loop already knows *what* to build and *how to check it's done*. +> ctx makes it know *what to bring* — the skills, agents, MCP servers, and model +> harnesses the job needs — and loads them before the first plan. + +This is the complete guide to the Loop ⇄ ctx integration: what it is, why it +matters, how to set it up, and how to drive the full capability set — including +running on **your own local or API model**. + +--- + +## 1. The 60-second pitch + +A `.loop` file is a plain-English, self-correcting workflow: a goal, a way to +verify "done", human gates, and a retry edge. It already runs your agent in a +tight plan → act → observe → reflect cycle until the tests pass. + +The one thing a loop *couldn't* do was **equip itself**. `use skills: a, b` +assumes `a` and `b` already exist on disk. Someone had to know the right skills, +find them, and install them by hand. + +**ctx closes that gap.** Point a loop at a goal and ctx recommends the smallest +useful bundle of capabilities for it and provisions them — so the loop walks in +already holding the right tools: + +- **Skills & agents** — installed straight into `~/.claude/skills`, ready for the + loop's very first plan. +- **MCP servers** — recommended with a one-line install command (e.g. a + filesystem or database server the goal implies). +- **Model harnesses** — when you bring your own model (local Ollama, an API + model), ctx recommends a fitting agent harness (AutoGen, Langfuse, …) as a + ready-to-run, **dry-run** install command. + +It is **opt-in, fail-closed, and human-gated by design.** A loop with no ctx +attached runs exactly as before. Nothing heavier than a skill is ever installed +without you asking. + +**The outcome you're buying:** stop hand-curating tooling for every workflow. +Describe the goal; the loop arrives equipped. + +--- + +## 2. The problem it solves + +Teams writing agentic workflows hit the same wall: + +| Without ctx | With ctx | +|---|---| +| You must already know which skills a task needs. | Describe the goal; ctx recommends the bundle. | +| Skills are installed by hand, per machine, per person. | The loop installs them at run time, reproducibly. | +| MCP servers and model harnesses are wired up manually. | Recommended for the goal, with the exact install command. | +| "Bring your own model" means assembling a harness yourself. | Declare your model; ctx recommends a fitting harness. | +| Tooling drift between author's box and CI. | The `.loop` re-resolves its bundle on every headless run. | + +ctx is the **provisioning layer beneath Loop**. Loop stays the driver; ctx is +the quartermaster. + +--- + +## 3. What you get — the capability set + +ctx recommends across four capability groups. A `.loop` *grants* which ones +apply (see §6). Each group behaves differently, on purpose: + +| Group | Installed automatically? | What happens | +|-------|--------------------------|--------------| +| **skills** | ✅ into `~/.claude/skills` | Merged into the loop's skill set for plan/act. | +| **agents** | ✅ into `~/.claude` | Sub-agents the loop can invoke, loaded the same way. | +| **mcps** | ❌ **recommend-only** | Fitting MCP servers surfaced with a `ctx-mcp-install ` command. The loop never auto-registers one. | +| **harnesses** | ❌ **recommend-only, gated** | Recommended only when you declare your own model; shipped as a `ctx-harness-install --dry-run` command you run. Never auto-installed. | + +**Why the split?** Skills and agents are small, sandboxed, and the loop needs +them in hand to work. MCP servers and harnesses pull real software and touch your +machine's configuration — so ctx *recommends* them and hands you the exact +command, but the decision to install stays yours. That's the trust boundary that +makes this safe to run unattended. + +--- + +## 4. How it works + +``` + ┌────────────┐ grant + goal + own-model ┌─────────────────┐ + You → │ .loop │ ────────────────────────────► │ ctx-mcp-server │ + │ (Loop) │ ◄──────────────────────────── │ (recommender) │ + └─────┬──────┘ ctx.loop_adapter.v1 contract └────────┬────────┘ + │ │ + │ skills/agents → installed │ recommend_bundle + │ mcps/harnesses → surfaced (recommend-only) │ + harness recommender + ▼ ▼ + plan → act → observe → reflect ↺ ~/.claude/skills + the graph +``` + +1. A loop that opts into ctx calls `ctx__loop_provision` **once before the first + plan**, passing its goal, the capability grants, and (optionally) your model. +2. ctx returns a single read-only JSON contract (`ctx.loop_adapter.v1`): the + skills/agents it installed, and the MCP servers / harnesses it recommends. +3. The loop merges skills + agents into its working set, and surfaces the + recommend-only items on its event stream for you (or your host) to act on. +4. On a failed cycle, `top up skills from ctx` asks for *more* — the loop learns + what it was missing from the failure and re-equips before the next plan. + +If ctx isn't attached, every ctx line is inert and the loop runs unchanged. A +ctx call that fails emits one "skipped" event and the loop continues. **A loop +never fails because ctx is missing.** + +--- + +## 5. Setup + +### Prerequisites +- [Loop](https://github.com/tickets-forge-dev/loop-lang) (`.loop` runtime / the + `/loopflow` skill in Claude Code). +- Python 3.11+ for ctx. + +### Install & attach ctx + +```bash +# 1. Install ctx and seed its recommendation graph +pip install claude-ctx +ctx-init --graph --model-mode skip # extracts the recommendation graph into ~/.claude/skill-wiki + +# 2. Attach ctx's tools to Claude Code over MCP +claude mcp add ctx -- ctx-mcp-server +claude mcp list # → ctx: ✔ Connected +``` + +That exposes the tools the Loop bridge uses: + +- `ctx__recommend_bundle` — read-only preview of what ctx would recommend. +- `ctx__loop_provision` — recommend + install skills/agents, recommend mcps/harnesses, return the contract. +- `ctx__loop_topup` — the same, for *additional* capabilities after a failed cycle. + +> **No graph yet?** ctx will return an empty (but valid) contract — the loop runs +> on whatever it already names. Re-run `ctx-init --graph` to seed or refresh. + +--- + +## 6. The grammar + +Five lines, all additive, all inert without ctx attached. + +```loop +recommend skills with ctx # config: ctx is this file's capability source +grant ctx: skills, agents, mcps, harnesses # config: which groups ctx may recommend (fail-closed) +ctx may use my own model "ollama/llama3.1" # config: declare your model → unlocks harnesses + +loop "stand up a local agent loop": + goal: an MCP agent loop on local ollama with filesystem access, with passing tests + use skills recommended by ctx for "local ollama agent loop with filesystem MCP" # loop body + top up skills from ctx when a step needs more # loop body + done when "pytest agent/tests/test_loop.py" passes +``` + +| Line | Tier | Effect | +|------|------|--------| +| `recommend skills with ctx` | config | Declares ctx as the file's capability source. | +| `grant ctx: ` | config | Capability groups ctx may recommend. **Fails closed** — omit it and ctx defaults to `skills + agents`; list only what you want. | +| `ctx may use my own model "/"` | config | Declares a user-owned/local/API model. Required to unlock **harness** recommendations. | +| `use skills recommended by ctx [for ""]` | loop body | Provision the bundle for the goal (or an explicit intent) before the first plan. | +| `top up skills from ctx when a step needs more` | loop body | After a failed cycle reflects, pull additional capabilities before re-planning. | + +### Fail-closed permissions — what it means + +`grant ctx:` is an allow-list, not a wish-list. ctx returns **only** the groups +you name: + +- No `grant ctx:` line → `skills + agents` (the original, safe default). +- `grant ctx: skills` → skills only; agents/mcps/harnesses are never returned. +- `grant ctx: skills, mcps` → skills installed, MCP servers recommended; no agents, no harnesses. + +A typo in a group name grants nothing for that token — it can never accidentally +widen access. + +--- + +## 7. Using your own model (the harness story) + +This is the feature that turns Loop × ctx from "skill installer" into "bring your +own model agent platform". + +If you run on a **local model** (Ollama, llama.cpp) or **your own API model**, +you usually need a *harness* — an agent framework like AutoGen or an +observability layer like Langfuse — wired to that model. ctx recommends one for +your goal and model, and hands you the command to install it. + +### Step 1 — declare your model + +```loop +ctx may use my own model "ollama/llama3.1" +``` + +The string is `"/"`. The provider (before the first `/`) and the +full model id are both passed to ctx so it can score harnesses for your exact +setup. + +### Step 2 — grant the harness group + +```loop +grant ctx: skills, harnesses +``` + +Harnesses are **double-gated**: they're returned only when *both* `harnesses` is +granted *and* a model is declared. Grant `harnesses` without a model and ctx +fails closed with a clear warning instead of recommending something it can't fit: + +```json +"warnings": ["harnesses granted but no user-owned model declared + (set own_llm / model_provider / model) — skipping harness recs."] +``` + +### Step 3 — run, review, install + +ctx returns the recommended harnesses with fit scores and a **dry-run** install +command: + +```json +"capabilities": { + "harnesses": [ + { "name": "autogen", "type": "harness", "fit_score": 1.0, + "install_command": "ctx-harness-install autogen --dry-run" }, + { "name": "langfuse", "type": "harness", "fit_score": 0.9, + "install_command": "ctx-harness-install langfuse --dry-run" } + ] +}, +"harness_install": "ctx-harness-install autogen --dry-run" +``` + +The loop **never installs a harness for you.** It surfaces the command; you run +it. `--dry-run` shows exactly what would be installed before anything touches +your machine. Drop `--dry-run` when you're ready. + +> **Why gated and dry-run?** A harness is the one capability that pulls a full +> framework and runs code against your model. Keeping it an explicit, previewable +> step is what lets you grant `harnesses` in a workflow that otherwise runs +> unattended. + +--- + +## 8. A full worked example + +`examples/ctx_capabilities.loop`: + +```loop +recommend skills with ctx +grant ctx: skills, agents, mcps, harnesses +ctx may use my own model "ollama/llama3.1" + +loop "stand up a local agent loop": + goal: an MCP agent loop running on local ollama with filesystem access, with passing tests + look at: agent/loop.py, agent/tests/test_loop.py + use skills recommended by ctx for "local ollama agent loop with filesystem MCP" + top up skills from ctx when a step needs more + each cycle: plan, then act, then observe + done when "pytest agent/tests/test_loop.py" passes + when it fails: reflect on the failing assertion, then plan again + after 6 tries: stop and warn "local agent loop still red — needs a human" +``` + +Print its shape: + +```bash +loop show examples/ctx_capabilities.loop +``` +``` +loop "stand up a local agent loop" + ↻ plan → act → observe (each cycle) + ↺ on fail: reflect → plan (the back-edge) + ✓ done when: "pytest agent/tests/test_loop.py" passes + ⛔ guard: after 6 tries → stop & warn "local agent loop still red — needs a human" +``` + +Run it: + +```bash +loop run examples/ctx_capabilities.loop --events +``` + +What happens on the first cycle: +1. ctx provisions skills + agents for the goal → installed, merged into the plan. +2. The filesystem MCP server is **recommended** (with its install command) on the + `ctx` event — you decide whether to register it. +3. Because a model is declared, a fitting **harness** is recommended as a dry-run + command. +4. plan → act → observe runs. If the tests fail, `top up skills from ctx` pulls + more before the next plan. + +--- + +## 9. The contract (for integrators) + +Every provision/top-up call returns one stable, versioned JSON object. Build +against it directly if you're embedding Loop or driving ctx from another host: + +```jsonc +{ + "version": "ctx.loop_adapter.v1", + "permissions": { "skills": true, "agents": true, "mcps": true, "harnesses": true }, + "use_skills": ["..."], // skill + agent names now resolvable on disk + "installed": ["..."], // freshly installed this call + "skipped": ["..."], // already present + "unavailable": [{ "name": "...", "status": "not-in-wiki" }], + "recommended": [{ "name": "...", "type": "skill", "score": 146.9 }], + "capabilities": { + "skills": [{ "name": "...", "type": "skill", "status": "installed" }], + "agents": [{ "name": "...", "type": "agent", "status": "installed" }], + "mcps": [{ "name": "...", "type": "mcp-server", "status": "available", + "install_command": "ctx-mcp-install ..." }], + "harnesses": [{ "name": "...", "type": "harness", "fit_score": 1.0, + "install_command": "ctx-harness-install ... --dry-run" }] + }, + "harness_install": "ctx-harness-install ... --dry-run", // or null + "warnings": [] +} +``` + +The contract is **additive and back-compatible**: the original +`use_skills`/`installed`/`skipped` keys are unchanged, so existing skills-only +integrations keep working untouched. + +MCP tool parameters (`ctx__loop_provision` / `ctx__loop_topup`): + +| Param | Type | Meaning | +|-------|------|---------| +| `goal` | string | What the capabilities are for. | +| `intent` | string | Optional query override. | +| `permissions` | string[] | Granted groups. Omit → `skills + agents`. | +| `own_llm` / `model_provider` / `model` | bool / string / string | Your model — unlocks harnesses. | +| `top_k` | int | Recommendations per group (≤ 5). | +| `dry_run` | bool | Recommend without installing skills/agents. | + +--- + +## 10. Safety & trust + +Designed to be safe to grant in unattended workflows: + +- **Opt-in.** No ctx attached → every ctx line is a no-op. Existing loops are unaffected. +- **Fail-closed.** Capabilities are an allow-list. Nothing outside the grant is ever returned. +- **Recommend-only for heavy capabilities.** MCP servers and harnesses are never + auto-installed — ctx hands you the command; you run it. +- **Dry-run by default for harnesses.** See exactly what would be installed first. +- **Human-gated, double-gated for harnesses.** They require both the grant *and* a declared model. +- **Degrades quietly.** A failed ctx call emits one event and the loop continues + with whatever it already names. +- **Reproducible.** Author-time names are baked into a literal `use skills:` line, + while the directive re-resolves on headless runs — so CI matches the author's box. + +--- + +## 11. FAQ + +**Do I have to use ctx?** No. It's entirely optional and opt-in. Loops without +ctx lines behave identically. + +**Will it install things I didn't approve?** Only skills and agents are installed +automatically, and only from groups you granted. MCP servers and harnesses are +never auto-installed. + +**Can I preview before anything changes?** Yes — `ctx__recommend_bundle` is a +read-only preview, and harness/MCP recommendations are always commands you choose +to run. Use `dry_run: true` to recommend skills/agents without installing them. + +**Does it work headless / in CI?** Yes. `loop run ` re-resolves the bundle +through the ctx MCP server, so an unattended run equips itself the same way an +author's session did. + +**What if my goal needs a tool ctx doesn't know?** ctx recommends from its graph; +unknown items simply don't appear. The loop still runs with whatever it names. +Re-seed or extend the graph to teach ctx new capabilities. + +**Local model or API model?** Both. Declare it with +`ctx may use my own model "/"`. That's what unlocks harness +recommendations tuned to your setup. + +--- + +## 12. Reference + +- Worked examples: `examples/ctx_skills.loop` (skills only), + `examples/ctx_capabilities.loop` (full capability set). +- Grammar in context: `AGENTS.md` → *Skill source: ctx*. +- Mechanics & contract: `docs/ctx-skill-source.md`. +- ctx itself: (`pip install claude-ctx`). + +**One line to remember:** *ctx provisions; Loop drives.* You describe the goal — +the loop arrives equipped. diff --git a/docs/ctx-skill-source.md b/docs/ctx-skill-source.md index 8ab0132..d40eae0 100644 --- a/docs/ctx-skill-source.md +++ b/docs/ctx-skill-source.md @@ -19,9 +19,12 @@ ctx-init --graph --model-mode skip # seed the recommendation graph claude mcp add ctx -- ctx-mcp-server # attach ctx's MCP tools ``` -That exposes four tools the Loop bridge uses: `ctx__recommend_bundle` (preview), -`ctx__loop_provision` (recommend + install + return names), and -`ctx__loop_topup` (add more on a failing cycle). +That exposes the tools the Loop bridge uses: `ctx__recommend_bundle` (preview), +`ctx__loop_provision` (recommend + install + return names), and `ctx__loop_topup` +(add more on a failing cycle). `loop_provision`/`loop_topup` accept an optional +`permissions` array (`skills, agents, mcps, harnesses`) plus `own_llm` / +`model_provider` / `model`, and return the versioned `ctx.loop_adapter.v1` +contract (see *Capability groups* below). ## Grammar @@ -43,6 +46,32 @@ loop "harden the stripe webhook handler": | `recommend skills with ctx` | config | Declares ctx as the file's skill source. | | `use skills recommended by ctx [for ""]` | loop body | Author-time: bake resolved names into `use skills:`. Run-time: re-resolve before the first plan. | | `top up skills from ctx when a step needs more` | loop body | Run-time: after a cycle fails and reflects, pull additional skills before re-planning. | +| `grant ctx: skills, agents, mcps, harnesses` | config | Capability groups the file lets ctx recommend. Fails closed; default (no line) = skills+agents. | +| `ctx may use my own model "/"` | config | Declares a user-owned/local/API model — unlocks harness recommendations (dry-run only). | + +## Capability groups (beyond skills) + +ctx recommends across four entity types; a `.loop` grants which ones apply. The +model **fails closed** — with no `grant ctx:` line the grant defaults to +`skills + agents` (the original behaviour), and only listed groups are ever +returned. + +| Group | Installed? | Behaviour | +|-------|-----------|-----------| +| `skills` | yes → `~/.claude/skills` | Merged into the cycle's skill set, as before. | +| `agents` | yes → `~/.claude` | Loaded the same way Loop loads named (sub)agents. | +| `mcps` | **no — recommend-only** | Fitting MCP servers surfaced with a suggested `ctx-mcp-install `; emitted on the `ctx` event. The runtime never auto-registers one. | +| `harnesses` | **no — recommend-only, gated** | Recommended only when the loop declares a user-owned model (`ctx may use my own model …`); shipped as an explicit `ctx-harness-install --dry-run` command. Never an automatic install. | + +The provision/top-up calls return the `ctx.loop_adapter.v1` contract: +`{ version, permissions, use_skills, installed, skipped, unavailable, +recommended, capabilities{skills,agents,mcps,harnesses}, harness_install, +warnings }`. The runtime merges `use_skills` (skills + agents) into the loop and +surfaces `capabilities.mcps` / `capabilities.harnesses` / `harness_install` on +the `ctx` event for the host or a human to act on — it never installs an MCP +server or a harness on its own. + +See `examples/ctx_capabilities.loop` for the full-capability example. ## How it works diff --git a/examples/ctx_capabilities.loop b/examples/ctx_capabilities.loop new file mode 100644 index 0000000..c36d5bd --- /dev/null +++ b/examples/ctx_capabilities.loop @@ -0,0 +1,23 @@ +# ctx as the full capability source — not just skills. +# +# `grant ctx:` lets ctx recommend across four capability groups; it FAILS CLOSED, so only the +# groups you list are returned. skills + agents install into ~/.claude/skills (and merge into the +# loop's skill set). mcps are recommend-only (the runtime suggests them, never auto-registers). +# harnesses recommend ONLY when you declare a user-owned/local/API model, and always ship as a +# `--dry-run` install command a human runs — never an automatic install. +# +# Needs the ctx MCP server attached (claude mcp add ctx -- ctx-mcp-server). With no ctx attached, +# every ctx line is inert and the loop runs exactly as it would without them. +recommend skills with ctx +grant ctx: skills, agents, mcps, harnesses +ctx may use my own model "ollama/llama3.1" + +loop "stand up a local agent loop": + goal: an MCP agent loop running on local ollama with filesystem access, with passing tests + look at: agent/loop.py, agent/tests/test_loop.py + use skills recommended by ctx for "local ollama agent loop with filesystem MCP" + top up skills from ctx when a step needs more + each cycle: plan, then act, then observe + done when "pytest agent/tests/test_loop.py" passes + when it fails: reflect on the failing assertion, then plan again + after 6 tries: stop and warn "local agent loop still red — needs a human" diff --git a/packages/parser/src/parser.ts b/packages/parser/src/parser.ts index 9deb982..4657852 100644 --- a/packages/parser/src/parser.ts +++ b/packages/parser/src/parser.ts @@ -1,5 +1,6 @@ import { Action, + CapabilityGroup, Config, CycleStep, Definition, @@ -25,6 +26,14 @@ import { const RIGOR_LEVELS: Rigor[] = ["vibe coding", "structured ai-assisted", "agentic engineering"]; +// Capability-group names a `grant ctx:` line may list (singular/plural + entity-type spellings). +const CTX_CAPABILITY_GROUPS: Record = { + skills: "skills", skill: "skills", + agents: "agents", agent: "agents", + mcps: "mcps", mcp: "mcps", "mcp-server": "mcps", "mcp-servers": "mcps", + harnesses: "harnesses", harness: "harnesses", +}; + const HOOK_POINTS: Record = { "before each cycle": "before-cycle", "after plan": "after-plan", @@ -699,6 +708,26 @@ function parseConfigLine(config: Config, ln: Line): boolean { config.skillSource = { provider: "ctx" }; return true; } + // Config tier: capability groups the file grants ctx (`grant ctx: skills, agents, mcps, harnesses`). + // Fails closed — only listed groups are granted; default (no line) is skills+agents in the adapter. + if ((m = t.match(/^grant ctx:\s*(.+)$/i))) { + const groups: CapabilityGroup[] = []; + for (const raw of m[1].split(/,|\band\b/).map((s) => s.trim().toLowerCase()).filter(Boolean)) { + const g = CTX_CAPABILITY_GROUPS[raw]; + if (!g) { + throw new ParseError(`unknown ctx capability group "${raw}" (expected: skills, agents, mcps, harnesses)`, ln.lineNo); + } + if (!groups.includes(g)) groups.push(g); + } + config.ctxGrants = groups; + return true; + } + // Config tier: declare a user-owned/local/API model so ctx may recommend harnesses (gated, dry-run). + if ((m = t.match(/^ctx may use my own model\s+"([^"]+)"$/i))) { + const spec = m[1].trim(); + config.ownModel = { provider: spec.split("/")[0].trim(), model: spec }; + return true; + } return false; } diff --git a/packages/parser/src/types.ts b/packages/parser/src/types.ts index 03cd79b..7e31ac2 100644 --- a/packages/parser/src/types.ts +++ b/packages/parser/src/types.ts @@ -80,6 +80,22 @@ export interface SkillDiscovery { intent?: string; } +/** + * A capability group a `.loop` can grant ctx (`grant ctx: skills, agents, mcps, harnesses`). + * Fails closed: with no grant, ctx defaults to skills+agents. mcps/harnesses are recommend-only; + * harnesses additionally require an `ownModel`. + */ +export type CapabilityGroup = "skills" | "agents" | "mcps" | "harnesses"; + +/** + * A user-owned/local/API model (`ctx may use my own model "/"`). Declaring one + * unlocks ctx's harness recommendations (which otherwise stay gated, since harnesses pull software). + */ +export interface OwnModel { + provider: string; + model: string; +} + export interface Config { use?: string; useOverrides?: OverrideEntry[]; @@ -107,6 +123,13 @@ export interface Config { runsAs?: string; /** External skill recommender for the whole file (`recommend skills with ctx`). */ skillSource?: SkillSource; + /** + * Capability groups this file grants ctx (`grant ctx: skills, agents, mcps, harnesses`). + * Threaded to ctx as permissions; fail-closed, default skills+agents. Absent = legacy behaviour. + */ + ctxGrants?: CapabilityGroup[]; + /** A user-owned model (`ctx may use my own model "…"`) that unlocks ctx harness recommendations. */ + ownModel?: OwnModel; } export interface OverrideEntry { diff --git a/packages/parser/test/parser.test.js b/packages/parser/test/parser.test.js index 1a97321..5f4519f 100644 --- a/packages/parser/test/parser.test.js +++ b/packages/parser/test/parser.test.js @@ -422,6 +422,25 @@ test("ctx: discovery coexists with a hand-named 'use skills:' line", () => { assert.deepEqual(loop.skillDiscovery, { provider: "ctx" }); }); +test("ctx: config-tier 'grant ctx:' parses the capability groups", () => { + const file = parse('grant ctx: skills, agents, mcps, harnesses\nloop "x":\n goal: g\n use skills recommended by ctx\n done when "t" passes'); + assert.deepEqual(file.config.ctxGrants, ["skills", "agents", "mcps", "harnesses"]); +}); + +test("ctx: 'grant ctx:' dedups + normalizes singular/'and', rejects unknown groups", () => { + const file = parse('grant ctx: skill, mcp and harness and harness\nloop "x":\n goal: g\n done when "t" passes'); + assert.deepEqual(file.config.ctxGrants, ["skills", "mcps", "harnesses"]); + assert.throws( + () => parse('grant ctx: bogus\nloop "x":\n goal: g\n done when "t" passes'), + /unknown ctx capability group/, + ); +}); + +test("ctx: 'ctx may use my own model' parses provider + model", () => { + const file = parse('ctx may use my own model "ollama/llama3.1"\nloop "x":\n goal: g\n done when "t" passes'); + assert.deepEqual(file.config.ownModel, { provider: "ollama", model: "ollama/llama3.1" }); +}); + test("parallel stages: 'stages in parallel:' assigns a shared group id", () => { const pipe = parse( 'pipeline "p":\n stage "a":\n goal: g\n check: t\n stages in parallel:\n stage "b":\n goal: g\n check: t\n stage "c":\n goal: g\n check: t' diff --git a/packages/runtime/src/cli.ts b/packages/runtime/src/cli.ts index 8e316ba..a58553d 100644 --- a/packages/runtime/src/cli.ts +++ b/packages/runtime/src/cli.ts @@ -324,6 +324,8 @@ async function main() { human: ipc, git, ctx, + ctxGrants: file.config?.ctxGrants, + ownModel: file.config?.ownModel, baseDir: target, loadFile, readText, @@ -347,6 +349,8 @@ async function main() { human: new CliHumanIO(), git, ctx, + ctxGrants: file.config?.ctxGrants, + ownModel: file.config?.ownModel, baseDir: target, loadFile, readText, diff --git a/packages/runtime/src/ctx.ts b/packages/runtime/src/ctx.ts index a0bd268..4308d2e 100644 --- a/packages/runtime/src/ctx.ts +++ b/packages/runtime/src/ctx.ts @@ -22,13 +22,48 @@ function parseResult(raw: unknown): CtxProvisionResult { return EMPTY; } const names = (obj.use_skills ?? obj.useSkills) as unknown; + // capabilities. is a list of {name,...} entries (ctx.loop_adapter.v1); pull bare names. + const caps = obj.capabilities as Record | undefined; + const groupNames = (g: string): string[] | undefined => { + const list = caps?.[g]; + if (!Array.isArray(list)) return undefined; + return list + .map((e) => (e && typeof e === "object" ? (e as { name?: unknown }).name : e)) + .filter((s): s is string => typeof s === "string"); + }; + const capabilities = caps + ? { + skills: groupNames("skills"), + agents: groupNames("agents"), + mcps: groupNames("mcps"), + harnesses: groupNames("harnesses"), + } + : undefined; return { useSkills: Array.isArray(names) ? names.filter((s): s is string => typeof s === "string") : [], installed: Array.isArray(obj.installed) ? (obj.installed as string[]) : undefined, skipped: Array.isArray(obj.skipped) ? (obj.skipped as string[]) : undefined, + capabilities, + harnessInstall: typeof obj.harness_install === "string" ? (obj.harness_install as string) : null, + warnings: Array.isArray(obj.warnings) ? (obj.warnings as string[]) : undefined, }; } +/** Build the optional ctx tool args (permissions + own-model) shared by provision and topup. */ +function ctxArgs( + permissions?: string[], + ownModel?: { provider: string; model: string } +): Record { + const a: Record = {}; + if (permissions && permissions.length) a.permissions = permissions; + if (ownModel) { + a.own_llm = true; + a.model_provider = ownModel.provider; + a.model = ownModel.model; + } + return a; +} + /** * Talks to a ctx MCP server (`ctx-mcp-server`) over stdio to provision skills for a loop. * The child process is spawned lazily on the first call, so attaching this adapter to a loop @@ -66,12 +101,27 @@ export class McpCtxAdapter implements CtxAdapter { return parseResult(res); } - provision(input: { goal: string; intent?: string; baseDir: string }): Promise { - return this.call("ctx__loop_provision", { goal: input.goal, intent: input.intent }); + provision(input: { + goal: string; intent?: string; baseDir: string; + permissions?: string[]; ownModel?: { provider: string; model: string }; + }): Promise { + return this.call("ctx__loop_provision", { + goal: input.goal, + intent: input.intent, + ...ctxArgs(input.permissions, input.ownModel), + }); } - topup(input: { goal: string; reflection: string; loaded: string[]; baseDir: string }): Promise { - return this.call("ctx__loop_topup", { goal: input.goal, reflection: input.reflection, loaded: input.loaded }); + topup(input: { + goal: string; reflection: string; loaded: string[]; baseDir: string; + permissions?: string[]; ownModel?: { provider: string; model: string }; + }): Promise { + return this.call("ctx__loop_topup", { + goal: input.goal, + reflection: input.reflection, + loaded: input.loaded, + ...ctxArgs(input.permissions, input.ownModel), + }); } async close(): Promise { diff --git a/packages/runtime/src/engine.ts b/packages/runtime/src/engine.ts index 545c5c0..128adcb 100644 --- a/packages/runtime/src/engine.ts +++ b/packages/runtime/src/engine.ts @@ -141,8 +141,18 @@ async function executeLoop(loop: Loop, opts: RunOptions): Promise { goal: loop.goal, intent: loop.skillDiscovery.intent, baseDir: opts.baseDir, + permissions: opts.ctxGrants, + ownModel: opts.ownModel, + }); + // Skills/agents merge into the working set; mcps/harnesses are recommend-only and only + // surfaced on the event (a host registers MCPs / a human runs the dry-run harness install). + emit(opts, { + type: "ctx", action: "provision", skills: mergeSkills(res.useSkills), ok: true, + mcps: res.capabilities?.mcps, + harnesses: res.capabilities?.harnesses, + harnessInstall: res.harnessInstall ?? undefined, + warnings: res.warnings, }); - emit(opts, { type: "ctx", action: "provision", skills: mergeSkills(res.useSkills), ok: true }); } catch (err) { emit(opts, { type: "ctx", action: "provision", skills: [], ok: false, detail: String((err as Error)?.message ?? err) }); } @@ -333,9 +343,22 @@ async function executeLoop(loop: Loop, opts: RunOptions): Promise { // skills for the next plan. Excludes already-loaded skills; degrades quietly on any error. if (loop.skillTopUp && opts.ctx && reflection && reflection !== reflectionBefore) { try { - const res = await opts.ctx.topup({ goal: loop.goal, reflection, loaded: skills, baseDir: opts.baseDir }); + const res = await opts.ctx.topup({ + goal: loop.goal, reflection, loaded: skills, baseDir: opts.baseDir, + permissions: opts.ctxGrants, + ownModel: opts.ownModel, + }); const added = mergeSkills(res.useSkills); - if (added.length) emit(opts, { type: "ctx", action: "topup", skills: added, ok: true }); + const mcps = res.capabilities?.mcps; + const harnesses = res.capabilities?.harnesses; + if (added.length || mcps?.length || harnesses?.length) { + emit(opts, { + type: "ctx", action: "topup", skills: added, ok: true, + mcps, harnesses, + harnessInstall: res.harnessInstall ?? undefined, + warnings: res.warnings, + }); + } } catch (err) { emit(opts, { type: "ctx", action: "topup", skills: [], ok: false, detail: String((err as Error)?.message ?? err) }); } diff --git a/packages/runtime/src/types.ts b/packages/runtime/src/types.ts index 0c84019..32ee8ff 100644 --- a/packages/runtime/src/types.ts +++ b/packages/runtime/src/types.ts @@ -30,7 +30,13 @@ export type LoopEvent = | { type: "foreach-item-end"; var: string; index: number; satisfied: boolean } | { type: "foreach-end"; var: string; satisfied: boolean } | { type: "git"; action: "branch"|"worktree"|"commit"|"push"|"pr"; detail: string } - | { type: "ctx"; action: "provision" | "topup"; skills: string[]; ok?: boolean; detail?: string } + | { type: "ctx"; action: "provision" | "topup"; skills: string[]; ok?: boolean; detail?: string; + /** Recommend-only capabilities ctx surfaced (never auto-installed): MCP servers / harnesses. */ + mcps?: string[]; harnesses?: string[]; + /** Dry-run command to install the top recommended harness — an explicit user action. */ + harnessInstall?: string; + /** Non-fatal notes from ctx (e.g. harnesses granted without an own model). */ + warnings?: string[] } | { type: "model"; node: "plan" | "act" | "reflect" | "also"; tier: "fast" | "strong"; model?: string }; export type CycleNode = "plan" | "act" | "observe"; @@ -140,6 +146,16 @@ export interface CtxProvisionResult { installed?: string[]; /** Slugs ctx skipped because already present (informational). */ skipped?: string[]; + /** + * Recommend-only capability names by group (the ctx.loop_adapter.v1 contract). `mcps` and + * `harnesses` are NEVER merged into the loop's skill set — they are surfaced for the host/user + * to act on. Present only for groups the loop granted. + */ + capabilities?: { skills?: string[]; agents?: string[]; mcps?: string[]; harnesses?: string[] }; + /** Dry-run command to install the top recommended harness — an explicit, human-gated action. */ + harnessInstall?: string | null; + /** Non-fatal notes (e.g. harnesses granted without a declared own model). */ + warnings?: string[]; } /** @@ -149,8 +165,18 @@ export interface CtxProvisionResult { * loop's hand-named skills — ctx is always optional. */ export interface CtxAdapter { - provision(input: { goal: string; intent?: string; baseDir: string }): Promise; - topup(input: { goal: string; reflection: string; loaded: string[]; baseDir: string }): Promise; + provision(input: { + goal: string; intent?: string; baseDir: string; + /** Capability groups granted (`grant ctx: …`); omitted = ctx's default (skills+agents). */ + permissions?: string[]; + /** A user-owned model (`ctx may use my own model …`) — unlocks harness recommendations. */ + ownModel?: { provider: string; model: string }; + }): Promise; + topup(input: { + goal: string; reflection: string; loaded: string[]; baseDir: string; + permissions?: string[]; + ownModel?: { provider: string; model: string }; + }): Promise; close?(): Promise; } @@ -186,6 +212,10 @@ export interface RunOptions { cliModel?: string; /** External skill recommender (ctx). Present when the file opts in; absent = degrade to named skills. */ ctx?: CtxAdapter; + /** Capability groups the file grants ctx (`grant ctx: …`), threaded to provision/topup as permissions. */ + ctxGrants?: string[]; + /** A user-owned model (`ctx may use my own model …`) that unlocks ctx harness recommendations. */ + ownModel?: { provider: string; model: string }; } export interface LoopOutcome { diff --git a/packages/runtime/test/engine.test.js b/packages/runtime/test/engine.test.js index bef06a9..b3e9f9a 100644 --- a/packages/runtime/test/engine.test.js +++ b/packages/runtime/test/engine.test.js @@ -742,16 +742,22 @@ test("parallel stages run concurrently and all must satisfy", async () => { // ---- ctx skill provisioning ---- -/** Records provision/topup calls; returns scripted skill names. */ +/** Records provision/topup calls; returns scripted skill names + recommend-only capabilities. */ class MockCtxAdapter { - constructor({ provisionSkills = [], topupSkills = [] } = {}) { + constructor({ provisionSkills = [], topupSkills = [], capabilities, harnessInstall, warnings } = {}) { this.provisionCalls = []; this.topupCalls = []; this._p = provisionSkills; this._t = topupSkills; + this._caps = capabilities; + this._harness = harnessInstall; + this._warnings = warnings; } - async provision(input) { this.provisionCalls.push(input); return { useSkills: this._p }; } - async topup(input) { this.topupCalls.push(input); return { useSkills: this._t }; } + _result(useSkills) { + return { useSkills, capabilities: this._caps, harnessInstall: this._harness, warnings: this._warnings }; + } + async provision(input) { this.provisionCalls.push(input); return this._result(this._p); } + async topup(input) { this.topupCalls.push(input); return this._result(this._t); } } test("ctx: provision merges recommended skills into the first plan", async () => { @@ -826,3 +832,30 @@ test("ctx: no top up when a loop does not ask for it", async () => { }); assert.equal(ctx.topupCalls.length, 0, "no top up without `top up skills from ctx`"); }); + +test("ctx: grants + own model thread to provision; mcps/harnesses surface but never merge into skills", async () => { + const def = parse('loop "x":\n goal: build a local agent loop\n use skills recommended by ctx\n done when "t" passes').definitions[0]; + const runner = new MockRunner(); + const ctx = new MockCtxAdapter({ + provisionSkills: ["fastapi-patterns"], + capabilities: { mcps: ["local-ollama-files"], harnesses: ["autogen"] }, + harnessInstall: "ctx-harness-install autogen --dry-run", + warnings: [], + }); + const { events, onEvent } = collect(); + const outcome = await runDefinition(def, { + runner, verifier: new SeqVerifier([true]), human: new ScriptedHumanIO(), baseDir: process.cwd(), ctx, onEvent, + ctxGrants: ["skills", "agents", "mcps", "harnesses"], + ownModel: { provider: "ollama", model: "ollama/llama3.1" }, + }); + // The engine threads the file's grants + own model into the provision input. + assert.deepEqual(ctx.provisionCalls[0].permissions, ["skills", "agents", "mcps", "harnesses"]); + assert.deepEqual(ctx.provisionCalls[0].ownModel, { provider: "ollama", model: "ollama/llama3.1" }); + // Skills merge into the plan; mcps/harnesses are recommend-only and must NOT enter the skill set. + assert.deepEqual(runner.planCalls[0].skills, ["fastapi-patterns"]); + const ev = events.find((e) => e.type === "ctx" && e.action === "provision"); + assert.deepEqual(ev.mcps, ["local-ollama-files"]); + assert.deepEqual(ev.harnesses, ["autogen"]); + assert.equal(ev.harnessInstall, "ctx-harness-install autogen --dry-run"); + assert.equal(outcome.satisfied, true); +}); diff --git a/spec/loop-spec.schema.json b/spec/loop-spec.schema.json index 8d9a0d0..6cc2f08 100644 --- a/spec/loop-spec.schema.json +++ b/spec/loop-spec.schema.json @@ -98,6 +98,21 @@ "properties": { "provider": { "type": "string", "enum": ["ctx"] } } + }, + "ctxGrants": { + "type": "array", + "description": "Capability groups this file grants ctx (`grant ctx: skills, agents, mcps, harnesses`). Threaded to ctx as permissions; fails closed, default skills+agents. mcps/harnesses are recommend-only; harnesses also require `ownModel`.", + "items": { "type": "string", "enum": ["skills", "agents", "mcps", "harnesses"] } + }, + "ownModel": { + "type": "object", + "additionalProperties": false, + "required": ["provider", "model"], + "description": "A user-owned/local/API model (`ctx may use my own model \"/\"`) — unlocks ctx harness recommendations (gated, dry-run only).", + "properties": { + "provider": { "type": "string" }, + "model": { "type": "string" } + } } } }, From e9af92b07600f50d5dc9cb0129b7f515b526044c Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Tue, 30 Jun 2026 14:38:19 -0400 Subject: [PATCH 02/14] feat(ctx): warn when a declared own-model's local binary is missing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `ctx may use my own model "ollama/…"` only unlocks ctx harness recommendations; it never makes the loop run on that model. Warn at run time when the provider's local binary (e.g. ollama) isn't on PATH, so the author isn't surprised that running a recommended harness later would fail. API providers (no local binary) never warn. New ownModel.ts (ownModelBinaryWarning + commandOnPath), wired into the cli ctx path, +1 runtime test (injected PATH lookup). Co-Authored-By: Claude Opus 4.8 (1M context) --- packages/runtime/src/cli.ts | 6 ++++ packages/runtime/src/index.ts | 1 + packages/runtime/src/ownModel.ts | 51 ++++++++++++++++++++++++++++ packages/runtime/test/engine.test.js | 14 +++++++- 4 files changed, 71 insertions(+), 1 deletion(-) create mode 100644 packages/runtime/src/ownModel.ts diff --git a/packages/runtime/src/cli.ts b/packages/runtime/src/cli.ts index a58553d..f5a49eb 100644 --- a/packages/runtime/src/cli.ts +++ b/packages/runtime/src/cli.ts @@ -11,6 +11,7 @@ import { CliHumanIO } from "./human.js"; import { ClaudeCodeRunner } from "./runners/claudeCode.js"; import { IpcHumanIO } from "./ipc.js"; import { ShellGitIO } from "./runners/shellGit.js"; +import { ownModelBinaryWarning } from "./ownModel.js"; import type { LoopEvent } from "./types.js"; import { summarizeModels, formatModelSummary, summarizeOpex, formatOpexSummary } from "./summary.js"; @@ -302,6 +303,11 @@ async function main() { } catch (err) { console.error(`⚠ ctx skill source requested but the MCP client could not load: ${String((err as Error)?.message ?? err)}`); } + // A declared own-model (`ctx may use my own model …`) only unlocks ctx harness recommendations; + // it never makes the loop run on that model. Warn if its local binary is missing so the author + // isn't surprised that running a recommended harness later would fail. + const ownModelWarning = ownModelBinaryWarning(file.config?.ownModel); + if (ownModelWarning) console.error(ownModelWarning); } // `--events`: machine-readable NDJSON protocol for a UI host (e.g. the VSCode diff --git a/packages/runtime/src/index.ts b/packages/runtime/src/index.ts index 7d1b896..a4c3216 100644 --- a/packages/runtime/src/index.ts +++ b/packages/runtime/src/index.ts @@ -1,5 +1,6 @@ export * from "./types.js"; export { run, runDefinition } from "./engine.js"; +export { ownModelBinaryWarning, commandOnPath } from "./ownModel.js"; export { ShellVerifier } from "./verify.js"; export type { ShellVerifierOptions } from "./verify.js"; export { CliHumanIO, ScriptedHumanIO } from "./human.js"; diff --git a/packages/runtime/src/ownModel.ts b/packages/runtime/src/ownModel.ts new file mode 100644 index 0000000..dee4044 --- /dev/null +++ b/packages/runtime/src/ownModel.ts @@ -0,0 +1,51 @@ +import { existsSync } from "node:fs"; +import { join, delimiter } from "node:path"; + +/** + * Local model providers that ship a CLI binary we can detect on PATH. API providers + * (openai, anthropic, openrouter, litellm, …) authenticate by key — there is no binary to + * check, so a declared `ctx may use my own model "openai/…"` never warns. + */ +const LOCAL_MODEL_BINARIES: Record = { + ollama: "ollama", +}; + +/** True if `bin` resolves to an executable on PATH. Cross-platform, no subprocess. */ +export function commandOnPath(bin: string): boolean { + const exts = + process.platform === "win32" + ? (process.env.PATHEXT ?? ".EXE;.CMD;.BAT").split(";") + : [""]; + for (const dir of (process.env.PATH ?? "").split(delimiter)) { + if (!dir) continue; + for (const ext of exts) { + try { + if (existsSync(join(dir, bin + ext))) return true; + } catch { + /* unreadable PATH entry — skip it */ + } + } + } + return false; +} + +/** + * Warn when a `.loop` declares its own local model (`ctx may use my own model …`) but the + * provider's binary isn't installed. Returns the warning string, or null when there is nothing + * to warn about: no model declared, an API/unknown provider with no local binary, or the binary + * is present. Pure — pass `onPath` to test without touching the real PATH. + */ +export function ownModelBinaryWarning( + ownModel: { provider: string; model: string } | undefined, + onPath: (bin: string) => boolean = commandOnPath, +): string | null { + if (!ownModel) return null; + const bin = LOCAL_MODEL_BINARIES[ownModel.provider.trim().toLowerCase()]; + if (!bin) return null; // API/unknown provider — no local binary to check + if (onPath(bin)) return null; + return ( + `⚠ ctx: this file declares its own model "${ownModel.model}" (provider "${ownModel.provider}"), ` + + `but the \`${bin}\` binary isn't on PATH. The loop still runs on its normal runner; ctx will ` + + `recommend ${ownModel.provider} harnesses, but actually running one needs \`${bin}\` installed.` + ); +} diff --git a/packages/runtime/test/engine.test.js b/packages/runtime/test/engine.test.js index b3e9f9a..6c13150 100644 --- a/packages/runtime/test/engine.test.js +++ b/packages/runtime/test/engine.test.js @@ -4,7 +4,7 @@ import { readFileSync } from "node:fs"; import { fileURLToPath } from "node:url"; import { dirname, join } from "node:path"; import { parse } from "@loop-lang/parser"; -import { run, runDefinition, MockRunner, ScriptedHumanIO } from "../dist/index.js"; +import { run, runDefinition, MockRunner, ScriptedHumanIO, ownModelBinaryWarning } from "../dist/index.js"; import { MockGitIO } from "../dist/runners/mockGit.js"; const here = dirname(fileURLToPath(import.meta.url)); @@ -833,6 +833,18 @@ test("ctx: no top up when a loop does not ask for it", async () => { assert.equal(ctx.topupCalls.length, 0, "no top up without `top up skills from ctx`"); }); +test("ctx: own-model binary warning fires only for a missing LOCAL provider binary", () => { + const ollama = { provider: "ollama", model: "ollama/llama3.1" }; + // local provider, binary missing -> warn + assert.match(ownModelBinaryWarning(ollama, () => false), /`ollama` binary isn't on PATH/); + // local provider, binary present -> silent + assert.equal(ownModelBinaryWarning(ollama, () => true), null); + // API/unknown provider has no local binary -> silent even if onPath is false + assert.equal(ownModelBinaryWarning({ provider: "openai", model: "gpt-4o" }, () => false), null); + // no own-model declared -> silent + assert.equal(ownModelBinaryWarning(undefined, () => false), null); +}); + test("ctx: grants + own model thread to provision; mcps/harnesses surface but never merge into skills", async () => { const def = parse('loop "x":\n goal: build a local agent loop\n use skills recommended by ctx\n done when "t" passes').definitions[0]; const runner = new MockRunner(); From db782f20c857ea8d727d4b7fefd4137f06777c3e Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Tue, 30 Jun 2026 16:32:55 -0400 Subject: [PATCH 03/14] feat(runtime): LOOP_EVENTS_URL telemetry sink for the control plane Stream every LoopEvent to a collector when LOOP_EVENTS_URL is set (+ LOOP_EVENTS_TOKEN, LOOP_RUN_ID). Best-effort, fire-and-forget, flushes on exit; silent when unset/unreachable. +3 tests. Co-Authored-By: Claude Opus 4.8 (1M context) --- packages/runtime/src/cli.ts | 17 ++++-- packages/runtime/src/eventSink.ts | 71 +++++++++++++++++++++++++ packages/runtime/test/eventSink.test.js | 68 +++++++++++++++++++++++ 3 files changed, 153 insertions(+), 3 deletions(-) create mode 100644 packages/runtime/src/eventSink.ts create mode 100644 packages/runtime/test/eventSink.test.js diff --git a/packages/runtime/src/cli.ts b/packages/runtime/src/cli.ts index f5a49eb..cd84985 100644 --- a/packages/runtime/src/cli.ts +++ b/packages/runtime/src/cli.ts @@ -12,6 +12,7 @@ import { ClaudeCodeRunner } from "./runners/claudeCode.js"; import { IpcHumanIO } from "./ipc.js"; import { ShellGitIO } from "./runners/shellGit.js"; import { ownModelBinaryWarning } from "./ownModel.js"; +import { eventSinkFromEnv } from "./eventSink.js"; import type { LoopEvent } from "./types.js"; import { summarizeModels, formatModelSummary, summarizeOpex, formatOpexSummary } from "./summary.js"; @@ -310,6 +311,14 @@ async function main() { if (ownModelWarning) console.error(ownModelWarning); } + // Control-plane telemetry: when LOOP_EVENTS_URL is set, stream every event to the collector. + // Best-effort and off by default — no env, no sink, behaves exactly as before. + const eventSink = eventSinkFromEnv({ + loop_path: path, + principal: file.config?.runsAs, + runner: file.config?.runner, + }); + // `--events`: machine-readable NDJSON protocol for a UI host (e.g. the VSCode // extension) — streams Claude's live activity and answers human gates over stdin. if (rest.includes("--events")) { @@ -339,11 +348,12 @@ async function main() { flowStack: [path], modelPolicy: file.config?.models, cliModel: model, - onEvent: (e) => emit({ kind: "event", event: e }), + onEvent: (e) => { eventSink?.post(e); emit({ kind: "event", event: e }); }, }); const ok = outcomes.every((o) => o.satisfied); emit({ kind: "end", ok }); rl.close(); + await eventSink?.flush(); process.exit(ok ? 0 : 1); } @@ -382,7 +392,7 @@ async function main() { const html = renderLiveHtml(file, { title: basename(fileArg) }); const srv = await startLiveServer(html); console.error(`\n ↻ Loop live dashboard → http://127.0.0.1:${srv.port}\n`); - const outcomes = await run(file, { ...baseOpts, onEvent: (e) => { srv.emit(e); onTrace(e); } }); + const outcomes = await run(file, { ...baseOpts, onEvent: (e) => { eventSink?.post(e); srv.emit(e); onTrace(e); } }); reportSummary(); const ok = outcomes.every((o) => o.satisfied); // Keep the dashboard up after the run — a fast loop can finish before the browser even @@ -394,13 +404,14 @@ async function main() { return; } - const outcomes = await run(file, { ...baseOpts, onEvent: onTrace }); + const outcomes = await run(file, { ...baseOpts, onEvent: (e) => { eventSink?.post(e); onTrace(e); } }); reportSummary(); // Observability: when `observe:` is on, print the OpEx report (token burn made visible). if (file.config?.observe?.trace || file.config?.observe?.meter) { console.error(formatOpexSummary(summarizeOpex(traceEvents))); } const ok = outcomes.every((o) => o.satisfied); + await eventSink?.flush(); process.exit(ok ? 0 : 1); } diff --git a/packages/runtime/src/eventSink.ts b/packages/runtime/src/eventSink.ts new file mode 100644 index 0000000..978c4b6 --- /dev/null +++ b/packages/runtime/src/eventSink.ts @@ -0,0 +1,71 @@ +import { randomUUID } from "node:crypto"; +import type { LoopEvent } from "./types.js"; + +/** Run metadata sent alongside events so the collector can attribute a run. */ +export interface RunMeta { + loop_path?: string; + loop_name?: string; + git_sha?: string; + principal?: string; + runner?: string; +} + +export interface EventSink { + /** Fire-and-forget: enqueue one event for delivery. Never throws. */ + post(event: LoopEvent): void; + /** Await all in-flight deliveries (call before exit so the tail isn't lost). Never throws. */ + flush(): Promise; +} + +/** + * Stream Loop runtime events to a control-plane collector over HTTP. + * + * Best-effort by design: an unreachable or failing collector never throws and never blocks a + * run — telemetry must not be able to break a loop. The server is idempotent on the monotonic + * per-run `seq`, so out-of-order or retried delivery is safe. + */ +export function makeHttpEventSink(opts: { + url: string; + runId: string; + token?: string; + meta?: RunMeta; +}): EventSink { + const endpoint = `${opts.url.replace(/\/+$/, "")}/api/v1/runs/${encodeURIComponent(opts.runId)}/events`; + const headers: Record = { "content-type": "application/json" }; + if (opts.token) headers["x-api-token"] = opts.token; + let seq = 0; + const inflight = new Set>(); + return { + post(event) { + const body = JSON.stringify({ meta: opts.meta, events: [{ seq: seq++, event }] }); + const p = fetch(endpoint, { method: "POST", headers, body }) + .then((r) => r.body?.cancel?.()) + .catch(() => { + /* best-effort: telemetry must never break a run */ + }); + inflight.add(p); + void p.finally(() => inflight.delete(p)); + }, + async flush() { + await Promise.allSettled([...inflight]); + }, + }; +} + +/** + * Build a sink from the environment. Returns undefined when `LOOP_EVENTS_URL` is unset, so a run + * with no control plane configured streams nowhere and behaves exactly as before. + * LOOP_EVENTS_URL — collector base URL (required to enable) + * LOOP_EVENTS_TOKEN — shared API token (optional) + * LOOP_RUN_ID — correlate events to one run (optional; a UUID is generated if unset) + */ +export function eventSinkFromEnv(meta?: RunMeta): EventSink | undefined { + const url = process.env.LOOP_EVENTS_URL; + if (!url) return undefined; + return makeHttpEventSink({ + url, + runId: process.env.LOOP_RUN_ID || randomUUID(), + token: process.env.LOOP_EVENTS_TOKEN, + meta, + }); +} diff --git a/packages/runtime/test/eventSink.test.js b/packages/runtime/test/eventSink.test.js new file mode 100644 index 0000000..76c4599 --- /dev/null +++ b/packages/runtime/test/eventSink.test.js @@ -0,0 +1,68 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import http from "node:http"; +import { makeHttpEventSink } from "../dist/eventSink.js"; + +/** A throwaway HTTP server that records every request body. */ +function recordingServer(received) { + return http.createServer((req, res) => { + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + received.push({ url: req.url, token: req.headers["x-api-token"], body: JSON.parse(body) }); + res.writeHead(200, { "content-type": "application/json" }); + res.end('{"ok":true}'); + }); + }); +} + +test("http event sink posts each event with a monotonic seq + token to the run endpoint", async () => { + const received = []; + const srv = recordingServer(received); + await new Promise((r) => srv.listen(0, r)); + const port = srv.address().port; + + const sink = makeHttpEventSink({ + url: `http://127.0.0.1:${port}`, + runId: "r1", + token: "secret", + meta: { loop_path: "x.loop", principal: "ci" }, + }); + sink.post({ type: "loop-start", name: "n" }); + sink.post({ type: "ctx", action: "provision", skills: ["a"] }); + sink.post({ type: "loop-end", satisfied: true }); + await sink.flush(); + srv.close(); + + assert.equal(received.length, 3); + for (const r of received) { + assert.equal(r.url, "/api/v1/runs/r1/events"); + assert.equal(r.token, "secret"); + assert.equal(r.body.meta.loop_path, "x.loop"); + } + // seq is assigned synchronously at post() time, so it's deterministic regardless of arrival order. + const bySeq = new Map(received.map((r) => [r.body.events[0].seq, r.body.events[0].event])); + assert.deepEqual([...bySeq.keys()].sort((a, b) => a - b), [0, 1, 2]); + assert.equal(bySeq.get(0).type, "loop-start"); + assert.equal(bySeq.get(1).type, "ctx"); +}); + +test("sink degrades silently when the collector is unreachable (never throws)", async () => { + const sink = makeHttpEventSink({ url: "http://127.0.0.1:1", runId: "r2" }); // nothing listening + sink.post({ type: "loop-start" }); + await assert.doesNotReject(sink.flush()); +}); + +test("no token header when none configured", async () => { + const received = []; + const srv = recordingServer(received); + await new Promise((r) => srv.listen(0, r)); + const port = srv.address().port; + const sink = makeHttpEventSink({ url: `http://127.0.0.1:${port}/`, runId: "r3" }); + sink.post({ type: "stop", reason: "done" }); + await sink.flush(); + srv.close(); + assert.equal(received.length, 1); + assert.equal(received[0].token, undefined); + assert.equal(received[0].url, "/api/v1/runs/r3/events"); // trailing slash on base url trimmed +}); From 31781631a8ea876180e65424716d8761c7687bb8 Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Wed, 1 Jul 2026 07:47:59 -0400 Subject: [PATCH 04/14] refactor: move root .loop files to examples/ - Moved agentic-engineering.loop from root to examples/ - Moved demo.loop from root to examples/ - Updated references in test and docs to reflect new paths Co-Authored-By: Claude Haiku 4.5 --- docs/agentic-engineering-plan.md | 2 +- agentic-engineering.loop => examples/agentic-engineering.loop | 0 demo.loop => examples/demo.loop | 0 packages/runtime/test/examples.test.js | 2 +- 4 files changed, 2 insertions(+), 2 deletions(-) rename agentic-engineering.loop => examples/agentic-engineering.loop (100%) rename demo.loop => examples/demo.loop (100%) diff --git a/docs/agentic-engineering-plan.md b/docs/agentic-engineering-plan.md index 6d74141..f38e6fb 100644 --- a/docs/agentic-engineering-plan.md +++ b/docs/agentic-engineering-plan.md @@ -1,7 +1,7 @@ # Plan — Agentic Engineering constructs for Loop > Bringing agentic-engineering discipline into Loop. -> Branch: `claude/loop-lang-concepts-3p5tz1`. Companion pipeline: [`agentic-engineering.loop`](../agentic-engineering.loop). +> Branch: `claude/loop-lang-concepts-3p5tz1`. Companion pipeline: [`agentic-engineering.loop`](../examples/agentic-engineering.loop). ## Context diff --git a/agentic-engineering.loop b/examples/agentic-engineering.loop similarity index 100% rename from agentic-engineering.loop rename to examples/agentic-engineering.loop diff --git a/demo.loop b/examples/demo.loop similarity index 100% rename from demo.loop rename to examples/demo.loop diff --git a/packages/runtime/test/examples.test.js b/packages/runtime/test/examples.test.js index 7e37d01..c39325b 100644 --- a/packages/runtime/test/examples.test.js +++ b/packages/runtime/test/examples.test.js @@ -22,7 +22,7 @@ function findLoops(dir, acc = []) { return acc; } -const files = [...findLoops(join(root, "examples")), join(root, "agentic-engineering.loop")]; +const files = findLoops(join(root, "examples")); // A verifier that always passes; human IO that always approves — so every example reaches done. class PassVerifier { async verify() { return { passed: true, output: "ok" }; } } From b04840c335e0311068ebb6899bb322617760a2cc Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Wed, 1 Jul 2026 16:06:54 -0400 Subject: [PATCH 05/14] =?UTF-8?q?feat(runtime):=20local=20NDJSON=20event?= =?UTF-8?q?=20log=20=E2=80=94=20LOOP=5FLOG=5FFILE=20env=20+=20--log=20flag?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Records the full loop event stream to a local file, the durable sibling of the LOOP_EVENTS_URL control-plane sink. Every event the engine already emits (loop/node/observe/reflect/transition/human/ctx/git/hook/model/stop) appends as one NDJSON line: a `loop.log.v1` header then `{seq, ts, event}` per event. - makeFileEventSink: synchronous appendFileSync per event, so the log is durable the instant it fires — survives Ctrl-C / crash with no lost tail (flush() is a no-op). Best-effort like the HTTP sink: creates the parent dir once, disables itself quietly on any write error, never throws. - combineSinks: fan one event stream out to HTTP + file, sharing one runId so the collector and the local log correlate. eventSinkFromEnv wires both, so all three run paths (--events, --live, default) log with no cli.ts churn. - --log flag overrides LOOP_LOG_FILE for one-off runs. Off by default: no env, no flag → no sink, behaves exactly as before. Tests: 5 sink + override/precedence cases; runtime suite 111/111. Co-Authored-By: Claude Opus 4.8 (1M context) --- packages/runtime/src/cli.ts | 23 +++-- packages/runtime/src/eventSink.ts | 87 +++++++++++++++--- packages/runtime/test/eventSink.test.js | 114 +++++++++++++++++++++++- 3 files changed, 203 insertions(+), 21 deletions(-) diff --git a/packages/runtime/src/cli.ts b/packages/runtime/src/cli.ts index cd84985..fc1faf0 100644 --- a/packages/runtime/src/cli.ts +++ b/packages/runtime/src/cli.ts @@ -231,7 +231,7 @@ async function main() { } if (!cmd || (cmd !== "run" && cmd !== "parse" && cmd !== "viz" && cmd !== "live") || !fileArg) { - console.error("usage: loop-run [--model ] [--live] [--events] [--out ]"); + console.error("usage: loop-run [--model ] [--live] [--events] [--log ] [--out ]"); process.exit(2); } @@ -289,6 +289,9 @@ async function main() { const modelIdx = rest.indexOf("--model"); const model = modelIdx >= 0 ? rest[modelIdx + 1] : undefined; + // `--log ` writes the full event stream to a local NDJSON log (overrides LOOP_LOG_FILE). + const logIdx = rest.indexOf("--log"); + const logFile = logIdx >= 0 ? rest[logIdx + 1] : undefined; const target = file.config?.target ? resolve(baseDir, file.config.target) : baseDir; const git = new ShellGitIO(); @@ -311,13 +314,17 @@ async function main() { if (ownModelWarning) console.error(ownModelWarning); } - // Control-plane telemetry: when LOOP_EVENTS_URL is set, stream every event to the collector. - // Best-effort and off by default — no env, no sink, behaves exactly as before. - const eventSink = eventSinkFromEnv({ - loop_path: path, - principal: file.config?.runsAs, - runner: file.config?.runner, - }); + // Telemetry: stream every event to the control-plane collector (LOOP_EVENTS_URL) and/or append + // it to a local NDJSON log (LOOP_LOG_FILE). Best-effort and off by default — no env, no sink, + // behaves exactly as before. + const eventSink = eventSinkFromEnv( + { + loop_path: path, + principal: file.config?.runsAs, + runner: file.config?.runner, + }, + { logFile } + ); // `--events`: machine-readable NDJSON protocol for a UI host (e.g. the VSCode // extension) — streams Claude's live activity and answers human gates over stdin. diff --git a/packages/runtime/src/eventSink.ts b/packages/runtime/src/eventSink.ts index 978c4b6..0c6b41e 100644 --- a/packages/runtime/src/eventSink.ts +++ b/packages/runtime/src/eventSink.ts @@ -1,4 +1,6 @@ import { randomUUID } from "node:crypto"; +import { appendFileSync, mkdirSync } from "node:fs"; +import { dirname } from "node:path"; import type { LoopEvent } from "./types.js"; /** Run metadata sent alongside events so the collector can attribute a run. */ @@ -53,19 +55,80 @@ export function makeHttpEventSink(opts: { } /** - * Build a sink from the environment. Returns undefined when `LOOP_EVENTS_URL` is unset, so a run - * with no control plane configured streams nowhere and behaves exactly as before. - * LOOP_EVENTS_URL — collector base URL (required to enable) - * LOOP_EVENTS_TOKEN — shared API token (optional) + * Append every Loop runtime event to a local NDJSON log file — one JSON object per line. + * + * The first line is a `loop.log.v1` header (runId + meta); every subsequent line is + * `{ seq, ts, event }`. Writes are synchronous appends, so each event is durable the instant it + * fires — the log survives a Ctrl-C or crash with no lost tail (nothing is buffered, so `flush()` + * is a no-op). Best-effort like the HTTP sink: a missing directory is created once, and any write + * error disables the sink quietly rather than breaking the run. + */ +export function makeFileEventSink(opts: { path: string; runId: string; meta?: RunMeta }): EventSink { + let seq = 0; + let broken = false; + let started = false; + const line = (obj: unknown): void => { + if (broken) return; + try { + appendFileSync(opts.path, JSON.stringify(obj) + "\n"); + } catch { + broken = true; + } + }; + const start = (): void => { + if (started) return; + started = true; + try { + mkdirSync(dirname(opts.path), { recursive: true }); + } catch { + /* dir may already exist or be uncreatable — the append below decides if we can log */ + } + line({ v: "loop.log.v1", runId: opts.runId, ts: new Date().toISOString(), meta: opts.meta }); + }; + return { + post(event) { + start(); + line({ seq: seq++, ts: new Date().toISOString(), event }); + }, + async flush() { + /* appendFileSync is durable — nothing is buffered */ + }, + }; +} + +/** Fan one event stream out to several sinks. Returns undefined when none are active. */ +export function combineSinks(sinks: Array): EventSink | undefined { + const active = sinks.filter((s): s is EventSink => !!s); + if (!active.length) return undefined; + if (active.length === 1) return active[0]; + return { + post(event) { + for (const s of active) s.post(event); + }, + async flush() { + await Promise.allSettled(active.map((s) => s.flush())); + }, + }; +} + +/** + * Build a sink from the environment (plus optional explicit overrides). Returns undefined when no + * telemetry is configured, so a run with no sink streams nowhere and behaves exactly as before. + * When more than one is active, events fan out to all — sharing one run id so the HTTP collector + * and the local log correlate. + * LOOP_EVENTS_URL — control-plane collector base URL (enables the HTTP sink) + * LOOP_EVENTS_TOKEN — shared API token for the collector (optional) + * LOOP_LOG_FILE — local NDJSON log path (enables the file sink) * LOOP_RUN_ID — correlate events to one run (optional; a UUID is generated if unset) + * + * `override.logFile` (the CLI's `--log `) takes precedence over `LOOP_LOG_FILE`. */ -export function eventSinkFromEnv(meta?: RunMeta): EventSink | undefined { +export function eventSinkFromEnv(meta?: RunMeta, override?: { logFile?: string }): EventSink | undefined { + const runId = process.env.LOOP_RUN_ID || randomUUID(); const url = process.env.LOOP_EVENTS_URL; - if (!url) return undefined; - return makeHttpEventSink({ - url, - runId: process.env.LOOP_RUN_ID || randomUUID(), - token: process.env.LOOP_EVENTS_TOKEN, - meta, - }); + const logPath = override?.logFile || process.env.LOOP_LOG_FILE; + return combineSinks([ + url ? makeHttpEventSink({ url, runId, token: process.env.LOOP_EVENTS_TOKEN, meta }) : undefined, + logPath ? makeFileEventSink({ path: logPath, runId, meta }) : undefined, + ]); } diff --git a/packages/runtime/test/eventSink.test.js b/packages/runtime/test/eventSink.test.js index 76c4599..6265dd7 100644 --- a/packages/runtime/test/eventSink.test.js +++ b/packages/runtime/test/eventSink.test.js @@ -1,7 +1,10 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import http from "node:http"; -import { makeHttpEventSink } from "../dist/eventSink.js"; +import { readFileSync, rmSync, existsSync, mkdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { tmpdir } from "node:os"; +import { makeHttpEventSink, makeFileEventSink, combineSinks, eventSinkFromEnv } from "../dist/eventSink.js"; /** A throwaway HTTP server that records every request body. */ function recordingServer(received) { @@ -66,3 +69,112 @@ test("no token header when none configured", async () => { assert.equal(received[0].token, undefined); assert.equal(received[0].url, "/api/v1/runs/r3/events"); // trailing slash on base url trimmed }); + +test("file sink writes a header line then one NDJSON line per event", async () => { + const dir = join(tmpdir(), `loop-log-${process.pid}-${Date.now()}`); + const path = join(dir, "nested", "run.log"); // nested dir must be created + try { + const sink = makeFileEventSink({ path, runId: "r1", meta: { loop_path: "x.loop", principal: "ci" } }); + sink.post({ type: "loop-start", name: "n" }); + sink.post({ type: "observe", passed: true, output: "ok" }); + sink.post({ type: "loop-end", satisfied: true }); + await sink.flush(); + + const lines = readFileSync(path, "utf8").trim().split("\n").map((l) => JSON.parse(l)); + assert.equal(lines.length, 4); // header + 3 events + + // header + assert.equal(lines[0].v, "loop.log.v1"); + assert.equal(lines[0].runId, "r1"); + assert.equal(lines[0].meta.loop_path, "x.loop"); + assert.ok(lines[0].ts, "header has a timestamp"); + + // events carry a monotonic seq starting at 0, a ts, and the original event + assert.deepEqual(lines.slice(1).map((l) => l.seq), [0, 1, 2]); + assert.equal(lines[1].event.type, "loop-start"); + assert.equal(lines[2].event.type, "observe"); + assert.equal(lines[3].event.type, "loop-end"); + assert.ok(lines[1].ts, "event has a timestamp"); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); + +test("file sink degrades silently on an unwritable path (never throws)", async () => { + // A path whose parent is a file, not a directory → mkdir/append both fail. + const dir = join(tmpdir(), `loop-log-bad-${process.pid}-${Date.now()}`); + const blocker = join(dir, "blocker"); + const path = join(blocker, "run.log"); // blocker will be a file, so this dir can't exist + try { + mkdirSync(dir, { recursive: true }); + writeFileSync(blocker, "x"); + const sink = makeFileEventSink({ path, runId: "r2" }); + assert.doesNotThrow(() => sink.post({ type: "loop-start" })); + assert.doesNotThrow(() => sink.post({ type: "loop-end", satisfied: false })); + await assert.doesNotReject(sink.flush()); + assert.ok(!existsSync(path), "nothing was written"); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); + +test("combineSinks fans one event out to every sink and flushes all", async () => { + const dir = join(tmpdir(), `loop-log-combine-${process.pid}-${Date.now()}`); + const path = join(dir, "run.log"); + const received = []; + const srv = recordingServer(received); + await new Promise((r) => srv.listen(0, r)); + const port = srv.address().port; + try { + const combined = combineSinks([ + makeHttpEventSink({ url: `http://127.0.0.1:${port}`, runId: "rc" }), + makeFileEventSink({ path, runId: "rc" }), + undefined, // an inactive sink is skipped + ]); + combined.post({ type: "loop-start", name: "n" }); + combined.post({ type: "loop-end", satisfied: true }); + await combined.flush(); + srv.close(); + + assert.equal(received.length, 2, "HTTP sink got both events"); + const fileLines = readFileSync(path, "utf8").trim().split("\n"); + assert.equal(fileLines.length, 3, "file sink got header + both events"); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); + +test("combineSinks returns undefined when no sink is active", () => { + assert.equal(combineSinks([undefined, undefined]), undefined); +}); + +test("eventSinkFromEnv is inert with no env and no override", () => { + const saved = { url: process.env.LOOP_EVENTS_URL, log: process.env.LOOP_LOG_FILE }; + delete process.env.LOOP_EVENTS_URL; + delete process.env.LOOP_LOG_FILE; + try { + assert.equal(eventSinkFromEnv({ loop_path: "x.loop" }), undefined); + } finally { + if (saved.url !== undefined) process.env.LOOP_EVENTS_URL = saved.url; + if (saved.log !== undefined) process.env.LOOP_LOG_FILE = saved.log; + } +}); + +test("eventSinkFromEnv: --log override beats LOOP_LOG_FILE", async () => { + const dir = join(tmpdir(), `loop-log-override-${process.pid}-${Date.now()}`); + const envPath = join(dir, "from-env.log"); + const flagPath = join(dir, "from-flag.log"); + const saved = process.env.LOOP_LOG_FILE; + process.env.LOOP_LOG_FILE = envPath; + try { + const sink = eventSinkFromEnv({ loop_path: "x.loop" }, { logFile: flagPath }); + sink.post({ type: "loop-start", name: "n" }); + await sink.flush(); + assert.ok(existsSync(flagPath), "flag path was written"); + assert.ok(!existsSync(envPath), "env path was NOT written (flag overrode it)"); + } finally { + if (saved !== undefined) process.env.LOOP_LOG_FILE = saved; + else delete process.env.LOOP_LOG_FILE; + rmSync(dir, { recursive: true, force: true }); + } +}); From c4f0e4abb8a915ef21cf5fa5f5bfcca5b1a6b0d9 Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Wed, 1 Jul 2026 16:09:50 -0400 Subject: [PATCH 06/14] docs: document the event log & telemetry (--log / LOOP_LOG_FILE / LOOP_EVENTS_URL) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit MANUAL.md §4 gains an "Event log & telemetry" section: what the event stream records, the NDJSON format (header + seq'd event lines) with jq read-back examples, durability + best-effort guarantees, and the env-var table for the local log and the remote HTTP collector. Usage line + flag list pick up --log. README points at it from the live-dashboard section. Co-Authored-By: Claude Opus 4.8 (1M context) --- README.md | 9 +++++++ docs/MANUAL.md | 68 +++++++++++++++++++++++++++++++++++++++++++++++++- 2 files changed, 76 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 4e2f3d8..60152d6 100644 --- a/README.md +++ b/README.md @@ -192,6 +192,15 @@ When you run a loop via `/loopflow`, the skill asks if you want the dashboard an opens it and updates it as each step happens — pipeline stages, flow steps, and sprint stories filling in as the loop progresses. +Prefer a file you can grep later? Persist the same event stream as NDJSON: + +``` +loop-run run file.loop --log run.log # append every event to a local log (also via LOOP_LOG_FILE) +``` + +See **Event log & telemetry** in [`docs/MANUAL.md`](docs/MANUAL.md) for the format and the +`LOOP_EVENTS_URL` remote collector. + ## Project layout | Package | Purpose | diff --git a/docs/MANUAL.md b/docs/MANUAL.md index ba69a5a..800078c 100644 --- a/docs/MANUAL.md +++ b/docs/MANUAL.md @@ -116,7 +116,7 @@ the thrash guard). ## 4. The CLI ``` -loop-run [--model ] [--live] [--out ] +loop-run [--model ] [--live] [--log ] [--out ] ``` | Command | What it does | @@ -136,6 +136,8 @@ loop-run [--model ] Code. Omit to use the CLI default. - `--live` — for `run`, open a live browser dashboard and stream every step to it as the loop executes (see below). +- `--log ` — for `run`, append the full event stream to a local NDJSON log + (see **Event log & telemetry** below). Overrides `LOOP_LOG_FILE`. - `--out ` — for `viz`, the HTML output file (default: the `.loop` name with an `.html` extension). - `--json` — print the parsed loop-spec JSON before the command's normal work (handy with @@ -168,6 +170,70 @@ on connect (and dedupes on reconnect via `Last-Event-ID`), so events fired befor connects are not lost and a transient drop doesn't double-deliver. The dashboard is self-contained (no external assets) and binds to `127.0.0.1` only. +### Event log & telemetry + +Every meaningful thing a run does is a **structured event** — `loop-start`, each +`node-enter` / `node-exit` (with the attempt number), `observe` (pass/fail + output), +`transition`, `reflect`, `loop-back`, human gates, `ctx` provision/top-up, `git` actions, +`hook` results, the `model` tier per phase, `stop` (with the reason), `loop-end`, and the +pipeline / flow / for-each envelopes. The live dashboard renders this stream; you can also +**persist it** — to a local file and/or a remote collector — for auditing, debugging a +thrashing loop, or metering cost. + +**Off by default.** With no flag and no env var, a run persists nothing and behaves exactly +as before. Persistence is **best-effort**: a failing log or an unreachable collector never +throws and never blocks a run — telemetry can't break a loop. + +#### Local log — `--log` / `LOOP_LOG_FILE` + +```bash +loop-run run test.loop --log run.log # one-off (overrides LOOP_LOG_FILE) +LOOP_LOG_FILE=run.log loop-run run test.loop # env — same effect +``` + +The log is **NDJSON**: one JSON object per line, so it streams and greps cleanly. The first +line is a `loop.log.v1` header (the run id + metadata); every line after is one event with a +monotonic `seq` and an ISO timestamp: + +```jsonc +{"v":"loop.log.v1","runId":"…","ts":"2026-07-01T20:04:43.443Z","meta":{"loop_path":"test.loop","principal":"idan"}} +{"seq":0,"ts":"…","event":{"type":"loop-start","name":"test loop"}} +{"seq":1,"ts":"…","event":{"type":"node-enter","node":"plan","attempt":1}} +{"seq":2,"ts":"…","event":{"type":"observe","passed":false,"output":"1 failing test"}} +{"seq":3,"ts":"…","event":{"type":"reflect","focus":"which layer broke","text":"the API returned 500"}} +{"seq":4,"ts":"…","event":{"type":"stop","reason":"done"}} +``` + +Each event is written with a synchronous append, so it's on disk the instant it fires — the +log survives a `Ctrl-C` or a crash with **no lost tail**. The parent directory is created if +missing. It works the same across all run modes (default, `--live`, `--events`). + +Read it back with any NDJSON tool — e.g. with `jq`: + +```bash +jq 'select(.event.type == "observe")' run.log # every verification result +jq -r 'select(.event.type=="reflect") | .event.text' run.log # what each failure taught it +jq 'select(.event.type == "stop") | .event.reason' run.log # how it ended +``` + +#### Remote collector — `LOOP_EVENTS_URL` + +To stream the same events to a control plane over HTTP, set `LOOP_EVENTS_URL` (and, if the +collector needs it, `LOOP_EVENTS_TOKEN`). Events POST to +`/api/v1/runs//events`, each carrying the monotonic `seq` so the server is +idempotent on retries and out-of-order delivery. + +| Env var | Meaning | +|---|---| +| `LOOP_LOG_FILE` | Local NDJSON log path (the `--log` flag overrides it). | +| `LOOP_EVENTS_URL` | Control-plane collector base URL — enables the HTTP sink. | +| `LOOP_EVENTS_TOKEN` | Shared API token, sent as `x-api-token` (optional). | +| `LOOP_RUN_ID` | Correlate a run's events across sinks (optional; a UUID is generated if unset). | + +The file log and the HTTP collector can run **together** — the same event fans out to both, +sharing one `runId`, so a local trace and the control-plane record line up. Set `LOOP_RUN_ID` +yourself when you want a run's id to match something you already track (a CI job, a ticket). + ## 5. Language reference A `.loop` file is indentation-structured. `loop` / `pipeline` sit at column 0; their body From d76278de2bb821b3844625bc950a009f50ba9b75 Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Wed, 1 Jul 2026 16:30:47 -0400 Subject: [PATCH 07/14] =?UTF-8?q?feat(parser,runtime):=20flake-guard=20pre?= =?UTF-8?q?dicate=20=E2=80=94=20`done=20when=20""=20passes=20N=20time?= =?UTF-8?q?s`?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Re-runs a test/command `done when` check N times and requires EVERY run to pass, so "done" means "passes reliably" not "passed once" — a guard against a green that only holds by luck (timing- or order-dependent tests). - parser: an optional `N times` suffix on `passes` / `succeeds` / `finds nothing` and `the test "…" passes`, surfaced as `runs` on the predicate. Only set when > 1, so a plain check (and "1 time") keep the existing single-run shape. The `check:`/`verify:` sugar accepts it too. - verify: ShellVerifier loops `runs` times; the first failing run short-circuits (labelled `run i/N failed —`); a clean pass reports `(passed N/N runs)`. - show/explain: rendered as `×N` (ASCII) and "N times in a row" (prose). Docs: AGENTS.md + MANUAL.md predicate sections. Parser 58/58, runtime 115/115 (4 new verifier tests prove the exact run count + short-circuit). Co-Authored-By: Claude Opus 4.8 (1M context) --- AGENTS.md | 6 +++ docs/MANUAL.md | 8 ++++ packages/parser/src/parser.ts | 28 ++++++++----- packages/parser/src/types.ts | 6 ++- packages/parser/test/parser.test.js | 26 ++++++++++++ packages/runtime/src/show.ts | 11 +++-- packages/runtime/src/verify.ts | 18 ++++++-- packages/runtime/test/verify.test.js | 62 ++++++++++++++++++++++++++++ 8 files changed, 146 insertions(+), 19 deletions(-) create mode 100644 packages/runtime/test/verify.test.js diff --git a/AGENTS.md b/AGENTS.md index 395e67a..67f9888 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -116,6 +116,7 @@ stages in parallel: (inside a pipeline: the indented stages run concurrently) done when the test "billing.spec.ts::apostrophe" passes # a named test done when "pnpm test" passes # a shell command, exit 0 done when "semgrep --severity=high" finds nothing # a shell command, empty output +done when "pnpm test flaky" passes 3 times # flake guard: re-run the check, EVERY run must pass done when a human confirms "looks right at 375px" # a human check done when the skill "email-review" approves # an eval: approved / not done when the skill "email-review" scores 8 or more # an eval: numeric threshold @@ -127,6 +128,11 @@ done when the skill "path-review" approves on the trajectory # an eval The command in a predicate runs in the user's shell with their privileges (like an npm script). It IS meant to be a real command. Prefer a fast, deterministic check. +**Flake guard — `passes N times`.** Append `N times` to a `test` or command predicate to re-run +it `N` times and require every run to pass (the first failure short-circuits). Reach for it when a +green can pass by luck — a timing- or order-dependent test — so "done" means "passes *reliably*", +not "passed *once*". + ### Tests vs evals — list as many `done when` as you need A loop may have **multiple `done when` lines, and ALL must pass** (a conjunction). Use this to diff --git a/docs/MANUAL.md b/docs/MANUAL.md index 800078c..ebbd32b 100644 --- a/docs/MANUAL.md +++ b/docs/MANUAL.md @@ -362,6 +362,7 @@ what you intend. done when the test "billing.spec.ts::apostrophe" passes # a named test done when "pnpm test" passes # shell command, exit 0 (`succeeds` also works) done when "semgrep --severity=high" finds nothing # shell command, empty stdout +done when "pnpm test flaky" passes 3 times # flake guard: re-run, every run must pass done when a human confirms "looks right at 375px" # a human check done when the skill "email-review" approves # an eval: approved / not done when the skill "email-review" scores 8 or more # an eval: numeric threshold @@ -373,6 +374,13 @@ done when the skill "path-review" approves on the trajectory # an eval The command runs in your shell with your privileges (like an npm script). Keep it fast and deterministic. +**Flake guard — `passes N times`.** Append `N times` to a `test` or command predicate +(`passes` / `succeeds` / `finds nothing`) to re-run the check `N` times and require **every** +run to pass. The first failing run short-circuits (the rest don't execute). Use it when a +green can hold by luck — a timing-dependent or order-dependent test — so "done" means +"passes *reliably*", not "passed *once*". `show` renders it as `×N`; a plain check (or +`1 time`) is the usual single run. + #### Tests vs evals A loop can list **several `done when` lines, and all must pass.** Use this to combine the diff --git a/packages/parser/src/parser.ts b/packages/parser/src/parser.ts index 4657852..23a83c3 100644 --- a/packages/parser/src/parser.ts +++ b/packages/parser/src/parser.ts @@ -124,17 +124,25 @@ function quoted(s: string): string | null { function parsePredicate(s: string, lineNo: number): Predicate { const text = s.trim(); - // the test "X" passes - let m = text.match(/^the test\s+"([^"]+)"\s+passes$/i); - if (m) return { type: "test", target: m[1] }; - // "CMD" finds nothing -> command, expect empty - m = text.match(/^"([^"]+)"\s+finds nothing$/i); - if (m) return { type: "command", command: m[1], expect: "empty" }; + // An optional `N times` suffix re-runs a check to guard against a flaky green — every run must + // pass. Only surfaced as `runs` when > 1 (so a plain check, and "1 time", stay the default shape). + const times = (n: string | undefined): { runs?: number } => { + const r = n ? parseInt(n, 10) : 1; + return r > 1 ? { runs: r } : {}; + }; + + // the test "X" passes [N times] + let m = text.match(/^the test\s+"([^"]+)"\s+passes(?:\s+(\d+)\s+times?)?$/i); + if (m) return { type: "test", target: m[1], ...times(m[2]) }; + + // "CMD" finds nothing [N times] -> command, expect empty + m = text.match(/^"([^"]+)"\s+finds nothing(?:\s+(\d+)\s+times?)?$/i); + if (m) return { type: "command", command: m[1], expect: "empty", ...times(m[2]) }; - // "CMD" passes / succeeds -> command, exit-zero - m = text.match(/^"([^"]+)"\s+(?:passes|succeeds)$/i); - if (m) return { type: "command", command: m[1], expect: "exit-zero" }; + // "CMD" passes / succeeds [N times] -> command, exit-zero + m = text.match(/^"([^"]+)"\s+(?:passes|succeeds)(?:\s+(\d+)\s+times?)?$/i); + if (m) return { type: "command", command: m[1], expect: "exit-zero", ...times(m[2]) }; // a human confirms "..." m = text.match(/^a human confirms\s+"([^"]+)"$/i); @@ -388,7 +396,7 @@ function interpretLoopBody(name: string | null, body: Line[], defaults?: ParseDe // command (`check: npm test`); a predicate phrase (`check: the skill "x" approves`) is parsed as-is. if ((m = t.match(/^(?:check|verify):\s*(.+)$/i))) { const val = m[1].trim(); - const isPhrase = /^(the test|the skill|a human)\b/i.test(val) || /^".*"\s+(passes|succeeds|finds nothing)$/i.test(val); + const isPhrase = /^(the test|the skill|a human)\b/i.test(val) || /^".*"\s+(passes|succeeds|finds nothing)(\s+\d+\s+times?)?$/i.test(val); const pred: Predicate = isPhrase ? parsePredicate(val, ln.lineNo) : { type: "command", command: val.replace(/^"|"$/g, ""), expect: "exit-zero" }; diff --git a/packages/parser/src/types.ts b/packages/parser/src/types.ts index 7e31ac2..7559af8 100644 --- a/packages/parser/src/types.ts +++ b/packages/parser/src/types.ts @@ -234,8 +234,10 @@ export interface PlanSource { } export type Predicate = - | { type: "test"; target: string } - | { type: "command"; command: string; expect?: "exit-zero" | "empty" } + // `runs` (from `… passes N times`) re-runs the check N times and requires every run to pass — + // a flake guard against a green that only holds by luck. Absent / 1 = the usual single run. + | { type: "test"; target: string; runs?: number } + | { type: "command"; command: string; expect?: "exit-zero" | "empty"; runs?: number } | { type: "human"; description: string } /** * An eval: a review skill judges the goal (approve / score). `subject` selects what it diff --git a/packages/parser/test/parser.test.js b/packages/parser/test/parser.test.js index 5f4519f..fa40c15 100644 --- a/packages/parser/test/parser.test.js +++ b/packages/parser/test/parser.test.js @@ -213,6 +213,32 @@ test("use skills: also accepts 'and' as a separator", () => { assert.deepEqual(loop.skills, ["a", "b", "c"]); }); +test("flake guard: `passes N times` carries a runs count on a command predicate", () => { + const loop = parse('loop "x":\n goal: g\n done when "pnpm test" passes 3 times').definitions[0]; + assert.deepEqual(loop.doneWhen, [{ type: "command", command: "pnpm test", expect: "exit-zero", runs: 3 }]); +}); + +test("flake guard: `finds nothing N times` and `the test passes N times` carry runs", () => { + const cmd = parse('loop "x":\n goal: g\n done when "semgrep" finds nothing 5 times').definitions[0]; + assert.deepEqual(cmd.doneWhen, [{ type: "command", command: "semgrep", expect: "empty", runs: 5 }]); + + const tst = parse('loop "x":\n goal: g\n done when the test "a::b" passes 2 times').definitions[0]; + assert.deepEqual(tst.doneWhen, [{ type: "test", target: "a::b", runs: 2 }]); +}); + +test("flake guard: a plain check (no `N times`) and `1 time` stay the single-run shape", () => { + const plain = parse('loop "x":\n goal: g\n done when "t" passes').definitions[0]; + assert.deepEqual(plain.doneWhen, [{ type: "command", command: "t", expect: "exit-zero" }]); // no `runs` key + + const one = parse('loop "x":\n goal: g\n done when "t" passes 1 time').definitions[0]; + assert.deepEqual(one.doneWhen, [{ type: "command", command: "t", expect: "exit-zero" }]); // 1 == default → omitted +}); + +test("flake guard: the `check:` sugar also accepts `N times`", () => { + const loop = parse('loop "x":\n goal: g\n check: "pnpm test" passes 3 times').definitions[0]; + assert.deepEqual(loop.doneWhen, [{ type: "command", command: "pnpm test", expect: "exit-zero", runs: 3 }]); +}); + test("memory: 'keep a memory in' is an accepted alias", () => { const loop = parse('loop "x":\n goal: g\n keep a memory in "notes.md"').definitions[0]; assert.deepEqual(loop.memory, { file: "notes.md" }); diff --git a/packages/runtime/src/show.ts b/packages/runtime/src/show.ts index c2da003..c09262d 100644 --- a/packages/runtime/src/show.ts +++ b/packages/runtime/src/show.ts @@ -8,8 +8,10 @@ import type { LoopFile, Definition, Loop, Pipeline, Flow, Predicate, Transition, function predicateStr(p?: Predicate | null): string | null { if (!p) return null; - if (p.type === "test") return `test "${p.target}"`; - if (p.type === "command") return p.expect === "empty" ? `"${p.command}" finds nothing` : `"${p.command}" passes`; + // `… passes N times` re-runs the check as a flake guard; surface it compactly as `×N`. + const times = (p.type === "test" || p.type === "command") && p.runs && p.runs > 1 ? ` ×${p.runs}` : ""; + if (p.type === "test") return `test "${p.target}"${times}`; + if (p.type === "command") return (p.expect === "empty" ? `"${p.command}" finds nothing` : `"${p.command}" passes`) + times; if (p.type === "human") return `a human confirms "${p.description}"`; // skill = an eval: name the verdict and, when not the default, the subject it judges. const verdict = p.minScore !== undefined ? `scores ${p.minScore}+` : "approves"; @@ -137,8 +139,9 @@ function joinList(items: string[], conj = "then"): string { } function predicateProse(p: Predicate): string { - if (p.type === "test") return `the test "${p.target}" passes`; - if (p.type === "command") return p.expect === "empty" ? `running \`${p.command}\` reports nothing` : `running \`${p.command}\` succeeds`; + const times = (p.type === "test" || p.type === "command") && p.runs && p.runs > 1 ? ` ${p.runs} times in a row` : ""; + if (p.type === "test") return `the test "${p.target}" passes${times}`; + if (p.type === "command") return (p.expect === "empty" ? `running \`${p.command}\` reports nothing` : `running \`${p.command}\` succeeds`) + times; if (p.type === "human") return `you confirm "${p.description}"`; // skill = an eval const verdict = p.minScore !== undefined ? `the "${p.skill}" review scores ${p.minScore} or more` : `the "${p.skill}" review approves`; diff --git a/packages/runtime/src/verify.ts b/packages/runtime/src/verify.ts index 67dd531..6ab3176 100644 --- a/packages/runtime/src/verify.ts +++ b/packages/runtime/src/verify.ts @@ -63,8 +63,20 @@ export class ShellVerifier implements Verifier { expectEmpty = predicate.expect === "empty"; } - const { code, out } = await sh(command, baseDir); - const passed = expectEmpty ? code === 0 && out.length === 0 : code === 0; - return { passed, output: out.slice(0, 4000) }; + // `… passes N times` re-runs the check as a flake guard: EVERY run must pass, and the first + // failing run short-circuits (no point running the rest). Absent / 1 → a single run as before. + const runs = Math.max(1, predicate.runs ?? 1); + let lastOut = ""; + for (let i = 1; i <= runs; i++) { + const { code, out } = await sh(command, baseDir); + lastOut = out; + const ok = expectEmpty ? code === 0 && out.length === 0 : code === 0; + if (!ok) { + const prefix = runs > 1 ? `run ${i}/${runs} failed — ` : ""; + return { passed: false, output: (prefix + out).slice(0, 4000) }; + } + } + const suffix = runs > 1 ? `\n(passed ${runs}/${runs} runs)` : ""; + return { passed: true, output: (lastOut + suffix).slice(0, 4000) }; } } diff --git a/packages/runtime/test/verify.test.js b/packages/runtime/test/verify.test.js new file mode 100644 index 0000000..dc7ca65 --- /dev/null +++ b/packages/runtime/test/verify.test.js @@ -0,0 +1,62 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync, rmSync, existsSync } from "node:fs"; +import { join } from "node:path"; +import { tmpdir } from "node:os"; +import { ShellVerifier } from "../dist/verify.js"; + +const verifier = new ShellVerifier(); + +// A shell snippet that appends one line to `f` (so the file's line count == how many times the +// command actually ran), optionally then exiting non-zero to simulate a failing run. +const bump = (f, ok = true) => `echo x >> ${JSON.stringify(f)}${ok ? "" : "; exit 1"}`; +const runCount = (f) => (existsSync(f) ? readFileSync(f, "utf8").trim().split("\n").filter(Boolean).length : 0); + +test("a single-run command still runs exactly once and passes on exit 0", async () => { + const f = join(tmpdir(), `verify-once-${process.pid}-${Date.now()}`); + try { + const r = await verifier.verify({ type: "command", command: bump(f), expect: "exit-zero" }, tmpdir()); + assert.equal(r.passed, true); + assert.equal(runCount(f), 1); + assert.ok(!/passed \d+\/\d+ runs/.test(r.output), "no flake-guard suffix for a single run"); + } finally { + rmSync(f, { force: true }); + } +}); + +test("`passes N times` runs the command N times when every run passes", async () => { + const f = join(tmpdir(), `verify-ntimes-${process.pid}-${Date.now()}`); + try { + const r = await verifier.verify({ type: "command", command: bump(f), expect: "exit-zero", runs: 4 }, tmpdir()); + assert.equal(r.passed, true); + assert.equal(runCount(f), 4, "ran exactly 4 times"); + assert.match(r.output, /passed 4\/4 runs/); + } finally { + rmSync(f, { force: true }); + } +}); + +test("`passes N times` short-circuits on the first failing run", async () => { + const f = join(tmpdir(), `verify-shortcircuit-${process.pid}-${Date.now()}`); + try { + // Always appends, then exits 1 → the first run fails, so the remaining runs must NOT execute. + const r = await verifier.verify({ type: "command", command: bump(f, false), expect: "exit-zero", runs: 3 }, tmpdir()); + assert.equal(r.passed, false); + assert.equal(runCount(f), 1, "stopped after the first failure — did not run 3 times"); + assert.match(r.output, /run 1\/3 failed/); + } finally { + rmSync(f, { force: true }); + } +}); + +test("`finds nothing N times` requires empty output on every run", async () => { + // `true` produces no output and exits 0 → empty, so N clean runs pass. + const clean = await verifier.verify({ type: "command", command: "true", expect: "empty", runs: 3 }, tmpdir()); + assert.equal(clean.passed, true); + assert.match(clean.output, /passed 3\/3 runs/); + + // A command that prints has non-empty output → fails on the first run. + const noisy = await verifier.verify({ type: "command", command: "echo found-a-match", expect: "empty", runs: 3 }, tmpdir()); + assert.equal(noisy.passed, false); + assert.match(noisy.output, /run 1\/3 failed/); +}); From fcc6a7ba4863af1d9d9b29f50bda4a9eed307966 Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Thu, 2 Jul 2026 00:06:27 -0400 Subject: [PATCH 08/14] =?UTF-8?q?feat(parser,runtime):=20verification=20re?= =?UTF-8?q?liability=20+=20crash=20recovery=20=E2=80=94=20judge=20panels,?= =?UTF-8?q?=20secret=20redaction,=20--resume?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three features that turn a demo loop into one you can trust unattended: • Judge panels — `done when the skill "x" approves by N judges`. N independent verdicts, majority wins, early-exit once mathematically decided. Each vote is its own skill-verify event (`judge i/N: …`); observe reports the tally. The eval-side counterpart of `passes N times`: flake guard for tests, judge panel for evals. Composes with scores/subject/the bar. • Secret redaction — every event is scrubbed before any sink persists it: values of secret-named env vars (*_TOKEN/_SECRET/_PASSWORD/_KEY/…) become [redacted:], and well-known shapes (GitHub/Slack/AWS/sk- keys, JWTs, PEM blocks, Bearer headers, password= assignments) are masked. On by default; LOOP_REDACT=off to disable. Best-effort: a redactor fault falls back to the raw event rather than breaking telemetry. • Resume — `loop run file.loop --resume run.log`. The NDJSON log doubles as a journal: buildResumePlan() replays it (container stack + step depth keep nested sub-runs honest) and the engine skips every unit whose end event says satisfied — whole definitions, pipeline stages, flow steps, foreach items — emitting ⏩ resumed events. Flow handoff summaries now ride on flow-step-end so a resumed flow restores carry-forward context; the log header carries a sha256 of the .loop source and the CLI warns on drift. Interrupted or failed units re-run from scratch: nothing is trusted that a check didn't prove. Tests: parser 61/61 (judge grammar), runtime 132/132 — 6 redaction, 4 panel (majority/early-exit/single-judge parity), 7 resume (log-scan unit tests + end-to-end crash-resume for pipeline, flow w/ summary restore, foreach). Co-Authored-By: Claude Fable 5 --- packages/parser/src/parser.ts | 18 ++- packages/parser/src/types.ts | 6 +- packages/parser/test/parser.test.js | 19 +++ packages/runtime/src/cli.ts | 29 ++++- packages/runtime/src/engine.ts | 79 +++++++++++- packages/runtime/src/eventSink.ts | 28 +++- packages/runtime/src/index.ts | 4 + packages/runtime/src/redact.ts | 65 ++++++++++ packages/runtime/src/resume.ts | 136 ++++++++++++++++++++ packages/runtime/src/show.ts | 6 +- packages/runtime/src/types.ts | 36 +++++- packages/runtime/test/judges.test.js | 82 ++++++++++++ packages/runtime/test/redact.test.js | 77 +++++++++++ packages/runtime/test/resume.test.js | 183 +++++++++++++++++++++++++++ packages/vscode/src/extension.ts | 1 + 15 files changed, 749 insertions(+), 20 deletions(-) create mode 100644 packages/runtime/src/redact.ts create mode 100644 packages/runtime/src/resume.ts create mode 100644 packages/runtime/test/judges.test.js create mode 100644 packages/runtime/test/redact.test.js create mode 100644 packages/runtime/test/resume.test.js diff --git a/packages/parser/src/parser.ts b/packages/parser/src/parser.ts index 23a83c3..c6ad404 100644 --- a/packages/parser/src/parser.ts +++ b/packages/parser/src/parser.ts @@ -151,14 +151,20 @@ function parsePredicate(s: string, lineNo: number): Predicate { // An eval predicate may name its subject: `on the output` (default) or `on the trajectory`. const subj = (s: string | undefined): { subject?: "output" | "trajectory" } => s ? { subject: s.toLowerCase() as "output" | "trajectory" } : {}; + // …and a consensus panel: `by N judges` runs the eval N times, majority wins. Only surfaced + // when > 1 (a single judge is the default shape). + const judges = (n: string | undefined): { judges?: number } => { + const j = n ? parseInt(n, 10) : 1; + return j > 1 ? { judges: j } : {}; + }; - // the skill "X" scores N or more [on the output|trajectory] -> eval with a numeric threshold - m = text.match(/^the skill\s+"([^"]+)"\s+scores\s+(\d+)(?:\s+or more)?(?:\s+on the (output|trajectory))?$/i); - if (m) return { type: "skill", skill: m[1], expect: "approve", minScore: parseInt(m[2], 10), ...subj(m[3]) }; + // the skill "X" scores N or more [on the output|trajectory] [by N judges] + m = text.match(/^the skill\s+"([^"]+)"\s+scores\s+(\d+)(?:\s+or more)?(?:\s+on the (output|trajectory))?(?:\s+by\s+(\d+)\s+judges?)?$/i); + if (m) return { type: "skill", skill: m[1], expect: "approve", minScore: parseInt(m[2], 10), ...subj(m[3]), ...judges(m[4]) }; - // the skill "X" approves [on the output|trajectory] -> eval (approved / not) - m = text.match(/^the skill\s+"([^"]+)"\s+approves(?:\s+on the (output|trajectory))?$/i); - if (m) return { type: "skill", skill: m[1], expect: "approve", ...subj(m[2]) }; + // the skill "X" approves [on the output|trajectory] [by N judges] + m = text.match(/^the skill\s+"([^"]+)"\s+approves(?:\s+on the (output|trajectory))?(?:\s+by\s+(\d+)\s+judges?)?$/i); + if (m) return { type: "skill", skill: m[1], expect: "approve", ...subj(m[2]), ...judges(m[3]) }; throw new ParseError(`could not understand "done when ${text}"`, lineNo); } diff --git a/packages/parser/src/types.ts b/packages/parser/src/types.ts index 7559af8..4d1bba3 100644 --- a/packages/parser/src/types.ts +++ b/packages/parser/src/types.ts @@ -243,9 +243,11 @@ export type Predicate = * An eval: a review skill judges the goal (approve / score). `subject` selects what it * inspects — the produced `output` (default) or the `trajectory` (the path and tool calls * the agent took to get there). `bar` is an optional inline rubric (`the bar:`) naming the - * conditions the judge scores against. + * conditions the judge scores against. `judges` (from `… by N judges`) runs the eval N times + * independently and takes a majority vote — consensus smooths single-judge wobble. Absent / + * 1 = the usual single verdict. */ - | { type: "skill"; skill: string; expect: "approve"; minScore?: number; subject?: "output" | "trajectory"; bar?: string }; + | { type: "skill"; skill: string; expect: "approve"; minScore?: number; subject?: "output" | "trajectory"; bar?: string; judges?: number }; export interface Transition { on: "pass" | "fail" | "blocked" | "attempts"; diff --git a/packages/parser/test/parser.test.js b/packages/parser/test/parser.test.js index fa40c15..8b658a6 100644 --- a/packages/parser/test/parser.test.js +++ b/packages/parser/test/parser.test.js @@ -239,6 +239,25 @@ test("flake guard: the `check:` sugar also accepts `N times`", () => { assert.deepEqual(loop.doneWhen, [{ type: "command", command: "pnpm test", expect: "exit-zero", runs: 3 }]); }); +test("multi-judge: `approves by N judges` carries a judges count", () => { + const loop = parse('loop "x":\n goal: g\n done when the skill "review" approves by 3 judges').definitions[0]; + assert.deepEqual(loop.doneWhen, [{ type: "skill", skill: "review", expect: "approve", judges: 3 }]); +}); + +test("multi-judge: composes with a score threshold and a subject", () => { + const loop = parse('loop "x":\n goal: g\n done when the skill "review" scores 8 or more on the trajectory by 5 judges').definitions[0]; + assert.deepEqual(loop.doneWhen, [ + { type: "skill", skill: "review", expect: "approve", minScore: 8, subject: "trajectory", judges: 5 }, + ]); +}); + +test("multi-judge: a single judge (default or `by 1 judge`) keeps the single-verdict shape", () => { + const plain = parse('loop "x":\n goal: g\n done when the skill "review" approves').definitions[0]; + assert.deepEqual(plain.doneWhen, [{ type: "skill", skill: "review", expect: "approve" }]); // no judges key + const one = parse('loop "x":\n goal: g\n done when the skill "review" approves by 1 judge').definitions[0]; + assert.deepEqual(one.doneWhen, [{ type: "skill", skill: "review", expect: "approve" }]); +}); + test("memory: 'keep a memory in' is an accepted alias", () => { const loop = parse('loop "x":\n goal: g\n keep a memory in "notes.md"').definitions[0]; assert.deepEqual(loop.memory, { file: "notes.md" }); diff --git a/packages/runtime/src/cli.ts b/packages/runtime/src/cli.ts index fc1faf0..005d18a 100644 --- a/packages/runtime/src/cli.ts +++ b/packages/runtime/src/cli.ts @@ -1,5 +1,6 @@ #!/usr/bin/env node import { appendFileSync, readFileSync, writeFileSync, readdirSync, existsSync } from "node:fs"; +import { createHash } from "node:crypto"; import { dirname, resolve, relative, basename } from "node:path"; import { createInterface } from "node:readline"; import { parse } from "@loop-lang/parser"; @@ -13,6 +14,7 @@ import { IpcHumanIO } from "./ipc.js"; import { ShellGitIO } from "./runners/shellGit.js"; import { ownModelBinaryWarning } from "./ownModel.js"; import { eventSinkFromEnv } from "./eventSink.js"; +import { buildResumePlan } from "./resume.js"; import type { LoopEvent } from "./types.js"; import { summarizeModels, formatModelSummary, summarizeOpex, formatOpexSummary } from "./summary.js"; @@ -37,6 +39,8 @@ function render(e: LoopEvent): string { return ` ■ stage "${e.name}"`; case "stage-end": return ` ■ stage "${e.name}" → ${e.satisfied ? "satisfied" : "FAILED"}`; + case "resumed": + return ` ⏩ ${e.unit} ${e.name ? `"${e.name}"` : ""}${e.index !== undefined ? ` #${e.index + 1}` : ""} — resumed (already satisfied)`; case "loop-start": return `↻ loop ${e.name ? `"${e.name}"` : ""}`.trimEnd(); case "node-enter": @@ -231,7 +235,7 @@ async function main() { } if (!cmd || (cmd !== "run" && cmd !== "parse" && cmd !== "viz" && cmd !== "live") || !fileArg) { - console.error("usage: loop-run [--model ] [--live] [--events] [--log ] [--out ]"); + console.error("usage: loop-run [--model ] [--live] [--events] [--log ] [--resume ] [--out ]"); process.exit(2); } @@ -294,6 +298,26 @@ async function main() { const logFile = logIdx >= 0 ? rest[logIdx + 1] : undefined; const target = file.config?.target ? resolve(baseDir, file.config.target) : baseDir; + // `--resume `: rebuild what a prior run already finished from its event log and skip it — + // satisfied definitions / stages / flow steps / foreach items don't re-run; the first + // incomplete unit picks up (with flow carry-forward summaries restored from the log). + const resumeIdx = rest.indexOf("--resume"); + const resumeLog = resumeIdx >= 0 ? rest[resumeIdx + 1] : undefined; + const sourceHash = createHash("sha256").update(readFileSync(path)).digest("hex"); + let resume: import("./types.js").ResumePlan | undefined; + if (resumeLog) { + resume = buildResumePlan(readFileSync(resolve(baseDir, resumeLog), "utf8")); + if (resume.sourceHash && resume.sourceHash !== sourceHash) { + console.error(`⚠ resume: ${basename(path)} changed since the logged run — completed units are matched by name/position and may not correspond`); + } + if (resume.completed.size === 0) { + console.error(`↻ resume: nothing recorded as satisfied in ${resumeLog} — running from the start`); + resume = undefined; + } else { + console.error(`↻ resume: skipping ${resume.completed.size} completed unit(s) from ${resumeLog}`); + } + } + const git = new ShellGitIO(); // Attach the ctx skill recommender only when the file opts in (`recommend skills with ctx` @@ -320,6 +344,7 @@ async function main() { const eventSink = eventSinkFromEnv( { loop_path: path, + loop_sha256: sourceHash, principal: file.config?.runsAs, runner: file.config?.runner, }, @@ -355,6 +380,7 @@ async function main() { flowStack: [path], modelPolicy: file.config?.models, cliModel: model, + resume, onEvent: (e) => { eventSink?.post(e); emit({ kind: "event", event: e }); }, }); const ok = outcomes.every((o) => o.satisfied); @@ -381,6 +407,7 @@ async function main() { flowStack: [path], modelPolicy: file.config?.models, cliModel: model, + resume, }; const traceEvents: LoopEvent[] = []; const onTrace = (e: LoopEvent) => { diff --git a/packages/runtime/src/engine.ts b/packages/runtime/src/engine.ts index 128adcb..b9e5c69 100644 --- a/packages/runtime/src/engine.ts +++ b/packages/runtime/src/engine.ts @@ -1,6 +1,7 @@ import { resolve, dirname, basename } from "node:path"; import type { Loop, Pipeline, Stage, Flow, FlowStep, Definition, Transition, Action, LoopFile, HookPoint } from "@loop-lang/parser"; import type { CycleNode, LoopEvent, LoopOutcome, RunOptions, StopReason } from "./types.js"; +import { scopeResume } from "./resume.js"; import { enumerateItems, labelOf } from "./iterate.js"; import { resolveGit, isProtected } from "./git.js"; import { resolveModels, modelForPhase } from "./models.js"; @@ -280,7 +281,7 @@ async function executeLoop(loop: Loop, opts: RunOptions): Promise { // A trajectory eval judges HOW the agent got there (the captured path); // an output eval judges WHAT it produced (the act summary). const isTrajectory = pred.subject === "trajectory"; - const sr = await opts.runner.runSkill({ + const skillInput = { skill: pred.skill, goal: loop.goal, context: isTrajectory ? (lastTrajectory || lastActSummary) : lastActSummary, @@ -289,9 +290,28 @@ async function executeLoop(loop: Loop, opts: RunOptions): Promise { subject: pred.subject, bar: pred.bar, baseDir: opts.baseDir, - }); - emit(opts, { type: "skill-verify", skill: pred.skill, passed: sr.passed, detail: sr.detail }); - r = { passed: sr.passed, output: sr.detail }; + }; + // `by N judges` — a consensus panel: N independent verdicts, majority wins. A single + // LM judge wobbles near the bar; independent samples average the noise out. Stops + // early once the vote is mathematically decided. + const panel = Math.max(1, pred.judges ?? 1); + let approvals = 0; + const verdicts: string[] = []; + for (let j = 1; j <= panel; j++) { + const sr = await opts.runner.runSkill(skillInput); + if (sr.passed) approvals++; + const detail = panel > 1 ? `judge ${j}/${panel}: ${sr.detail}` : sr.detail; + verdicts.push(detail); + emit(opts, { type: "skill-verify", skill: pred.skill, passed: sr.passed, detail }); + if (approvals * 2 > panel || (j - approvals) * 2 > panel) break; // majority decided + } + const approved = approvals * 2 > panel; + r = { + passed: approved, + output: panel > 1 + ? `judges: ${approvals}/${verdicts.length} approved (majority of ${panel} ${approved ? "reached" : "not reached"})\n${verdicts.join("\n")}` + : verdicts[0], + }; } else { r = await opts.verifier.verify(pred, opts.baseDir); } @@ -438,6 +458,14 @@ async function applyActions( /** Run one stage: its gate, its loop, and a story-commit on success. */ async function runStage(stage: Stage, opts: RunOptions): Promise { + // Resume: a stage the prior run's log proves satisfied is skipped whole — gate included + // (it was already approved once; re-asking would gate work that won't happen). + if (opts.resumeScope?.stages.has(stage.name)) { + emit(opts, { type: "stage-start", name: stage.name }); + emit(opts, { type: "resumed", unit: "stage", name: stage.name }); + emit(opts, { type: "stage-end", name: stage.name, satisfied: true }); + return { satisfied: true, reason: "done", attempts: 0, summary: `[${stage.name}] resumed — already satisfied` }; + } emit(opts, { type: "stage-start", name: stage.name }); if (stage.gate) { emit(opts, { type: "human", kind: "gate", prompt: stage.gate.message }); @@ -508,10 +536,19 @@ async function executeForEach(step: FlowStep, opts: RunOptions, stack: string[]) let failedAccepted = 0; for (let i = 0; i < items.length; i++) { + // Resume: skip items the prior run already delivered (keyed by step + var + index). + if (opts.resumeScope?.items.get(`${step.name}:${varName}`)?.has(i)) { + emit(opts, { type: "foreach-item-start", var: varName, index: i, total: items.length }); + emit(opts, { type: "resumed", unit: "foreach-item", name: varName, index: i }); + emit(opts, { type: "foreach-item-end", var: varName, index: i, satisfied: true }); + continue; + } emit(opts, { type: "foreach-item-start", var: varName, index: i, total: items.length }); const tmpl = await opts.loadFile(step.ref, opts.baseDir); const outcomes = await run(tmpl, { ...opts, + resume: undefined, + resumeScope: undefined, // a template's own units are summarised by this item's end event baseDir: dirname(templatePath), flowStack: [...stack, templatePath], upstream: items[i], @@ -543,6 +580,18 @@ async function executeFlow(flow: Flow, opts: RunOptions): Promise { let carried = opts.upstream; // upstream from a parent flow, if any for (const step of flow.steps) { + // Resume: a step the prior run's log proves satisfied is skipped, and its recorded handoff + // summary is restored so the NEXT step still receives the carry-forward context it expects. + const resumed = opts.resumeScope?.steps.get(step.name); + if (resumed !== undefined) { + const summary = typeof resumed === "string" ? resumed : `[${step.name}] satisfied (resumed)`; + summaries[step.name] = summary; + carried = summary; + emit(opts, { type: "flow-step-start", name: step.name, ref: step.ref }); + emit(opts, { type: "resumed", unit: "flow-step", name: step.name }); + emit(opts, { type: "flow-step-end", name: step.name, satisfied: true, summary }); + continue; + } emit(opts, { type: "flow-step-start", name: step.name, ref: step.ref }); if (step.gate) { @@ -576,6 +625,8 @@ async function executeFlow(flow: Flow, opts: RunOptions): Promise { const stepUpstream = step.fromStep ? summaries[step.fromStep] : carried; const outcomes = await run(subFile, { ...opts, + resume: undefined, + resumeScope: undefined, // a sub-file's units are summarised by this step's end event baseDir: dirname(stepPath), flowStack: [...stack, stepPath], upstream: stepUpstream, @@ -590,7 +641,8 @@ async function executeFlow(flow: Flow, opts: RunOptions): Promise { summaries[step.name] = summary; carried = summary; - emit(opts, { type: "flow-step-end", name: step.name, satisfied }); + // The summary rides on the end event so a `--resume` of this log can restore the carry-forward. + emit(opts, { type: "flow-step-end", name: step.name, satisfied, summary: summary.slice(0, 4000) }); if (!satisfied) { emit(opts, { type: "flow-end", name: flow.name, satisfied: false }); return { satisfied: false, reason: "blocked", attempts: 0, summary }; @@ -640,6 +692,19 @@ export async function runDefinition(def: Definition, opts: RunOptions): Promise< return executeLoopFull(def, opts); } +/** + * Run the i-th top-level definition under the resume plan: skip it whole when the prior log + * proves it satisfied, otherwise run it with its slice of the plan (stages/steps/items). + */ +async function runDefinitionResumable(def: Definition, i: number, opts: RunOptions): Promise { + if (opts.resume?.completed.has(`def:${i}`)) { + const name = ("name" in def ? (def as { name: string | null }).name : null) ?? def.kind; + emit(opts, { type: "resumed", unit: "definition", name }); + return { satisfied: true, reason: "done", attempts: 0, summary: `[${name}] resumed — already satisfied in a prior run` }; + } + return runDefinition(def, { ...opts, resumeScope: scopeResume(opts.resume, i) }); +} + /** Run every definition in a parsed file, in order. */ export async function run(file: LoopFile, opts: RunOptions): Promise { const outer = !opts.gitStarted && !!opts.git; @@ -663,7 +728,7 @@ export async function run(file: LoopFile, opts: RunOptions): Promise x.satisfied); if (allOk && policy.commit === "done") await gitCommit(o, `loop: ${slug(file.definitions[0] && (file.definitions[0] as any).name)} — goal met`); if (allOk && policy.push) { await o.git!.push({ branch, dir: o.baseDir }); emit(o, { type: "git", action: "push", detail: branch }); } @@ -672,6 +737,6 @@ export async function run(file: LoopFile, opts: RunOptions): Promise string): EventSink { + return { + post(event) { + let scrubbed = event; + try { + scrubbed = redactEvent(event, redact); + } catch { + /* a redactor bug must never break telemetry — fall back to the raw event */ + } + sink.post(scrubbed); + }, + flush: () => sink.flush(), + }; +} + /** Fan one event stream out to several sinks. Returns undefined when none are active. */ export function combineSinks(sinks: Array): EventSink | undefined { const active = sinks.filter((s): s is EventSink => !!s); @@ -122,13 +144,17 @@ export function combineSinks(sinks: Array): EventSink | u * LOOP_RUN_ID — correlate events to one run (optional; a UUID is generated if unset) * * `override.logFile` (the CLI's `--log `) takes precedence over `LOOP_LOG_FILE`. + * + * Secret redaction is ON by default (see redact.ts); set `LOOP_REDACT=off` to disable. */ export function eventSinkFromEnv(meta?: RunMeta, override?: { logFile?: string }): EventSink | undefined { const runId = process.env.LOOP_RUN_ID || randomUUID(); const url = process.env.LOOP_EVENTS_URL; const logPath = override?.logFile || process.env.LOOP_LOG_FILE; - return combineSinks([ + const sink = combineSinks([ url ? makeHttpEventSink({ url, runId, token: process.env.LOOP_EVENTS_TOKEN, meta }) : undefined, logPath ? makeFileEventSink({ path: logPath, runId, meta }) : undefined, ]); + if (!sink || process.env.LOOP_REDACT === "off") return sink; + return redactingSink(sink, buildRedactor()); } diff --git a/packages/runtime/src/index.ts b/packages/runtime/src/index.ts index a4c3216..c2ff4e8 100644 --- a/packages/runtime/src/index.ts +++ b/packages/runtime/src/index.ts @@ -1,6 +1,10 @@ export * from "./types.js"; export { run, runDefinition } from "./engine.js"; export { ownModelBinaryWarning, commandOnPath } from "./ownModel.js"; +export { buildResumePlan, scopeResume } from "./resume.js"; +export { buildRedactor, redactEvent } from "./redact.js"; +export { eventSinkFromEnv, makeFileEventSink, makeHttpEventSink, combineSinks, redactingSink } from "./eventSink.js"; +export type { EventSink, RunMeta } from "./eventSink.js"; export { ShellVerifier } from "./verify.js"; export type { ShellVerifierOptions } from "./verify.js"; export { CliHumanIO, ScriptedHumanIO } from "./human.js"; diff --git a/packages/runtime/src/redact.ts b/packages/runtime/src/redact.ts new file mode 100644 index 0000000..f066279 --- /dev/null +++ b/packages/runtime/src/redact.ts @@ -0,0 +1,65 @@ +import type { LoopEvent } from "./types.js"; + +/** + * Secret redaction for telemetry. Loop events carry raw command output (`observe`, `node-exit`, + * hook details …), and in CI that output can echo tokens from the environment. Everything a sink + * persists — the local NDJSON log and the HTTP collector — passes through here first, so a leaked + * credential is scrubbed *before* it ever touches disk or the network. + * + * Two layers, both best-effort: + * 1. **Environment values** — the value of any env var whose NAME looks secret-bearing + * (TOKEN / SECRET / PASSWORD / *_KEY / CREDENTIALS / AUTH) is replaced wherever it appears, + * labelled with the variable name so the trace stays debuggable. + * 2. **Well-known shapes** — GitHub / Slack / AWS / generic `sk-` API keys, JWTs, PEM private + * keys, `Bearer …` headers, and `password=…`-style assignments. + * + * On by default; `LOOP_REDACT=off` disables (e.g. for local debugging of the redactor itself). + */ + +const SECRET_ENV_NAME = /(TOKEN|SECRET|PASSWORD|PASSWD|API_?KEY|ACCESS_KEY|PRIVATE_KEY|CREDENTIALS?|AUTH)/i; + +/** Value-shape patterns for common credentials. Order matters: multi-line PEM first. */ +const PATTERNS: Array<[RegExp, string]> = [ + [/-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g, "[redacted:private-key]"], + [/\bgh[pousr]_[A-Za-z0-9]{20,}\b/g, "[redacted:github-token]"], + [/\bgithub_pat_[A-Za-z0-9_]{20,}\b/g, "[redacted:github-token]"], + [/\bxox[baprs]-[A-Za-z0-9-]{10,}\b/g, "[redacted:slack-token]"], + [/\bAKIA[0-9A-Z]{16}\b/g, "[redacted:aws-key-id]"], + [/\bsk-[A-Za-z0-9_-]{20,}\b/g, "[redacted:api-key]"], + [/\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b/g, "[redacted:jwt]"], + [/\b(Bearer|Basic)\s+[A-Za-z0-9._~+/=-]{16,}/g, "$1 [redacted]"], + // `password=hunter2!`, `api_key: "…"` — keep the key, scrub the value. The lookahead skips + // values already handled above (`Bearer [redacted]`, an env-layer `[redacted:X]`). + [/\b(password|passwd|secret|token|api[_-]?key|authorization)(\s*[=:]\s*)(?!\[redacted|Bearer\b|Basic\b)("[^"]{4,}"|'[^']{4,}'|[^\s"']{4,})/gi, "$1$2[redacted]"], +]; + +/** + * Build a string redactor from an environment map. Values shorter than 8 chars are ignored — + * scrubbing something like "true" or "dev" would shred ordinary output. + */ +export function buildRedactor(env: Record = process.env): (s: string) => string { + const values = Object.entries(env) + .filter((e): e is [string, string] => !!e[1] && e[1].length >= 8 && SECRET_ENV_NAME.test(e[0])) + .sort((a, b) => b[1].length - a[1].length); // longest first so substrings can't pre-empt + return (s: string): string => { + let out = s; + for (const [name, value] of values) out = out.split(value).join(`[redacted:${name}]`); + for (const [re, rep] of PATTERNS) out = out.replace(re, rep); + return out; + }; +} + +/** Deep-walk an event and redact every string field (arrays and nested objects included). */ +export function redactEvent(e: LoopEvent, redact: (s: string) => string): LoopEvent { + const walk = (v: unknown): unknown => { + if (typeof v === "string") return redact(v); + if (Array.isArray(v)) return v.map(walk); + if (v && typeof v === "object") { + const out: Record = {}; + for (const [k, val] of Object.entries(v)) out[k] = walk(val); + return out; + } + return v; + }; + return walk(e) as LoopEvent; +} diff --git a/packages/runtime/src/resume.ts b/packages/runtime/src/resume.ts new file mode 100644 index 0000000..0b03e0b --- /dev/null +++ b/packages/runtime/src/resume.ts @@ -0,0 +1,136 @@ +import type { LoopEvent, ResumePlan, ResumeScope } from "./types.js"; + +/** + * Resume-from-log: rebuild "what already finished" from a prior run's NDJSON event log + * (`--log run.log` → crash/Ctrl-C → `--resume run.log`). + * + * The log is a flat stream, but nested work (a flow step running a whole sub-file, a foreach + * item running a template) emits its own loop/pipeline/flow events inside it. Two counters keep + * the scan honest: + * + * - a **container stack** of open loop/pipeline/flow definitions — an `*-end` that empties the + * stack closes a *top-level* definition (stage loops and sub-file runs pop back to a non-empty + * stack, so they never miscount); + * - a **step depth** of open flow-steps / foreach-items — units nested inside another step are + * someone else's business (their parent's end event already summarises them). + * + * Only units that ended `satisfied: true` are recorded. An interrupted unit has no end event, a + * failed one has `satisfied: false` — both re-run on resume, which is exactly the semantic you + * want: "skip what's proven done, redo what isn't". + * + * A log may contain several headers (a resumed run appending to the same file) — keys simply + * accumulate; the latest header's hash wins. + */ +export function buildResumePlan(logText: string): ResumePlan { + const completed = new Set(); + const summaries = new Map(); + let sourceHash: string | undefined; + let runId: string | undefined; + + const stack: string[] = []; // open definition containers (loop / pipeline / flow) + let defOrdinal = 0; // index of the next top-level definition to close + let stepDepth = 0; // open flow-steps + foreach-items + let curStep: string | null = null; // the open top-level flow step (for foreach item keys) + + for (const line of logText.split("\n")) { + if (!line.trim()) continue; + let o: { v?: string; runId?: string; meta?: { loop_sha256?: string }; event?: LoopEvent }; + try { + o = JSON.parse(line); + } catch { + continue; // a torn tail line (crash mid-write) is expected — ignore + } + if (o.v === "loop.log.v1") { + sourceHash = o.meta?.loop_sha256 ?? sourceHash; + runId = o.runId ?? runId; + continue; + } + const e = o.event; + if (!e || typeof (e as { type?: unknown }).type !== "string") continue; + + switch (e.type) { + case "loop-start": + case "pipeline-start": + case "flow-start": + stack.push(e.type); + break; + + case "loop-end": + case "pipeline-end": + case "flow-end": + stack.pop(); + if (stack.length === 0 && stepDepth === 0) { + if (e.satisfied) completed.add(`def:${defOrdinal}`); + defOrdinal++; + } + break; + + case "stage-end": + // A top-level pipeline's stage: exactly the pipeline itself on the stack. + if (stack.length === 1 && stepDepth === 0 && e.satisfied) { + completed.add(`stage:${defOrdinal}:${e.name}`); + } + break; + + case "flow-step-start": + if (stepDepth === 0 && stack.length === 1) curStep = e.name; + stepDepth++; + break; + + case "flow-step-end": + stepDepth--; + if (stepDepth === 0 && stack.length === 1) { + if (e.satisfied) { + const key = `step:${defOrdinal}:${e.name}`; + completed.add(key); + if (e.summary) summaries.set(key, e.summary); + } + curStep = null; + } + break; + + case "foreach-item-start": + stepDepth++; + break; + + case "foreach-item-end": + stepDepth--; + // Items of a top-level flow's step sit one step deep (inside the open flow-step). + if (stepDepth === 1 && stack.length === 1 && curStep && e.satisfied) { + completed.add(`item:${defOrdinal}:${curStep}:${e.var}:${e.index}`); + } + break; + } + } + return { completed, summaries, sourceHash, runId }; +} + +/** Slice a plan down to one definition's units — what the executors actually consult. */ +export function scopeResume(plan: ResumePlan | undefined, defIndex: number): ResumeScope | undefined { + if (!plan) return undefined; + const stages = new Set(); + const steps = new Map(); + const items = new Map>(); + const stagePrefix = `stage:${defIndex}:`; + const stepPrefix = `step:${defIndex}:`; + const itemPrefix = `item:${defIndex}:`; + for (const key of plan.completed) { + if (key.startsWith(stagePrefix)) stages.add(key.slice(stagePrefix.length)); + else if (key.startsWith(stepPrefix)) { + const name = key.slice(stepPrefix.length); + steps.set(name, plan.summaries.get(key) ?? true); + } else if (key.startsWith(itemPrefix)) { + // item:::: — index is the final segment; step/var may themselves hold ':'-free names. + const rest = key.slice(itemPrefix.length); + const lastColon = rest.lastIndexOf(":"); + const idx = parseInt(rest.slice(lastColon + 1), 10); + const stepVar = rest.slice(0, lastColon); + if (!Number.isNaN(idx)) { + if (!items.has(stepVar)) items.set(stepVar, new Set()); + items.get(stepVar)!.add(idx); + } + } + } + if (!stages.size && !steps.size && !items.size) return undefined; + return { stages, steps, items }; +} diff --git a/packages/runtime/src/show.ts b/packages/runtime/src/show.ts index c09262d..3ae2970 100644 --- a/packages/runtime/src/show.ts +++ b/packages/runtime/src/show.ts @@ -16,7 +16,8 @@ function predicateStr(p?: Predicate | null): string | null { // skill = an eval: name the verdict and, when not the default, the subject it judges. const verdict = p.minScore !== undefined ? `scores ${p.minScore}+` : "approves"; const on = p.subject && p.subject !== "output" ? ` on the ${p.subject}` : ""; - return `eval: skill "${p.skill}" ${verdict}${on}`; + const panel = p.judges && p.judges > 1 ? ` · ${p.judges} judges` : ""; + return `eval: skill "${p.skill}" ${verdict}${on}${panel}`; } /** Render every `done when` predicate (a conjunction) as labelled strings. */ function predicateStrs(dw?: Predicate[] | null): string[] { @@ -145,7 +146,8 @@ function predicateProse(p: Predicate): string { if (p.type === "human") return `you confirm "${p.description}"`; // skill = an eval const verdict = p.minScore !== undefined ? `the "${p.skill}" review scores ${p.minScore} or more` : `the "${p.skill}" review approves`; - return verdict + (p.subject === "trajectory" ? " (judging how it got there, not just the result)" : ""); + const panel = p.judges && p.judges > 1 ? ` by a majority of ${p.judges} judges` : ""; + return verdict + panel + (p.subject === "trajectory" ? " (judging how it got there, not just the result)" : ""); } /** A plain-English description of a single loop. */ diff --git a/packages/runtime/src/types.ts b/packages/runtime/src/types.ts index 32ee8ff..8298fd8 100644 --- a/packages/runtime/src/types.ts +++ b/packages/runtime/src/types.ts @@ -23,8 +23,11 @@ export type LoopEvent = | { type: "loop-end"; name: string | null; satisfied: boolean } | { type: "flow-start"; name: string } | { type: "flow-step-start"; name: string; ref: string } - | { type: "flow-step-end"; name: string; satisfied: boolean } + /** `summary` (the step's handoff text) is recorded so a resumed run can restore the flow carry-forward. */ + | { type: "flow-step-end"; name: string; satisfied: boolean; summary?: string } | { type: "flow-end"; name: string; satisfied: boolean } + /** A unit skipped because a prior run's event log (`--resume`) shows it already satisfied. */ + | { type: "resumed"; unit: "definition" | "stage" | "flow-step" | "foreach-item"; name?: string; index?: number } | { type: "foreach-start"; var: string; source: string; count: number; labels?: string[] } | { type: "foreach-item-start"; var: string; index: number; total: number } | { type: "foreach-item-end"; var: string; index: number; satisfied: boolean } @@ -216,6 +219,37 @@ export interface RunOptions { ctxGrants?: string[]; /** A user-owned model (`ctx may use my own model …`) that unlocks ctx harness recommendations. */ ownModel?: { provider: string; model: string }; + /** + * Crash/interrupt recovery (`--resume `): the plan built from a prior run's event log. + * Only the top-level `run()` consults it (to skip whole completed definitions); per-definition + * units are handed down via `resumeScope`. Nested runs (flow steps, foreach templates) never + * see either — their completion is already summarised by their parent unit's end event. + */ + resume?: ResumePlan; + /** The current definition's slice of the resume plan (set by run(), consumed by the executors). */ + resumeScope?: ResumeScope; +} + +/** What a prior run's event log says is already satisfied. Built by `buildResumePlan`. */ +export interface ResumePlan { + /** Canonical keys of completed top-level units (`def:`, `stage::`, `step::`, `item::::`). */ + completed: Set; + /** flow-step key → its recorded handoff summary, so a resumed flow restores carry-forward context. */ + summaries: Map; + /** sha256 of the .loop source recorded in the log header — mismatch means the file changed since. */ + sourceHash?: string; + /** The prior run's id (for operator messages). */ + runId?: string; +} + +/** The per-definition slice of a ResumePlan (keys with the `def index` prefix stripped). */ +export interface ResumeScope { + /** Names of pipeline stages already satisfied. */ + stages: Set; + /** Flow step name → recorded summary (or true when the log predates summaries). */ + steps: Map; + /** `${stepName}:${var}` → indices of foreach items already satisfied. */ + items: Map>; } export interface LoopOutcome { diff --git a/packages/runtime/test/judges.test.js b/packages/runtime/test/judges.test.js new file mode 100644 index 0000000..84759ec --- /dev/null +++ b/packages/runtime/test/judges.test.js @@ -0,0 +1,82 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { parse } from "@loop-lang/parser"; +import { runDefinition, MockRunner, ScriptedHumanIO } from "../dist/index.js"; + +/** Loop whose only check is a 3-judge eval panel. */ +const SRC = [ + 'loop "panel":', + " goal: the design is sound", + ' done when the skill "design-review" approves by 3 judges', + " each cycle: act, then observe", + ' after 1 tries: stop and warn "give up"', +].join("\n"); + +function stubs(verdicts, events) { + let call = 0; + return { + runner: new MockRunner({ + skill: () => { + const passed = verdicts[Math.min(call, verdicts.length - 1)]; + call++; + return { passed, detail: passed ? "APPROVED" : "REJECTED" }; + }, + }), + verifier: { verify: async () => ({ passed: true, output: "unused" }) }, + human: new ScriptedHumanIO(), + baseDir: "/p", + hardCap: 2, + readText: async () => "", + writeText: async () => {}, + onEvent: (e) => events?.push(e), + }; +} + +test("majority approves (2/3) → satisfied; early-exit skips the 3rd judge", async () => { + const events = []; + const def = parse(SRC).definitions[0]; + const s = stubs([true, true, false], events); + const outcome = await runDefinition(def, s); + assert.equal(outcome.satisfied, true); + // 2 approvals reach majority of 3 → decided after judge 2; judge 3 never runs. + assert.equal(s.runner.skillCalls.length, 2, "early-exit once the vote is decided"); + const votes = events.filter((e) => e.type === "skill-verify"); + assert.equal(votes.length, 2); + assert.match(votes[0].detail, /^judge 1\/3: /); + const observe = events.find((e) => e.type === "observe"); + assert.match(observe.output, /judges: 2\/2 approved \(majority of 3 reached\)/); +}); + +test("majority rejects → not satisfied; each cycle's vote decided after 2 rejections", async () => { + const events = []; + const def = parse(SRC).definitions[0]; + const s = stubs([false], events); // every judge rejects, every cycle + const outcome = await runDefinition(def, s); + assert.equal(outcome.satisfied, false, "panel rejected → loop not done"); + // Majority-to-reject is decided after 2 of 3 rejections → exactly 2 votes per cycle. + const observes = events.filter((e) => e.type === "observe"); + const votes = events.filter((e) => e.type === "skill-verify"); + assert.equal(votes.length, observes.length * 2, "2 votes per cycle (early-exit on decided rejection)"); + assert.ok(observes.every((o) => o.passed === false)); + assert.match(observes[0].output, /judges: 0\/2 approved \(majority of 3 not reached\)/); +}); + +test("split then flip: 1 reject + 2 approves → majority approves", async () => { + const def = parse(SRC).definitions[0]; + const s = stubs([false, true, true]); + const outcome = await runDefinition(def, s); + assert.equal(outcome.satisfied, true, "2/3 approvals carry the vote even after a first rejection"); + assert.equal(s.runner.skillCalls.length, 3, "vote undecided until the 3rd judge"); +}); + +test("single judge (no `by N judges`) behaves exactly as before", async () => { + const src = SRC.replace(" by 3 judges", ""); + const def = parse(src).definitions[0]; + const events = []; + const s = stubs([true], events); + const outcome = await runDefinition(def, s); + assert.equal(outcome.satisfied, true); + assert.equal(s.runner.skillCalls.length, 1); + const vote = events.find((e) => e.type === "skill-verify"); + assert.ok(!/^judge \d/.test(vote.detail), "no judge prefix for a single verdict"); +}); diff --git a/packages/runtime/test/redact.test.js b/packages/runtime/test/redact.test.js new file mode 100644 index 0000000..ac90919 --- /dev/null +++ b/packages/runtime/test/redact.test.js @@ -0,0 +1,77 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { buildRedactor, redactEvent } from "../dist/redact.js"; +import { redactingSink } from "../dist/eventSink.js"; + +test("env layer: values of secret-named env vars are scrubbed and labelled", () => { + const redact = buildRedactor({ + GITHUB_TOKEN: "supersecret-value-123", + DB_PASSWORD: "hunter2hunter2", + PUBLIC_HOME: "public-info-here", // name doesn't look secret → untouched + SHORT_TOKEN: "abc", // too short → untouched (would shred output) + }); + const out = redact("push failed: auth supersecret-value-123 rejected; db=hunter2hunter2; abc public-info-here"); + assert.ok(!out.includes("supersecret-value-123")); + assert.ok(!out.includes("hunter2hunter2")); + assert.match(out, /\[redacted:GITHUB_TOKEN\]/); + assert.match(out, /\[redacted:DB_PASSWORD\]/); + assert.ok(out.includes("public-info-here"), "non-secret env value untouched"); + assert.ok(out.includes("abc"), "short value untouched"); +}); + +test("pattern layer: well-known credential shapes are scrubbed", () => { + const redact = buildRedactor({}); + // Fixtures are assembled at runtime so no literal token shape exists in this source file — + // otherwise GitHub push protection (correctly!) refuses to let the test suite be pushed. + const j = (...parts) => parts.join(""); + const cases = [ + [j("ghp_", "abcdefghijklmnopqrstuvwxyz0123456789"), "github-token"], + [j("github_pat_", "11ABCDEFG0123456789_abcdefghij"), "github-token"], + [j("xoxb-", "123456789012-abcdefghijklmnop"), "slack-token"], + [j("AKIA", "IOSFODNN7EXAMPLE"), "aws-key-id"], + [j("sk-", "abcdefghijklmnopqrstuvwxyz123456"), "api-key"], + [j("eyJhbGciOiJIUzI1NiJ9", ".", "eyJzdWIiOiIxMjM0In0", ".", "SflKxwRJSMeKKF2QT4fwpM"), "jwt"], + ]; + for (const [secret, label] of cases) { + const out = redact(`before ${secret} after`); + assert.ok(!out.includes(secret), `${label}: value scrubbed`); + assert.match(out, new RegExp(`\\[redacted:${label}\\]`)); + assert.ok(out.startsWith("before ") && out.endsWith(" after"), `${label}: surroundings intact`); + } +}); + +test("pattern layer: assignments and auth headers keep the key, lose the value", () => { + const redact = buildRedactor({}); + assert.equal(redact("password=hunter2!x"), "password=[redacted]"); + assert.equal(redact('api_key: "abcd1234efgh"'), "api_key: [redacted]"); + const bearer = redact("Authorization: Bearer abcdefghijklmnop.qrstuvwxyz-12345"); + assert.ok(!bearer.includes("abcdefghijklmnop"), "bearer token scrubbed"); + assert.match(bearer, /Bearer \[redacted\]/); +}); + +test("PEM private keys are scrubbed whole (multi-line)", () => { + const redact = buildRedactor({}); + const pem = "-----BEGIN RSA PRIVATE KEY-----\nMIIEow...lines...\n-----END RSA PRIVATE KEY-----"; + assert.equal(redact(`key:\n${pem}\ndone`), "key:\n[redacted:private-key]\ndone"); +}); + +test("redactEvent walks nested fields and arrays, leaves non-strings alone", () => { + const redact = buildRedactor({ MY_SECRET_TOKEN: "deadbeefcafe1234" }); + const e = redactEvent( + { type: "observe", passed: false, output: "err deadbeefcafe1234", extra: { arr: ["x deadbeefcafe1234"] } }, + redact + ); + assert.equal(e.passed, false, "boolean untouched"); + assert.match(e.output, /\[redacted:MY_SECRET_TOKEN\]/); + assert.match(e.extra.arr[0], /\[redacted:MY_SECRET_TOKEN\]/); +}); + +test("redactingSink scrubs before the inner sink sees the event", async () => { + const seen = []; + const inner = { post: (e) => seen.push(e), flush: async () => {} }; + const sink = redactingSink(inner, buildRedactor({ CI_TOKEN: "0123456789abcdef" })); + sink.post({ type: "observe", passed: true, output: "ok 0123456789abcdef" }); + await sink.flush(); + assert.equal(seen.length, 1); + assert.ok(!JSON.stringify(seen[0]).includes("0123456789abcdef"), "raw secret never reached the sink"); +}); diff --git a/packages/runtime/test/resume.test.js b/packages/runtime/test/resume.test.js new file mode 100644 index 0000000..ec67447 --- /dev/null +++ b/packages/runtime/test/resume.test.js @@ -0,0 +1,183 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { parse } from "@loop-lang/parser"; +import { run, buildResumePlan, MockRunner, ScriptedHumanIO } from "../dist/index.js"; + +/** Serialize captured events the way the file sink writes them (header + seq'd lines). */ +const asLog = (events, meta) => + [ + JSON.stringify({ v: "loop.log.v1", runId: "r1", ts: "t", meta }), + ...events.map((event, seq) => JSON.stringify({ seq, ts: "t", event })), + ].join("\n") + "\n"; + +const PIPELINE = [ + 'pipeline "ship":', + ' stage "build":', + " goal: it builds", + ' done when "build" passes', + " each cycle: act, then observe", + ' stage "verify":', + " goal: tests pass", + ' done when "test" passes', + " each cycle: act, then observe", + ' after 1 tries: stop and warn "stuck"', +].join("\n"); + +/** A verifier scripted per predicate command. */ +const verifierFor = (results) => ({ + verify: async (pred) => ({ passed: !!results[pred.command], output: `${pred.command}: ${results[pred.command] ? "ok" : "boom"}` }), +}); + +function stubs(verifier, events) { + return { + runner: new MockRunner(), + verifier, + human: new ScriptedHumanIO(), + baseDir: "/p", + hardCap: 2, + readText: async () => "", + writeText: async () => {}, + onEvent: (e) => events?.push(e), + }; +} + +test("buildResumePlan: satisfied stages recorded, failed/incomplete not; nested sub-runs ignored", () => { + const log = asLog( + [ + { type: "pipeline-start", name: "ship" }, + { type: "stage-start", name: "build" }, + { type: "loop-start", name: "build" }, // the stage's own loop — must not count as a definition + { type: "loop-end", name: "build", satisfied: true }, + { type: "stage-end", name: "build", satisfied: true }, + { type: "stage-start", name: "verify" }, + { type: "loop-start", name: "verify" }, + // crash here — no end events for the verify stage or the pipeline + ], + { loop_sha256: "abc" } + ); + const plan = buildResumePlan(log); + assert.deepEqual([...plan.completed], ["stage:0:build"]); + assert.equal(plan.sourceHash, "abc"); + assert.equal(plan.runId, "r1"); +}); + +test("buildResumePlan: a fully satisfied top-level definition becomes def:", () => { + const log = asLog([ + { type: "loop-start", name: "a" }, + { type: "loop-end", name: "a", satisfied: true }, + { type: "loop-start", name: "b" }, + { type: "loop-end", name: "b", satisfied: false }, + ]); + assert.deepEqual([...buildResumePlan(log).completed], ["def:0"]); +}); + +test("buildResumePlan: flow steps keep their summaries; foreach items keyed by step+var+index", () => { + const log = asLog([ + { type: "flow-start", name: "deliver" }, + { type: "flow-step-start", name: "design", ref: "design.loop" }, + { type: "loop-start", name: "d" }, // nested sub-file run + { type: "loop-end", name: "d", satisfied: true }, + { type: "flow-step-end", name: "design", satisfied: true, summary: "DESIGN OK" }, + { type: "flow-step-start", name: "stories", ref: "story.loop" }, + { type: "foreach-start", var: "story", source: "sprint.yaml", count: 3 }, + { type: "foreach-item-start", var: "story", index: 0, total: 3 }, + { type: "loop-start", name: "s0" }, + { type: "loop-end", name: "s0", satisfied: true }, + { type: "foreach-item-end", var: "story", index: 0, satisfied: true }, + { type: "foreach-item-start", var: "story", index: 1, total: 3 }, + // crash mid-item 1 + ]); + const plan = buildResumePlan(log); + assert.deepEqual([...plan.completed].sort(), ["item:0:stories:story:0", "step:0:design"]); + assert.equal(plan.summaries.get("step:0:design"), "DESIGN OK"); +}); + +test("end-to-end: crash after stage 1 → resume skips stage 1, runs only stage 2", async () => { + const file = parse(PIPELINE); + + // Run 1: build passes, verify fails (thrash guard stops it) — like a run that died mid-way. + const events1 = []; + const s1 = stubs(verifierFor({ build: true, test: false }), events1); + const out1 = await run(file, s1); + assert.equal(out1[0].satisfied, false, "first run ends unsatisfied"); + assert.equal(s1.runner.actCalls.length >= 2, true, "both stages acted in run 1"); + + // Run 2: resume from run 1's log; the test is now fixed. + const plan = buildResumePlan(asLog(events1)); + const events2 = []; + const s2 = stubs(verifierFor({ build: true, test: true }), events2); + const out2 = await run(file, { ...s2, resume: plan }); + + assert.equal(out2[0].satisfied, true, "resumed run finishes"); + // Stage "build" must NOT re-run: every act in run 2 belongs to the verify stage. + assert.ok(s2.runner.actCalls.length >= 1); + assert.ok(s2.runner.actCalls.every((a) => a.goal.includes("tests pass")), "build stage never re-acted"); + const resumed = events2.find((e) => e.type === "resumed"); + assert.deepEqual({ unit: resumed.unit, name: resumed.name }, { unit: "stage", name: "build" }); +}); + +test("end-to-end: a wholly satisfied definition is skipped via def:", async () => { + const src = 'loop "one":\n goal: g\n done when "ok" passes\n each cycle: act, then observe'; + const file = parse(src); + const events1 = []; + const s1 = stubs(verifierFor({ ok: true }), events1); + await run(file, s1); + + const plan = buildResumePlan(asLog(events1)); + assert.ok(plan.completed.has("def:0")); + const s2 = stubs(verifierFor({ ok: true })); + const out = await run(file, { ...s2, resume: plan }); + assert.equal(out[0].satisfied, true); + assert.equal(s2.runner.actCalls.length, 0, "nothing re-ran"); +}); + +test("flow resume: skipped step's recorded summary is restored as the next step's upstream", async () => { + const flowSrc = ['flow "ship":', ' run "a.loop"', ' then run "b.loop"'].join("\n"); + const subA = 'loop "a":\n goal: step a\n done when "a-ok" passes\n each cycle: act, then observe'; + const subB = 'loop "b":\n goal: step b\n done when "b-ok" passes\n each cycle: plan, then act, then observe\n after 1 tries: stop and warn "stuck"'; + const loadFile = async (ref) => parse(ref === "a.loop" ? subA : subB); + + // Run 1: a succeeds, b fails → flow stops. The log records a's summary on its step-end. + const events1 = []; + const s1 = { ...stubs(verifierFor({ "a-ok": true, "b-ok": false }), events1), loadFile, flowStack: ["/p/f.loop"] }; + const out1 = await run(parse(flowSrc), s1); + assert.equal(out1[0].satisfied, false); + const aEnd = events1.find((e) => e.type === "flow-step-end" && e.satisfied); + assert.ok(aEnd.summary, "step end carries its summary for future resumes"); + + // Run 2: resume — step a skipped, step b runs with a's recorded summary as upstream. + const plan = buildResumePlan(asLog(events1)); + const s2 = { ...stubs(verifierFor({ "a-ok": true, "b-ok": true })), loadFile, flowStack: ["/p/f.loop"] }; + const out2 = await run(parse(flowSrc), { ...s2, resume: plan }); + assert.equal(out2[0].satisfied, true); + assert.ok(s2.runner.actCalls.every((a) => a.goal === "step b"), "step a never re-ran"); + assert.equal(s2.runner.planCalls[0].upstream, aEnd.summary, "carry-forward restored from the log"); +}); + +test("foreach resume: delivered items are skipped, the failed one re-runs", async () => { + const flowSrc = ['flow "deliver":', ' for each story in "sprint.yaml":', ' run "story.loop"'].join("\n"); + const tmpl = 'loop "story":\n goal: build it\n done when "story-ok" passes\n each cycle: act, then observe\n after 1 tries: stop and warn "stuck"'; + const loadFile = async () => parse(tmpl); + const readText = async () => "- login\n- signup\n- reset\n"; + + // Run 1: item 0 passes, item 1 fails and the human stops the flow. + let call = 0; + const flaky = { verify: async () => ({ passed: call++ === 0, output: "x" }) }; + const events1 = []; + const human1 = new ScriptedHumanIO({ defaults: { gate: false } }); // stop at the failed item + const s1 = { ...stubs(flaky, events1), human: human1, loadFile, readText, flowStack: ["/p/f.loop"] }; + const out1 = await run(parse(flowSrc), s1); + assert.equal(out1[0].satisfied, false); + + // Run 2: resume — item 0 skipped, items 1..2 run (verifier green now). + const plan = buildResumePlan(asLog(events1)); + assert.ok([...plan.completed].some((k) => k.startsWith("item:")), "item 0 recorded"); + const green = { verify: async () => ({ passed: true, output: "ok" }) }; + const events2 = []; + const s2 = { ...stubs(green, events2), loadFile, readText, flowStack: ["/p/f.loop"] }; + const out2 = await run(parse(flowSrc), { ...s2, resume: plan }); + assert.equal(out2[0].satisfied, true); + const skipped = events2.filter((e) => e.type === "resumed" && e.unit === "foreach-item"); + assert.deepEqual(skipped.map((e) => e.index), [0], "exactly item 0 skipped"); + assert.equal(s2.runner.actCalls.length, 2, "items 1 and 2 ran, item 0 did not"); +}); diff --git a/packages/vscode/src/extension.ts b/packages/vscode/src/extension.ts index 1a6bcc3..e2a5be9 100644 --- a/packages/vscode/src/extension.ts +++ b/packages/vscode/src/extension.ts @@ -255,6 +255,7 @@ function renderEvent(e: any): string | null { case "pipeline-start": return `▶ pipeline "${e.name}"`; case "stage-start": return ` ■ stage "${e.name}"`; case "stage-end": return ` ■ stage "${e.name}" → ${e.satisfied ? "satisfied" : "FAILED"}`; + case "resumed": return ` ⏩ ${e.unit} ${e.name ? `"${e.name}"` : ""}${e.index !== undefined ? ` #${e.index + 1}` : ""} — resumed`; case "loop-start": return `↻ loop ${e.name ? `"${e.name}"` : ""}`.trimEnd(); case "node-enter": return ` · ${e.node} (try ${e.attempt})`; case "observe": return ` = ${e.passed ? "PASS" : "fail"}${e.output ? ` — ${String(e.output).split("\n")[0].slice(0, 80)}` : ""}`; From 2e630a0735bea470860b49cea92e8af6beb5d44c Mon Sep 17 00:00:00 2001 From: Idan Ayalon Date: Thu, 2 Jul 2026 00:06:27 -0400 Subject: [PATCH 09/14] =?UTF-8?q?feat(docs):=20browser=20playground=20?= =?UTF-8?q?=E2=80=94=20type=20a=20.loop,=20see=20its=20shape=20live?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The parser, the ASCII "show" renderer, explain, and the soft linter are all pure TypeScript — one esbuild pass (npm run build:playground) bundles the whole language toolchain to 27 kB of client-side JS. docs/playground.html is a split-pane editor: parse-on-type with inline errors, the compact flow view, lint nudges, and the plain-English explain, plus a picker of examples that showcase the grammar (flake guard, judge panels, pipelines, flows). Linked from the tutorial nav and the hands-on cards. No install, no server. Co-Authored-By: Claude Fable 5 --- docs/index.html | 6 +- docs/playground.html | 187 +++++++++++++++++++++++++++++++++++ docs/playground/loop-lang.js | 12 +++ package.json | 1 + scripts/playground-entry.ts | 11 +++ 5 files changed, 215 insertions(+), 2 deletions(-) create mode 100644 docs/playground.html create mode 100644 docs/playground/loop-lang.js create mode 100644 scripts/playground-entry.ts diff --git a/docs/index.html b/docs/index.html index a4a4904..875de86 100644 --- a/docs/index.html +++ b/docs/index.html @@ -223,6 +223,7 @@
  • Go deeper
  • 📖 Full manual →
  • 📖 Keyword reference →
  • +
  • ⚡ Playground →
  • 🛠️ Workshop →
  • 🎮 LoopFlow Lab →
  • @@ -442,8 +443,9 @@

    2 · Run your first loop — in a Claude Code chat Safe from the first run: with no git: block, LoopFlow works on a branch and commits when the goal is met, and never pushes to main/master — see the git keyword.

    Rather learn hands-on?

    -

    Two guided ways to get the language into your fingers before the deep dive below:

    -
    +

    Three guided ways to get the language into your fingers before the deep dive below:

    + diff --git a/docs/playground.html b/docs/playground.html new file mode 100644 index 0000000..7d9db17 --- /dev/null +++ b/docs/playground.html @@ -0,0 +1,187 @@ + + + + + + +LoopFlow — playground + + + + + + +
    +

    LoopFlow playground

    + type a .loop — see its shape, live · everything runs in your browser + ← the tutorial +
    + +
    +
    +
    + + + ✓ parses +
    + +
    +
    +
    +
    +
    + + + + + diff --git a/docs/playground/loop-lang.js b/docs/playground/loop-lang.js new file mode 100644 index 0000000..88ea8bb --- /dev/null +++ b/docs/playground/loop-lang.js @@ -0,0 +1,12 @@ +var Loop=(()=>{var y=Object.defineProperty;var K=Object.getOwnPropertyDescriptor;var Y=Object.getOwnPropertyNames;var X=Object.prototype.hasOwnProperty;var J=(e,n)=>{for(var o in n)y(e,o,{get:n[o],enumerable:!0})},Q=(e,n,o,t)=>{if(n&&typeof n=="object"||typeof n=="function")for(let i of Y(n))!X.call(e,i)&&i!==o&&y(e,i,{get:()=>n[i],enumerable:!(t=K(n,i))||t.enumerable});return e};var Z=e=>Q(y({},"__esModule",{value:!0}),e);var Pe={};J(Pe,{ParseError:()=>u,explainFile:()=>_,lint:()=>U,parse:()=>x,renderFile:()=>H});var S="0.1",u=class extends Error{constructor(o,t){super(`Loop parse error (line ${t}): ${o}`);this.line=t;this.name="ParseError"}};var j=["vibe coding","structured ai-assisted","agentic engineering"],ee={skills:"skills",skill:"skills",agents:"agents",agent:"agents",mcps:"mcps",mcp:"mcps","mcp-server":"mcps","mcp-servers":"mcps",harnesses:"harnesses",harness:"harnesses"},te={"before each cycle":"before-cycle","after plan":"after-plan","after act":"after-act","after observe":"after-observe","on commit":"on-commit","on push":"on-push","on stop":"on-stop"};function ne(e){let n=e.trim().toLowerCase().replace(/[.,]$/,"");return{edit:"edit",edits:"edit",editing:"edit",migration:"migrate",migrations:"migrate",migrate:"migrate",push:"push",pushes:"push",pushing:"push",deploy:"deploy",deploys:"deploy",deployment:"deploy",deployments:"deploy",delete:"delete",deletes:"delete",deletion:"delete",deletions:"delete"}[n]??n}function C(e){return e.split(/,|\bor\b|\band\b/).map(n=>ne(n)).filter(n=>n.length>0)}function oe(e){let n=[];return e.replace(/\r\n/g,` +`).split(` +`).forEach((t,i)=>{let s=i+1,p="",l=!1;for(let h=0;ho;)t.push(e[i]),i++;return{body:t,next:i}}function b(e){let n=e.match(/"([^"]*)"/);return n?n[1]:null}function k(e,n){let o=e.trim(),t=l=>{let a=l?parseInt(l,10):1;return a>1?{runs:a}:{}},i=o.match(/^the test\s+"([^"]+)"\s+passes(?:\s+(\d+)\s+times?)?$/i);if(i)return{type:"test",target:i[1],...t(i[2])};if(i=o.match(/^"([^"]+)"\s+finds nothing(?:\s+(\d+)\s+times?)?$/i),i)return{type:"command",command:i[1],expect:"empty",...t(i[2])};if(i=o.match(/^"([^"]+)"\s+(?:passes|succeeds)(?:\s+(\d+)\s+times?)?$/i),i)return{type:"command",command:i[1],expect:"exit-zero",...t(i[2])};if(i=o.match(/^a human confirms\s+"([^"]+)"$/i),i)return{type:"human",description:i[1]};let s=l=>l?{subject:l.toLowerCase()}:{},p=l=>{let a=l?parseInt(l,10):1;return a>1?{judges:a}:{}};if(i=o.match(/^the skill\s+"([^"]+)"\s+scores\s+(\d+)(?:\s+or more)?(?:\s+on the (output|trajectory))?(?:\s+by\s+(\d+)\s+judges?)?$/i),i)return{type:"skill",skill:i[1],expect:"approve",minScore:parseInt(i[2],10),...s(i[3]),...p(i[4])};if(i=o.match(/^the skill\s+"([^"]+)"\s+approves(?:\s+on the (output|trajectory))?(?:\s+by\s+(\d+)\s+judges?)?$/i),i)return{type:"skill",skill:i[1],expect:"approve",...s(i[2]),...p(i[3])};throw new u(`could not understand "done when ${o}"`,n)}function A(e,n){let o=e.split(/,|\bthen\b/).map(i=>i.trim()).filter(i=>i.length>0),t=[];for(let i of o){let s=i.toLowerCase(),p=i.match(/^stop and warn\s+"([^"]+)"$/i);if(p){t.push({action:"stop",warn:p[1]});continue}if(s==="stop"){t.push({action:"stop"});continue}if(p=i.match(/^reflect(?:\s+on\s+(.+))?$/i),p){t.push(p[1]?{action:"reflect",focus:p[1].trim()}:{action:"reflect"});continue}if(s==="plan"||s==="plan again"||s==="replan"){t.push({action:"plan"});continue}if(s==="act"||s==="act again"){t.push({action:"act"});continue}if(s==="observe"){t.push({action:"observe"});continue}if(s==="ask a human"||s==="ask the human"||s==="ask human"){t.push({action:"ask-human"});continue}throw new u(`unknown action "${i}"`,n)}if(t.length===0)throw new u("expected at least one action",n);return t}function ie(e,n){let o=e.trim().toLowerCase();if(/^it passes and (the )?goal is met$/.test(o))return{on:"pass",requireGoalMet:!0};if(/^it passes$/.test(o))return{on:"pass"};if(/^it (fails|breaks)$/.test(o))return{on:"fail"};if(/^(it is |it gets )?(blocked|stuck)$/.test(o))return{on:"blocked"};throw new u(`unknown condition "when ${e}"`,n)}function se(e,n){let o=e.match(/^(before each cycle|after plan|after act|after observe|on commit|on push|on stop):\s*(.+)$/i);if(!o)throw new u(`unrecognized hook "${e}" (expected e.g. \`on commit: "cmd" finds nothing\`)`,n);let t=k(o[2],n);if(t.type!=="command"&&t.type!=="test")throw new u(`a hook must be a deterministic check (a command or test), not "${o[2]}"`,n);return{at:te[o[1].toLowerCase()],predicate:t}}function O(e,n){let o=e.split(/,|\bthen\b/).map(i=>i.trim().toLowerCase()).filter(i=>i.length>0),t=[];for(let i of o)if(i==="plan"||i==="act"||i==="observe")t.push(i);else throw new u(`unknown cycle step "${i}" (expected plan, act, or observe)`,n);if(t.length===0)throw new u("empty cycle",n);return t}function re(e,n,o){let t;if(/^work in place$/i.test(n)){e.isolation="in-place";return}if(t=n.match(/^work on a branch(?:\s+"([^"]+)")?$/i)){e.isolation="branch",t[1]&&(e.branch=t[1]);return}if(t=n.match(/^work in a worktree(?:\s+"([^"]+)")?$/i)){e.isolation="worktree",t[1]&&(e.branch=t[1]);return}if(/^commit when (?:the goal is met|done)$/i.test(n)){e.commit="done";return}if(/^commit each cycle$/i.test(n)){e.commit="cycle";return}if(/^commit each story$/i.test(n)){e.commit="story";return}if(/^(?:commit never|do not commit)$/i.test(n)){e.commit="never";return}if(/^(?:push when done|push)$/i.test(n)){e.push=!0;return}if(/^do not push$/i.test(n)){e.push=!1;return}if(/^open a (?:pull request|pr)$/i.test(n)){e.openPr=!0;return}throw new u(`unrecognized git line: "${n}"`,o)}function F(e,n){let o=e[n],{body:t,next:i}=w(e,n+1,o.indent);if(t.length===0)throw new u("empty git block",o.lineNo);let s={};for(let p of t)re(s,p.text,p.lineNo);return{git:s,next:i}}var E=["plan","act","reflect","also"];function T(e,n){let o={},t,i=new Set;for(let s of e.split(",")){let p=s.trim();if(!p)continue;let l=p.split(/\s+/),a=l[0].toLowerCase(),h=r=>{let c=l[r]?.toLowerCase();return c==="fast"||c==="strong"?c:void 0};if(a==="all"){let r=h(1);if(l.length!==2||!r)throw new u(`models: "all" needs a tier (fast|strong): "${p}"`,n);t=r}else if(a==="fast"||a==="strong"){if(l.length!==2)throw new u(`models: tier "${a}" needs one model: "${p}"`,n);(o.tiers??={})[a]=l[1]}else if(E.includes(a)){let r=h(1);if(l.length!==2||!r)throw new u(`models: phase "${a}" needs a tier (fast|strong): "${p}"`,n);(o.phases??={})[a]=r,i.add(a)}else{if(a==="observe")continue;throw new u(`models: unrecognized clause "${p}"`,n)}}if(t!==void 0){o.phases??={};for(let s of E)i.has(s)||(o.phases[s]=t)}return o}function M(e,n,o){let t={kind:"loop",name:e,goal:"",cycle:[]},i=null,s=!1,p=!1,l=0;for(;la.indent;)c.push(se(n[f].text,n[f].lineNo)),f++;if(c.length===0)throw new u("'hooks:' block is empty",a.lineNo);t.hooks=c,l=f;continue}if(r=h.match(/^models:\s*(.+)$/i)){t.models=T(r[1],a.lineNo),l++;continue}if(r=h.match(/^goal:\s*(.+)$/i)){t.goal=r[1].trim(),s=!0,l++;continue}if(r=h.match(/^done when\s+(.+)$/i)){let c=k(r[1],a.lineNo),f=n[l+1];if(f&&f.indent>a.indent){let d=f.text.match(/^the bar:\s*(.+)$/i);if(d){if(c.type!=="skill")throw new u(`'the bar:' applies to a skill eval, not "${r[1].trim()}"`,f.lineNo);c.bar=d[1].trim(),l++}}(t.doneWhen??=[]).push(c),l++;continue}if(r=h.match(/^(?:look at|look in|files|context|in):\s*(.+)$/i)){t.context=ae(r[1]),l++;continue}if(r=h.match(/^(?:check|verify):\s*(.+)$/i)){let c=r[1].trim(),d=/^(the test|the skill|a human)\b/i.test(c)||/^".*"\s+(passes|succeeds|finds nothing)(\s+\d+\s+times?)?$/i.test(c)?k(c,a.lineNo):{type:"command",command:c.replace(/^"|"$/g,""),expect:"exit-zero"};(t.doneWhen??=[]).push(d),l++;continue}if(/^allow\b/i.test(h)||/^ask me before\b/i.test(h)){le(t,h),l++;continue}if(r=h.match(/^(?:then\s+)?each cycle:\s*(.+)$/i)){t.cycle=O(r[1],a.lineNo),p=!0,l++;continue}if(r=h.match(/^also(?:\s+do)?:\s*(.+)$/i)){t.also=r[1].split(",").map(c=>c.trim()).filter(c=>c.length>0),l++;continue}if(r=h.match(/^use skills?:\s*(.+)$/i)){t.skills=r[1].split(/,|\band\b/).map(c=>c.trim()).filter(c=>c.length>0),l++;continue}if(r=h.match(/^use skills recommended by ctx(?:\s+for\s+"([^"]+)")?$/i)){t.skillDiscovery=r[1]?{provider:"ctx",intent:r[1].trim()}:{provider:"ctx"},l++;continue}if(/^top up skills from ctx(?:\s+when a step needs more)?$/i.test(h)){t.skillTopUp=!0,l++;continue}if(r=h.match(/^use tools from (?:the\s+)?"([^"]+)"(?:\s+server)?$/i)){(t.tools??=[]).push(r[1].trim()),l++;continue}if(r=h.match(/^examples?:\s*(.+)$/i)){(t.context??={}).examples=R(r[1]),l++;continue}if(r=h.match(/^knowledge:\s*(.+)$/i)){(t.context??={}).knowledge=R(r[1]),l++;continue}if(r=h.match(/^(?:remember|keep a memory)\s+in\s+"([^"]+)"$/i)){t.memory={file:r[1].trim()},l++;continue}if(r=h.match(/^plan from "([^"]+)"$/i)){t.planSource={type:"file",path:r[1]},l++;continue}if(/^a human approves the plan first$/i.test(h)){t.humanPlan=!0,l++;continue}if(/^a human reviews before stopping$/i.test(h)){t.humanReviewBeforeStop=!0,l++;continue}if(r=h.match(/^a human approves before\s+(.+)$/i)){i={message:`approve before ${r[1].trim()}`},l++;continue}if(r=h.match(/^when\s+(.+?):\s*(.+)$/i)){let c=ie(r[1],a.lineNo),f=A(r[2],a.lineNo);(t.transitions??=[]).push({...c,do:f}),l++;continue}if(r=h.match(/^after\s+(\d+)\s+tries:\s*(.+)$/i)){let c=A(r[2],a.lineNo);(t.transitions??=[]).push({on:"attempts",threshold:parseInt(r[1],10),do:c}),l++;continue}throw new u(`unrecognized line: "${h}"`,a.lineNo)}if(!s)throw new u(`loop "${e??"(anonymous)"}" is missing a goal`,n[0]?.lineNo??0);if(p||(t.cycle=o?.cycle?.length?[...o.cycle]:["plan","act","observe"]),o?.rigor==="structured ai-assisted"||o?.rigor==="agentic engineering"){let a=h=>(t.transitions??[]).some(r=>r.on===h);t.doneWhen?.length&&!a("fail")&&(t.transitions??=[]).push({on:"fail",do:[{action:"reflect"},{action:"plan"}]}),a("fail")&&!a("attempts")&&(t.transitions??=[]).push({on:"attempts",threshold:8,do:[{action:"stop",warn:"thrashing"}]})}return{loop:t,gate:i}}function R(e){return e.split(",").map(n=>n.replace(/^and\s+/i,"").trim()).filter(Boolean)}function ae(e){let n={},o=e.split(",").map(i=>i.trim()).filter(Boolean),t=[];for(let i of o)i=i.replace(/^and\s+/i,"").trim(),/^the last failure$/i.test(i)?n.includeLastFailure=!0:i.length>0&&t.push(i);return t.length&&(n.files=t),n}function le(e,n){let o=e.policy??{},t=n.match(/allow\s+(.+?)\s+automatically/i);t&&(o.auto=[...o.auto??[],...C(t[1])]),t=n.match(/ask me before\s+(.+?)(?:\.|$)/i),t&&(o.confirm=[...o.confirm??[],...C(t[1])]),e.policy=o}function N(e,n,o){let t=e[n],i=t.text.match(/^stage\s+(.+?):\s*$/i);if(!i)throw new u('expected "stage :"',t.lineNo);let s=i[1].trim(),p=b(s)??s,{body:l,next:a}=w(e,n+1,t.indent);if(l.length===0)throw new u(`stage "${p}" has no body`,t.lineNo);let{loop:h,gate:r}=M(null,l,o);return{stage:{name:p,gate:r,loop:h},next:a}}function ce(e,n,o){let t=e[n],i=b(t.text)??t.text.replace(/^pipeline\s+/i,"").replace(/:$/,"").trim(),{body:s,next:p}=w(e,n+1,t.indent),l=[],a=0,h=0;for(;af;){if(!/^stage\b/i.test(s[d].text))throw new u(`expected a "stage" inside the parallel group in pipeline "${i}"`,s[d].lineNo);let{stage:m,next:g}=N(s,d,o);m.parallelGroup=h,l.push(m),d=g}a=d;continue}if(!/^stage\b/i.test(s[a].text))throw new u(`expected a "stage" inside pipeline "${i}"`,s[a].lineNo);let{stage:r,next:c}=N(s,a,o);l.push(r),a=c}if(l.length===0)throw new u(`pipeline "${i}" has no stages`,t.lineNo);return{pipeline:{kind:"pipeline",name:i,stages:l},next:p}}function pe(e,n,o){let t=e[n],i=b(t.text),{body:s,next:p}=w(e,n+1,t.indent),{loop:l}=M(i,s,o);return{loop:l,next:p}}function he(e,n){let o=e[n],t=o.text.match(/^(?:then\s+)?for each\s+(\w+)\s+in\s+"([^"]+)":$/i);if(t){let r=t[1],c=t[2],{body:f,next:d}=w(e,n+1,o.indent),m=null,g=null;for(let $ of f){let L=$.text.match(/^run\s+"([^"]+)"$/i);if(L){m=L[1];continue}if(/^a human approves(?:\s+(?:the plan\s+)?first)?$/i.test($.text)){g={message:`approve before ${r}`};continue}let P=$.text.match(/^a human approves before\s+(.+)$/i);if(P){g={message:`approve before ${P[1].trim()}`};continue}throw new u(`unrecognized line in 'for each ${r}': "${$.text}"`,$.lineNo)}if(!m)throw new u(`'for each ${r}' needs a 'run "