diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index f778d8b51..6be906ea1 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -19,7 +19,9 @@ jobs: quality: name: quality (${{ matrix.os }}) runs-on: ${{ matrix.os }} - timeout-minutes: 30 + # The whole job's budget, not a test's: the Windows runner needed ~29 min + # by 0.10.0 and was cut off mid-accessibility gate at 30 (2026-10-01). + timeout-minutes: 45 strategy: fail-fast: false matrix: @@ -161,17 +163,21 @@ jobs: - run: npm ci - run: npm run package # What must be inside the package: the licence, the third-party - # notices, both dictation helpers and the five bundles (the webview + # notices, the native helpers and every runtime bundle (the webview # loads main.css beside main.js; the extension requires modelApi.js - # when the Model API backend first starts, M57). + # when the Model API backend first starts, M57, and starts + # pageWorker.js as a worker for each web page it converts, M69). - name: the .vsix carries the notices, the helpers, the bundles and the manifest's text run: | listing="$(unzip -Z1 ./*.vsix)" for entry in extension/LICENSE.txt extension/THIRD_PARTY_NOTICES.txt extension/package.nls.json \ - extension/native/windows/dictate.ps1 extension/native/darwin/muse-dictate \ + extension/native/windows/dictate.ps1 extension/native/windows/capture.ps1 \ + extension/native/windows/MuseSparkJob.cs extension/native/windows/MuseSparkMcpJob.cs \ + extension/native/windows/MuseSparkMcpLauncher.cs extension/native/darwin/muse-dictate \ extension/dist/extension.js extension/dist/modelApi.js extension/dist/searchWorker.js \ + extension/dist/pageWorker.js extension/dist/planMarkdown.js extension/dist/checkpointStore.js \ extension/dist/webview/main.js extension/dist/webview/main.css; do - if ! grep -qx "$entry" <<< "$listing"; then + if ! grep -Fxq -- "$entry" <<< "$listing"; then echo "::error::$entry is missing from the .vsix" >&2 exit 1 fi @@ -182,6 +188,27 @@ jobs: name: muse-spark-code-vsix path: '*.vsix' if-no-files-found: error + # The ACP agent's npm package (M63, PLAN.md D62), from the same + # production build `npm run package` just ran. + - run: node scripts/package-acp.mjs + - name: the agent's package carries its bundles, tables, notices and manifest + run: | + listing="$(tar -tzf dist/muse-spark-code-acp-*.tgz)" + for entry in package/package.json package/dist/acp.js package/dist/modelApi.js \ + package/dist/searchWorker.js package/dist/pageWorker.js \ + package/native/windows/MuseSparkJob.cs package/native/windows/MuseSparkMcpJob.cs \ + package/THIRD_PARTY_NOTICES.txt package/LICENSE package/README.md package/l10n/ui.de.json; do + if ! grep -Fxq -- "$entry" <<< "$listing"; then + echo "::error::$entry is missing from the agent's package" >&2 + exit 1 + fi + done + echo "the agent's package holds $(wc -l <<< "$listing") entries, the required ones included" + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: muse-spark-code-acp + path: dist/muse-spark-code-acp-*.tgz + if-no-files-found: error secrets: name: gitleaks diff --git a/.github/workflows/forks.yml b/.github/workflows/forks.yml new file mode 100644 index 000000000..bb5b2c308 --- /dev/null +++ b/.github/workflows/forks.yml @@ -0,0 +1,79 @@ +# The VS Code forks that ship Linux builds (PLAN.md M62b): Cursor, Devin +# Desktop (Windsurf until 2026), Kiro and Positron, each at its latest +# release from its own update feed (test/hosts/fork-release.mjs). Each job +# installs the .vsix with the fork's CLI and runs the integration tests in +# the fork; the job summary names the fork's version and its VS Code base. +# +# Apart from hosts.yml because these are the forks' latest builds, not +# pinned ones, and their feeds can change without notice: every Monday, by +# hand, and on pull requests that change this check. Every job has a +# timeout and leaves no token in the git config; none pushes. +name: Forks + +on: + pull_request: + paths: + - 'test/hosts/fork-release.mjs' + - 'test/hosts/installed.vscode-test.mjs' + - 'test/hosts/run-fork.sh' + - '.github/workflows/forks.yml' + schedule: + - cron: '41 6 * * 1' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: forks-${{ github.ref }} + cancel-in-progress: true + +jobs: + vsix: + name: vsix + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - run: npm run package + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: forks-vsix + path: '*.vsix' + if-no-files-found: error + + fork: + name: ${{ matrix.fork }} + needs: vsix + runs-on: ubuntu-latest + timeout-minutes: 25 + strategy: + fail-fast: false + matrix: + fork: [cursor, devin-desktop, kiro, positron] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - run: npm run build:dev + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: forks-vsix + path: forks-vsix + - name: ${{ matrix.fork }} + env: + # Positron's latest release is read from the GitHub API. + GH_TOKEN: ${{ github.token }} + run: xvfb-run -a sh test/hosts/run-fork.sh "${{ matrix.fork }}" forks-vsix/*.vsix "$RUNNER_TEMP/fork" diff --git a/.github/workflows/hosts.yml b/.github/workflows/hosts.yml new file mode 100644 index 000000000..cbe93ca50 --- /dev/null +++ b/.github/workflows/hosts.yml @@ -0,0 +1,267 @@ +# The host checks (PLAN.md M62, M63): the extension in the editors built on +# VS Code, and the ACP agent in the editors that speak ACP, each driven as +# docs/certification/m62.md and m63.md recorded it, against the fake Muse +# Code CLI (no model is called and no real key is used). Each job runs one +# script of test/hosts, the same script a local run uses. +# +# Pull requests and pushes to main that touch the product or these checks, +# every Monday (the hosts' own new releases), and by hand. Every job has a +# timeout and leaves no token in the git config; none pushes. +name: Hosts + +on: + pull_request: + paths: + - 'src/**' + - 'l10n/**' + - 'package.json' + - 'package-lock.json' + - 'scripts/build.mjs' + - 'scripts/package-acp.mjs' + - 'test/e2e/**' + - 'test/hosts/**' + - 'test/integration/**' + - '.github/workflows/hosts.yml' + push: + branches: [main] + paths: + - 'src/**' + - 'l10n/**' + - 'package.json' + - 'package-lock.json' + - 'scripts/build.mjs' + - 'scripts/package-acp.mjs' + - 'test/e2e/**' + - 'test/hosts/**' + - 'test/integration/**' + - '.github/workflows/hosts.yml' + schedule: + - cron: '17 6 * * 1' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: hosts-${{ github.ref }} + cancel-in-progress: true + +jobs: + # The .vsix (without the macOS dictation helper, which no check here + # needs) and the agent's npm package, from one production build. + packages: + name: packages + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - run: npm run package + - run: node scripts/package-acp.mjs + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: hosts-vsix + path: '*.vsix' + if-no-files-found: error + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: hosts-acp + path: dist/muse-spark-code-acp-*.tgz + if-no-files-found: error + + # The agent as npm installs it, on each platform: the stdio suite against + # the installed package (its own native keyring binding, no repository + # modules), then the key's round trip through the platform's credential + # store (D61). + agent: + name: agent package (${{ matrix.os }}) + needs: packages + runs-on: ${{ matrix.os }} + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest, macos-latest] + defaults: + run: + shell: bash + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: hosts-acp + path: hosts-acp + - run: npm install --global ./hosts-acp/muse-spark-code-acp-*.tgz + - name: the stdio suite against the installed package + run: MUSE_ACP_PACKAGE_DIR="$(npm root --global)/muse-spark-code-acp" npx vitest run test/e2e/acpStdio.e2e.test.ts + - name: the key through the Secret Service (linux) + if: runner.os == 'Linux' + run: | + sudo apt-get update -q + sudo apt-get install -y -q gnome-keyring + dbus-run-session -- sh -c 'printf "ci\n" | gnome-keyring-daemon --unlock --components=secrets > /dev/null && sh test/hosts/keystore.sh muse-spark-code-acp' + # A keychain of the job's own, unlocked, as the default: the runner's + # login keychain would ask a person to unlock it. + - name: the key through the Keychain (macos) + if: runner.os == 'macOS' + run: | + security create-keychain -p ci hosts.keychain + security list-keychains -d user -s hosts.keychain login.keychain + security default-keychain -d user -s hosts.keychain + security unlock-keychain -p ci hosts.keychain + security set-keychain-settings hosts.keychain + sh test/hosts/keystore.sh muse-spark-code-acp + - name: the key through Credential Manager (windows) + if: runner.os == 'Windows' + run: sh test/hosts/keystore.sh muse-spark-code-acp + + # The integration tests in VSCodium: the floor's release and the latest. + vscodium: + name: VSCodium ${{ matrix.release }} + runs-on: ubuntu-latest + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + release: ['1.99.32846', latest] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - run: npm run build:dev + - run: xvfb-run -a sh test/hosts/run-vscodium.sh "${{ matrix.release }}" "$RUNNER_TEMP/vscodium" + + # The .vsix in code-server, the floor's release (VS Code 1.99.3, Node + # 20.18) and the latest, driven in Chrome. + code-server: + name: code-server ${{ matrix.release }} + needs: packages + runs-on: ubuntu-latest + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + release: ['4.99.4', latest] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: hosts-vsix + path: hosts-vsix + - name: code-server + env: + GH_TOKEN: ${{ github.token }} + RELEASE: ${{ matrix.release }} + run: | + version="$RELEASE" + if [ "$version" = latest ]; then + version="$(gh release view --repo coder/code-server --json tagName --jq .tagName | sed 's/^v//')" + fi + mkdir -p "$RUNNER_TEMP/code-server" + curl -fsSL "https://github.com/coder/code-server/releases/download/v$version/code-server-$version-linux-amd64.tar.gz" \ + | tar -xz -C "$RUNNER_TEMP/code-server" --strip-components=1 + - run: sh test/hosts/run-code-server.sh "$RUNNER_TEMP/code-server/bin/code-server" hosts-vsix/*.vsix "$RUNNER_TEMP/code-server-check" + - if: failure() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: code-server-${{ matrix.release }}-evidence + path: | + ${{ runner.temp }}/code-server-check/shots + ${{ runner.temp }}/code-server-check/code-server.log + ${{ runner.temp }}/code-server-check/data/logs + + # The .vsix in Eclipse Theia (test/hosts/theia/package.json), built from npm. + theia: + name: Eclipse Theia + needs: packages + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - run: npm ci + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: hosts-vsix + path: hosts-vsix + - run: sh test/hosts/run-theia.sh hosts-vsix/*.vsix "$RUNNER_TEMP/theia-check" + - if: failure() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: theia-evidence + path: | + ${{ runner.temp }}/theia-check/shots + ${{ runner.temp }}/theia-check/theia.log + + # The agent in the ACP clients that install on Linux: JupyterLab (Jupyter + # AI), Emacs (acp.el, agent-shell) and Neovim (CodeCompanion). + acp-clients: + name: ACP client ${{ matrix.client }} + needs: packages + runs-on: ubuntu-latest + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + client: [jupyter, emacs, neovim] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + cache: npm + - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + if: matrix.client == 'jupyter' + with: + python-version: '3.12' + - run: npm ci + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: hosts-acp + path: hosts-acp + - run: npm install --global ./hosts-acp/muse-spark-code-acp-*.tgz + - if: matrix.client == 'emacs' + run: | + sudo apt-get update -q + sudo apt-get install -y -q emacs-nox + - run: sh "test/hosts/run-${{ matrix.client }}.sh" "$(command -v muse-spark-code-acp)" "$RUNNER_TEMP/${{ matrix.client }}-check" + - if: failure() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: ${{ matrix.client }}-evidence + path: | + ${{ runner.temp }}/${{ matrix.client }}-check/shots + ${{ runner.temp }}/${{ matrix.client }}-check/*.log + if-no-files-found: ignore diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 86ecfa1de..7a8110dea 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -3,7 +3,12 @@ # and the Marketplace publish runs when the VSCE_PAT secret of the # `marketplace` environment is set (a Marketplace "Manage" PAT scoped to the # publisher's organisation; without it the publish step is skipped and -# reported, and `npx vsce publish --packagePath` by hand still works). +# reported, and `npx vsce publish --packagePath` by hand still works). The +# same .vsix goes to Open VSX, for VSCodium, Cursor, Kiro, Positron and the +# other editors that install from it, when the environment's OVSX_PAT is +# set (PLAN.md D62, M62), under the same rules. The ACP agent's package +# (muse-spark-code-acp, M63) rides on the GitHub Release and goes to npm +# when the environment's NPM_TOKEN is set (Q65). # # Before anything is built, the tag must name the manifest's version and # point at a commit on main. The PAT reaches one step only, after an @@ -69,13 +74,17 @@ jobs: with: name: muse-spark-code-vsix path: release + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: muse-spark-code-acp + path: release - name: release notes from the CHANGELOG run: node scripts/changelog-notes.mjs "${GITHUB_REF_NAME#v}" > release/notes.md - name: create the GitHub Release env: GH_TOKEN: ${{ github.token }} run: | - gh release create "${GITHUB_REF_NAME}" release/*.vsix \ + gh release create "${GITHUB_REF_NAME}" release/*.vsix release/*.tgz \ --title "${GITHUB_REF_NAME}" \ --notes-file release/notes.md \ --verify-tag @@ -122,3 +131,82 @@ jobs: env: VSCE_PAT: ${{ secrets.VSCE_PAT }} run: ./node_modules/.bin/vsce publish --packagePath release/*.vsix + + openvsx: + name: Open VSX publish + needs: release + runs-on: ubuntu-latest + timeout-minutes: 10 + # Its own job, so either registry failing leaves the other published; + # the token lives in the same tag-only environment as VSCE_PAT. + environment: marketplace + env: + HAS_OVSX_PAT: ${{ secrets.OVSX_PAT != '' }} + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: muse-spark-code-vsix + path: release + - name: skipped without OVSX_PAT + if: env.HAS_OVSX_PAT != 'true' + run: echo "OVSX_PAT is not set; the Open VSX publish was skipped. Publish by hand with npx ovsx publish release/*.vsix" >> "$GITHUB_STEP_SUMMARY" + # The locked ovsx, with no install script of any package run. + - name: install the locked tools without install scripts + if: env.HAS_OVSX_PAT == 'true' + run: npm ci --ignore-scripts --no-audit + # Open VSX refuses a publish into a namespace that does not exist, and + # the package's publisher had none when this was checked (2026-09-27, + # `GET /api/RandyNorthrup` 404). The first run creates it with the same + # token; later runs find it (200) and skip. Any other answer stops here. + - name: create the Open VSX namespace if it is missing + if: env.HAS_OVSX_PAT == 'true' + env: + OVSX_PAT: ${{ secrets.OVSX_PAT }} + run: | + namespace=$(node -p "require('./package.json').publisher") + status=$(curl -s -o /dev/null -w '%{http_code}' "https://open-vsx.org/api/${namespace}") + if [ "$status" = "404" ]; then + ./node_modules/.bin/ovsx create-namespace "$namespace" + elif [ "$status" != "200" ]; then + echo "Open VSX answered HTTP $status for namespace $namespace" >&2 + exit 1 + fi + - name: publish to Open VSX + if: env.HAS_OVSX_PAT == 'true' + env: + OVSX_PAT: ${{ secrets.OVSX_PAT }} + run: ./node_modules/.bin/ovsx publish release/*.vsix + + npm: + name: npm publish (ACP agent) + needs: release + runs-on: ubuntu-latest + timeout-minutes: 10 + # The same tag-only environment as the other registries' tokens. + environment: marketplace + env: + HAS_NPM_TOKEN: ${{ secrets.NPM_TOKEN != '' }} + steps: + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 22 + registry-url: https://registry.npmjs.org + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: muse-spark-code-acp + path: release + - name: skipped without NPM_TOKEN + if: env.HAS_NPM_TOKEN != 'true' + run: echo "NPM_TOKEN is not set; muse-spark-code-acp was not published to npm. The GitHub Release carries its package." >> "$GITHUB_STEP_SUMMARY" + # The packed tarball as it is: no install, no script of any package. + - name: publish muse-spark-code-acp to npm + if: env.HAS_NPM_TOKEN == 'true' + env: + NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} + run: npm publish release/muse-spark-code-acp-*.tgz --access public --ignore-scripts diff --git a/.vscode-test.mjs b/.vscode-test.mjs index 20dc6dcdc..3b20c1bd0 100644 --- a/.vscode-test.mjs +++ b/.vscode-test.mjs @@ -1,5 +1,7 @@ +import { tmpdir } from 'node:os' import { defineConfig } from '@vscode/test-cli' import { minimumVsCodeVersion } from './scripts/lib/vscode-engine.mjs' +import { launchArgsFor } from './scripts/lib/vscodeTestProfile.mjs' // Integration tests run inside a real VS Code (Extension Development Host). // `npm run build:dev` bundles test/integration/**/*.test.ts to @@ -15,11 +17,20 @@ const MOCHA_TIMEOUT_MS = 20_000 const shared = { files: 'dist/test/integration/**/*.test.js', workspaceFolder: './test/fixtures/workspace', - launchArgs: ['--disable-extensions'], mocha: { ui: 'tdd', timeout: MOCHA_TIMEOUT_MS, color: true }, } -export default defineConfig([ - { ...shared, label: 'stable', version: 'stable' }, - { ...shared, label: 'minimum', version: minimumVsCodeVersion() }, -]) +// Under a long checkout path (macOS caps a Unix socket path at 104 bytes) each +// run gets a short user-data directory; see scripts/lib/vscodeTestProfile.mjs. +function run(label, version) { + const launchArgs = launchArgsFor({ + base: ['--disable-extensions'], + label, + checkout: process.cwd(), + platform: process.platform, + tmp: tmpdir(), + }) + return { ...shared, label, version, launchArgs } +} + +export default defineConfig([run('stable', 'stable'), run('minimum', minimumVsCodeVersion())]) diff --git a/.vscodeignore b/.vscodeignore index cdc36aa36..ccc647e1a 100644 --- a/.vscodeignore +++ b/.vscodeignore @@ -1,7 +1,10 @@ ** !dist/extension.js !dist/modelApi.js +!dist/planMarkdown.js +!dist/checkpointStore.js !dist/searchWorker.js +!dist/pageWorker.js !dist/webview/main.js !dist/webview/main.css !media/icon.svg diff --git a/AGENTS.md b/AGENTS.md index 8cc237e01..c49fbb67c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -53,6 +53,24 @@ them, the milestone plan, and the certification checklist. 8. **Secrets never leave SecretStorage.** No API keys in settings, logs, telemetry, tests, or fixtures. The pasted Model API key is never passed to any child process: the Muse Code CLI signs in on its own. + - **Outside VS Code (the ACP agent, PLAN.md D61).** The operating + system's credential store stands in for SecretStorage: it is the store + VS Code's SecretStorage itself rests on. The key goes in only through + `muse-spark-code-acp auth set`, from its standard input (the user's + terminal, or a pipe into it); never from an environment variable, an + argument or a file; and it is never passed to a child process + (`muse serve`, a tool, a check). A `META_API_KEY` the user sets in the + agent's own environment is theirs for Muse Code, not the stored key: + as the extension's `muse serve` inherits it (PLAN.md D1), it reaches + Muse Code's processes only, where it counts as Muse Code's credential. + The agent takes every credential variable (`*_API_KEY` and the named + ones hooks never get) out of its own environment at start, so no + shell command, hook, git or helper it starts sees one. + - **The one exception: M80's CI bootstrap** (PLAN.md M80, planned). + GitHub hands a secret to a step only through its environment or its + script, so the Action's own step shell is the one environment the key + is ever in: that shell pipes it to `auth set`'s standard input and + unsets it before `exec` starts. Nothing else is excepted. - **The CLI's credential file.** The extension reads only its structure (`src/core/backends/musecode/credentialFile.ts`): the schema version, which providers are named (only `meta` speaks for the sign-in), each @@ -92,7 +110,9 @@ them, the milestone plan, and the certification checklist. `muse serve`). Every use asks first in the paid-use popup (M58, D48: `PaidUseConsent` in `src/core/paid/paidConsent.ts`, Allow once / Allow always in this workspace / Deny), in every mode, Bypass included; a - paid call never gets an approval card or a session rule. The + paid call never gets an approval card or a session rule. In the ACP + agent (D62) a feature is on only with its flag, and the same question + is the editor's permission prompt (`src/acp/paid.ts`). The subscription never pays for one. One exception to the key is planned (PLAN.md D50, M85, experimental): the TypeSafe assist is billed to the user's own TypeSafe key instead of the Model API key; every other part @@ -110,14 +130,30 @@ them, the milestone plan, and the certification checklist. src/extension.ts activation: the view, the panel, the commands, the openers src/host/** VS Code adapters (views, conversation, backend managers, the Model API bundle's entry (dist/modelApi.js, loaded - when that backend first starts) and the search worker, + when that backend first starts), the plan reader's + (dist/planMarkdown.js, loaded on the first plan action), + the search worker and web fetch's page converter worker + (dist/pageWorker.js, started for each page), commands, auth, settings, mentions, - editor tracking, usage trace logs, voice, the diagnostics - MCP server, the MCP servers' spawner, the network posture) + editor tracking, usage trace logs, voice, the IDE tool + MCP server (diagnostics, code intelligence, images, web + fetch), VS Code's language services, the MCP servers' + spawner, the network posture, web fetch's pinned + transport and the verify loop's editor side: settled + diagnostics, format on edit and turn checkpoints' shadow repository) src/core/** backend-agnostic logic; must not import `vscode` (MSP host, Model API client and tools, the MCP client, context, Muse Code's memory, export, worktrees, usage, - dictation, Muse Voice, the paid gate, network failures) + dictation, Muse Voice, the paid gate, network failures, + code intelligence and the repo map, web fetch's + public-address checks and HTML converter, the verify + loop's check commands, diagnostics report and the files + it never opens because tools run them, the checkpoint restore plan) +src/acp/** the ACP agent (D62): the ACP side of a session and the + translation of the engine's events; must not import + `vscode` +src/runtime/** the agent's process: arguments, backends outside VS Code, + the OS credential store (D61), `auth` and `login` src/shared/** constants + zod protocol shared by host and webview src/shared/l10n/** the English table (en.ts), fill/plural/Intl helpers, the table checks and the list of translated languages @@ -137,15 +173,25 @@ test/unit/** vitest (node + jsdom via docblock); `vscode` is mocked test/e2e/** the fake Muse Code CLI driven through the real backend; the opt-in live drills (the Muse Code CLI; the Model API sweep, which bills the owner's key) -test/integration/** @vscode/test-cli, runs inside VS Code +test/integration/** @vscode/test-cli, runs inside VS Code, over the workspace + test/fixtures/workspace (code-intel/ is a TypeScript + project its language service reads) test/harness/ the webview behind a fake host, for screenshots and the accessibility gate; themes/ holds VS Code's four themes +test/hosts/ the extension and the ACP agent in other editors + against the fake CLI, one script per host (hosts.yml) scripts/** esbuild build; bundle-size, bundle-split, host-globals, - notices, audit, PSScriptAnalyzer, semgrep, accessibility - and localization gates; theme capture, the pseudo-locale, - harness screenshots, image rendering, changelog notes, - VS Code versions for CI + notices, audit, PSScriptAnalyzer, semgrep, accessibility, + localization and host API gates; theme capture, the + pseudo-locale, harness screenshots, image rendering, + changelog notes, VS Code versions for CI, the ACP + agent's package docs/certification/ per-milestone gate-fire records and screenshots +docs/ide-compatibility.md, docs/ide-compatibility/ + the plan for editors beyond VS Code (D60), the + generated record of what the extension asks of its host, + and hosts.md, what each editor was tested at +docs/acp.md the ACP agent's guide, shipped as its package's README media/ icons, banner, social preview, README screenshots ``` @@ -157,12 +203,15 @@ media/ icons, banner, social preview, README screenshots | The gates CI runs everywhere | `npm run quality:gates` | | Accessibility gate | `npm run test:a11y` | | Localization gate | `npm run check:l10n` | +| Host API record (D60) | `npm run check:host-api` (`-- --write`) | | Panel in the pseudo-locale | `npm run harness:shots -- --lang=pseudo` | | Unit tests with coverage | `npm run test:unit` | | Integration tests | `npm run test:integration` | | Dev build / watch | `npm run build:dev` / `npm run watch` | | Production build + size budget | `npm run build` | | Package `.vsix` | `npm run package` | +| Package the ACP agent (D62) | `npm run package:acp` | +| Host checks (hosts.yml) | `sh test/hosts/run-.sh` | ## Toolchain pins that matter diff --git a/CHANGELOG.md b/CHANGELOG.md index bf6b2fe44..a14c89f74 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,8 +7,475 @@ happened, not what was planned; superseded entries are kept. ## [Unreleased] +## [0.10.0] - 2026-10-01 + +### Highlights + +- **Other editors.** The ACP agent (`muse-spark-code-acp`) brings Muse Spark to Zed, JetBrains IDEs, Neovim, Emacs and more (`docs/acp.md`). +- **Code intelligence and rename.** The agent finds definitions, references and symbols through VS Code's language services, and renames a symbol everywhere it is used. +- **The agent checks its own edits.** After edits, the language servers' errors and your check commands reach the next request (Model API backend). +- **Web fetch.** The model reads one public HTTPS page on either backend, fetched from your machine and free. +- **Plans as files.** Save a Plan-mode reply to `.agents/plans/` and implement it in a fresh conversation. +- **Turn checkpoints (Preview, off by default).** Restore files, the conversation or both, then redo; the copies never touch your `.git` (stored restore needs a Model API session). Turn them on with `museSpark.turnCheckpoints`; their restore is being rebuilt on the tools' own writes (PLAN.md D63). +- **Every paid use asks first.** A popup (Allow once, Allow always in this workspace, Deny) in every mode, Bypass included. +- **A lighter start.** The Model API backend is a bundle of its own, loaded only when a conversation uses it. +- **VS Code 1.99 or newer** (was 1.125), so editors built on VS Code 1.99 or later can install the extension. +- **Open VSX and npm publishing.** A release tag also publishes the VSIX to Open VSX and the ACP agent to npm (each when its token is set). + +### Added + +- **The agent checks its own edits** (M68, PLAN.md D49). On the Model API + backend, after each round of tool calls that edited files, the next + request carries the edited files' errors and warnings from VS Code's + language servers, with what changed since each file's previous check + (`museSpark.diagnosticsAfterEdits`, on by default), and the results of + your **check commands** (`museSpark.checkCommands`: lint, test or + type-check commands, none by default), within one 64,000-character + budget. A **Check edits** row shows the files, their problems, how many + were **not checked** (no report arrived, unsaved, or past the first 8; + never reported clean) and how each check ended. +- **Check commands take the shell tool's hooks and permission path.** Your + PreToolUse, PostToolUse and PostToolUseFailure hooks see each check and + `then_run` command as a shell call (deny, rewrite, ask, add context, stop + the turn), and a hook's denial is shown apart from your Reject. Each asks + wherever a shell command would ask (every mode but Bypass permissions), + "Always allow in this session" allows that check (never the agent's own + shell call of the same command, nor the other way round) until the agent + edits a + file that decides what it runs (`package.json`, a `Makefile`, a config + the tools load, a file it names; any file, for a command with quotes, + variables or other shell syntax), and none runs in Plan mode or Restricted + Mode. `changedFiles` passes the edited files that still exist after `--`, + each quoted as one argument, and refuses a file name that starts with `-` + or `@`, or on Windows holds `"`, `&`, `|`, `<`, `>`, `^`, `%` or `!`; + `timeoutSeconds` caps each run. A rejected check is not asked again, and + after three failing rounds in a row the checks stop, until your next + message (a message you add while the agent works starts them again too); + the model and the panel say so. A round counts as failing only by checks + run on the files as they now are, and passes only when none of those + fails; an edit's `then_run` of the own command of a check that does not + take the changed files counts as that check (a pass only when the check's + time limit is no shorter than the shell's). No such check runs twice for + the same state of the files, and nothing runs after the turn's last + round. +- **`run_checks`**, the model's own call of the checks (on files that exist + in the workspace, or those edited since your message), and **`then_run`** + on `write_file` and `edit_file`: one command run right after the edit, + asked for like any shell command, run only if the file still holds what + the edit wrote, and shown under the diff as the call's second result + (SoL-Pi's Action Fusion, reimplemented from its description). +- **Format on edit** (`museSpark.formatOnEdit`, off by default): the file's + formatter runs on each file the Model API backend's edit tools write, + before anything checks it. The formatted text is written only while the + file still holds what the edit wrote and has no unsaved changes in an + editor; otherwise, or when it cannot be written, the edit stays as + written and the log says why. +- **Code the editor runs is never opened or formatted by the loop** + (`eslint.config.js`, `.prettierrc.cjs`, `package.json`, `node_modules`), + and once the agent writes such a file nothing more is opened or formatted + until your next message. +- **Muse Code** is told with each message to check the files it edits with + `mcp__ide__getDiagnostics` (when the session has the IDE tool server) and + to run your check commands. + +- **Plans as files** (M79, PLAN.md D49). In Plan mode the latest reply gets + two buttons, **Save plan** and **Implement in a fresh conversation**. + Pressing one is the approval: neither backend marks a plan or its approval + on the wire (Muse Code 1.4.0 was captured live). + - **Save plan** writes the plan byte for byte to + `.agents/plans/YYYY-MM-DD-.md`, Muse Code's own convention, with a + numeric suffix when the name is taken; an existing file is never + replaced. A Muse Code plan reply's two handoff lines ("Reply `go` to + execute this plan…") are left out. `.agents` is protected, so the save + asks first; Restricted Mode refuses it. + - **Implement in a fresh conversation** starts a new conversation on the + same backend. Its first message is the plan file, attached as named + text, and nothing else from the planning conversation, which stays in + History. Plan mode gives way to the starting mode. + - On the Model API backend, the plan's steps become the todo list before + the first request, and the brief names them. On Muse Code, which keeps + its todo list to the model, the brief asks Muse to list the steps. + - **Plans…** in the palette lists the saved plans, newest date first, to + open or implement. A plan file is untrusted content (PLAN.md D49): one + implemented from Plans… starts in Manual (Plan when that is the + starting mode) and is never presented to the model as approved. + - Only a reply to a message sent in Plan mode, in a turn that stayed in + it, counts as a plan. Save and Implement resume the conversation after + a restart, find a plan already saved instead of writing it twice, and + say why when they do nothing. What the model gets is what the user + saw: the reply is shown, and the brief written, from one rewritten + Markdown tree (a link's destination beside its text, a picture's + source, titles, definitions, footnotes and code-fence info as text), so + nothing in the brief is hidden in the panel. A plan with raw HTML is + saved with a warning and not started; one with a control or format + character (a direction override, a zero-width character) is neither + saved nor started. The log names a plan by a + verified date and a hash, or by the hash alone, never by its name. + - The plan reader (the panel's Markdown parser) is a bundle of its own, + `dist/planMarkdown.js` (budget 150 KiB), loaded on the first Save plan, + Implement or Plans…, so the activation bundle does not carry it. If it + cannot load, those actions are refused with the reason. + - A reload of the conversation (a delivery gap) keeps Save plan and + Implement under a plan reply. A plan that cannot open from Plans… says + why in the panel and logs only the kind of failure. + - Leaving Plan mode when the backend refuses the change keeps a running + Plan-mode turn a plan turn. A reasoning effort the session refuses is + no longer shown as applied. +- **Memory and plans: folder re-check.** A new memory note or plan is + refused when its folder was swapped for a link or junction after it was + checked (`createFileExclusively` checks again after making the folder and + before publishing). +- **Web fetch on both backends** (M69, PLAN.md D49; folds in M44b). The + model can read one public web page it found or you named: `web_fetch` on + the Model API backend, and `mcp__ide__webFetch` on Muse Code, whose own + `web_fetch` is switched off. The extension fetches the page from your + machine; it is free, not Meta's paid search. + - `https://` only, public internet addresses only: the name is resolved + here and refused when any answer is loopback, private, link-local, + carrier-grade NAT, cloud metadata or reserved (IPv4-mapped, NAT64 and + 6to4 forms judged by the IPv4 inside), and local or reserved names are + refused before any lookup. The connection goes to the address that was + checked, never to a second lookup; TLS still verifies the name. + - Same-host redirects are checked and pinned again, at most five; a + redirect to another host is handed back to the model. 5 MiB after + decompression, 30 seconds, an allow-list of text types; HTML becomes + Markdown, text stays as it is, anything else is refused with the reason. + - Through VS Code's proxy and certificates: the proxy is asked to tunnel + to the checked address, and a proxy's own answer is refused as such, + never read as the page. + - On the Model API backend it asks per host in Manual, Edit automatically + and Auto ("Always allow in this session" covers that host), runs in + Bypass, and is refused in Plan and in Restricted Mode. On Muse Code the + tool is listed only in a trusted workspace whose + `museSpark.sandboxNetwork` is not `restricted`, declares itself + open-world and not read-only, and the extension asks in its own dialog + before every fetch. + - The model receives the page between random markers, with a note that it + is untrusted content; the row shows the URL, the size and type, and what + the model read. 23 new strings in fifteen languages. + - After review: Muse Code's Stop (a closed request, or + `notifications/cancelled`) stops the fetch and voids a later answer in + the dialog, which is also asked only once per URL at a time and checks + the workspace again after it; the checked addresses are raced as RFC 8305 + says; failures name the page's host and the addresses tried instead of + M56's advice about Meta, and a network failure's detail only by its + error codes (never a certificate's names); a server's text reaches the + model outside the markers only as short tokens; the HTML converter is bounded; names with + trailing dots or empty labels are refused; a network's own NAT64 prefix + is discovered (RFC 7050), and while it cannot be learned no IPv6 answer + is used (only a DNS answer proves there is none); pages are parsed by + HTML's own rules (implied ends, misnested and self-closed tags, SVG and + MathML, comments and scripts), and the Markdown is the page's text as + served, which can include text a browser would not show (no stylesheet + or hiding attribute is read, since hiding cannot be worked out + completely and visible small print hides nothing), all of it between + the untrusted markers, as the tool's description and the note now say; + a hook's "allow" no longer replaces + the per-host card; trust + and the mode are asked again after the card, before each request and + before the page reaches the model; damaged compression and unknown charsets are + handled as a browser would; `museSpark.sandboxNetwork`'s description now + says it also hides web fetch from Muse Code. Fifteen more strings, one + changed and one dropped, and two changed setting descriptions, in + fifteen languages. +- **Dependencies.** Web fetch parses HTML with `parse5` 8.0.1 (MIT) + and sniffs its encoding with `html-encoding-sniffer` 6.0.0 (MIT), both + already in the tree through the test tools. They load only in + `dist/pageWorker.js` (201 KiB, budget 300 KiB), on a worker thread started for each page (at + most two at once) and stopped at 10 seconds or 512 MiB; `dist/extension.js` does not carry + them. `entities` is no longer a direct dependency. +- **Code intelligence** (M67, PLAN.md D49): the agent finds definitions, + references and symbols the way the editor does, from VS Code's own + language services, instead of searching text. `find_definition`, + `find_references`, `workspace_symbols`, `document_symbols`, `hover`, + `call_hierarchy` and `repo_map` are reads in every mode on the Model API + backend, Plan and Restricted Mode included; a symbol is named by path, + line and column, by its name on a line or in a file, or by name alone. + Answers are workspace-relative, sorted and capped, and say how many + results outside the workspace (a library's declarations, another folder) + they left out; a hover for a symbol defined only outside the workspace + (and outside the languages' own libraries) is held back. A file whose + language has no service, or that declares nothing, says so instead of + answering nothing, and an empty answer says that not every language + provides every kind. In a file with unsaved changes a line number is + refused and a name is found in the editor's text, said so, the editor + matched by the file's real path (a workspace opened through a link). A + call hierarchy names the file of each outgoing call site, and counts the + other functions a position names (overloads) that it did not ask. On the Muse + Code backend the same tools are served to Muse Code as + `mcp__ide__findDefinition` and the rest, each marked read-only; Muse Code + 1.4.0 still shows its own card for them in its on-request mode. +- **`rename_symbol`** renames a symbol everywhere it is used. On the Model + API backend it is an edit: the card names its files, a protected one + first (and is a protected write when one of them is); every file is + checked again after the card and once more right before its own write, + and Stop before the first write writes nothing; the row carries one + patch, so Revert and rewind undo it, a rename stopped partway included. + It refuses an edit that also creates, moves or deletes files, and one + whose ranges no longer cover the old name (made from an older version of + a file). A hook matching `Edit` runs for it, with the files it would + write, and the rename writes the plan the hook was shown. On Muse Code it changes nothing and hands Muse Code the diff to + apply with its own edit tool. +- **A repo map in the Model API's prompt** (`museSpark.modelApiRepoMap`, + off by default, machine-scoped, trusted workspaces only): the files other + files use most, with their most used definitions, within about 1,000 + tokens. It is kept once made, tried again on a later turn when a try finds + nothing (three tries at most), and shared with child tasks and forks. It + adds those tokens to every request, its heading and notes counted in + the budget. `repo_map` gives the same map on request either way, all of + its text within `max_tokens`; a budget too small for its own lead and + notes is refused with the size that would do. +- **Groundwork for editors other than VS Code** (M60, M61, PLAN.md D60). + The owner's IDE compatibility plan is filed in `docs/ide-compatibility.md`. + A new gate, `npm run check:host-api`, keeps a record of what the + extension asks of its host (`docs/ide-compatibility/host-api.md`): the + initial 201 VS Code APIs and where, the 13 files that imported `vscode`, + the Node built-ins, and what the webview needs (`acquireVsCodeApi` and 57 + theme variables); it fails when the record goes stale, and when the + engine, the protocol, the webview, the conversation controller, either + backend or the credential store reaches `vscode`. The webview now talks + to VS Code through one host bridge, and the chat surface, the log and the + dictation setup carry no VS Code types. Nothing changes in VS Code. +- **Muse Spark for editors that speak ACP** (M63, PLAN.md D61, D62). + `muse-spark-code-acp`, an npm package attached to each GitHub Release, + runs Muse Code or the Model API as an Agent Client Protocol agent for + Zed, JetBrains IDEs, Neovim, Emacs and the other ACP editors: the chat, + tool calls with diffs, the plan, permission prompts (a cancelled or + unknown answer rejects), questions as forms, the modes, the model and + effort, skills as commands, and sessions listed, loaded and resumed. The + backend is chosen when the editor starts it and never switches. + `muse-spark-code-acp auth set` keeps a Model API key in the operating + system's credential store (Windows Credential Manager, the macOS + Keychain, the Secret Service on Linux, with no plaintext fallback); the + key is never read from the environment or passed to Muse Code. Paid + features (web search, image generation) are off unless the editor + starts the agent with `--web-search` or `--image-generation`, and then + each use asks first in the editor's permission prompt, naming the price, + as the panel's popup does (M58): Allow once, Allow always in this + workspace (only with `--trust-workspace`, kept in the agent's data + folder and forgotten when the agent starts without the flag) or Deny. + Subagents stay off in the agent. The agent loads the Model API backend + from the same `dist/modelApi.js` the extension ships (M57), which its + package carries. The editor's MCP + servers (stdio and HTTP) are passed to Muse Code, so Jupyter AI's + notebook tools and Zed's context servers reach it. See `docs/acp.md`; + which editors have been tried is tracked in + `docs/ide-compatibility/hosts.md` (Zed, Emacs + with agent-shell, Neovim with CodeCompanion and JupyterLab with Jupyter + AI so far). On the Model API backend, a trusted folder gets Muse Code's + memory tools as the panel does; subagents, which are paid, are not + offered, and the package carries the C# of the shell tool's Windows job. +- **The ACP agent and proxies.** The agent runs outside VS Code, so VS + Code's proxy and certificate settings do not reach it, and Node's own + `fetch` ignores `HTTPS_PROXY` unless `NODE_USE_ENV_PROXY=1` (Node 22.21 or + later, or 24). The agent does not re-route by itself: on the Model API + backend it warns once in its log at start when a proxy variable is set + and would not be used (or this Node cannot use one), and a request that + never reaches Meta now names the variables to set in the agent's + environment instead of VS Code's `http.*` settings, in all 15 languages. + `docs/acp.md` has a new "Networks and proxies" section. +- **The ACP agent's Muse Code sign-in is read as the panel reads it** + (PR #49): from the credential file's structure, and the CLI's + `account/read` where only it can say, so an agent after `muse logout` + asks for sign-in instead of failing its first turn. A Muse Code agent no + longer clears the Model API agent's "Allow always" when it starts, and a + Model API agent whose `dist/modelApi.js` is missing says to reinstall the + agent, not the extension. +- **The ACP agent keeps credential variables to Muse Code.** A + `META_API_KEY` in the agent's environment still reaches Muse Code and + counts as its sign-in, as with the extension, but no shell command, hook + or git the agent runs sees it or any other `*_API_KEY` variable. A + loaded or resumed session now runs in the mode, model and effort the + editor shows (or fails to load), the model as Muse Code reports it; a + session the agent could not set up is let go rather than left running, + and closing or reloading a session ends its running prompt as + cancelled and stops that turn, with any answer still owed to it + ignored; + a question's form answer is used only + when it is one of the form's own options, in the allowed number; and the + agent's log names a backend failure by its kind, never its message. +- **The key outside VS Code, in the rules** (AGENTS.md rule 8, PLAN.md + D61). Outside VS Code the operating system's credential store stands in + for SecretStorage: the ACP agent's key goes in only through `auth set`'s + standard input and never reaches a child process. The one named + exception is the planned CI bootstrap (M80), whose step shell pipes the + key to `auth set` and unsets it before the run. +- **Open VSX and npm publishing** in the release workflow. A tag also + publishes the VSIX to Open VSX, for VS Code forks that install from + there, and the agent to npm, each only when its token is set in the + `marketplace` environment. +- **Host checks in CI** (the Hosts workflow, `test/hosts/`). Every pull + request that touches the product runs development-extension integration + in VSCodium and packaged browser checks in code-server (the 1.99 floor + and the latest) and Eclipse Theia, + and the packaged ACP agent on Linux, macOS and Windows (its key through + each credential store) and in JupyterLab, Emacs and Neovim, all against + a fake Muse Code CLI; the latest releases are tried again every Monday. + A second workflow, Forks, installs the VSIX in the latest Linux builds + of Cursor, Devin Desktop (formerly Windsurf), Kiro and Positron, then + runs development-extension integration there, weekly and by hand. + ### Changed +- **The diagnostics tool reads a file no editor shows.** VS Code's language + servers report only on files an editor shows (TypeScript and JSON, + measured in VS Code 1.139.1 and 1.125.0), so when the agent asks + `getDiagnostics` about one such file, on either backend, the extension + opens it in a tab beside your editor without taking focus, waits up to 10 + seconds for its report, and closes the tab again; a file outside the + workspace by its real path, or code the editor runs, is not opened, and a + file no report arrived for is "not checked", never clean. +- **VS Code 1.99 or newer** (was 1.125; M62, PLAN.md A8), so editors built + on VS Code 1.99 or later can install the extension. The extension uses + no VS Code API newer than 1.85, checked against every published + `@types/vscode` from 1.85 on, and its host bundles now need nothing + newer than Node 20.18, the Node of VS Code 1.99 and 1.100. Tested in + VSCodium 1.99.3 and 1.135 (the integration tests, 9 passing in each) and + in code-server 4.99.4 (VS Code 1.99.3: a conversation and an approval + in the browser), where the 1.125 floor was refused; and in the latest + Cursor, Devin Desktop (formerly Windsurf), Kiro and Positron, which + install it and pass the integration tests. On 1.99 and 1.100 Muse Voice + says it is unavailable, as their Node has no WebSocket. VS Code routes an + extension's WebSocket through its proxy support only from 1.112, so on + 1.101 to 1.111 Muse Voice's socket goes to Meta without it; **Muse Spark: + Diagnostics** now says whether this editor routes the extension's `fetch` + and WebSocket at all, instead of reading only the settings, and the + README's Proxies and certificates section says what to use. The Model + API's requests are routed on every supported version. +- **README: How this extension is built** (Development): the owner, Claude + Code as lead, up to four headless Muse Code builders on the contributor + model, Grok Build and Codex as reviewers, and the gates on dedicated test + machines and CI. + +- **Turn checkpoints: restore files, the conversation, or both, then redo** + (M72, PLAN.md D51). Each turn gets a checkpoint of the workspace's files at + its start and end, on both backends, untracked files included and + ignored files left out. A sent message's menu offers **Restore files to + here** and **Rewind conversation and restore files**. A restore undoes + what the conversation's turns changed from that message on, including + what shell commands changed. It names every file it left as it is: + changed since the turn (by you, a build, or another conversation's + overlapping turn), unsaved in an editor or notebook, not in the + checkpoint, or could not be changed. Each file is checked again just + before it changes, deletions come before writes (a case-only rename or a + file that became a folder comes back), and no file is written or deleted + through a link or junction. Its notice has **Redo**, which puts back what + the restore replaced; the redo record is saved before the first file + changes, so a restore that stops part way still reports what it changed + and keeps Redo, and a redo that could not do everything keeps its button. + **Rewind conversation and restore files** checks the conversation first + and rewinds it only when every file was restored. A folder that was there + before the turn, even an empty one, is never removed. +- **Independent windows share checkpoint records without overwriting them.** + Current-version windows use a shared canonical-root namespace under extension + global storage, including different VS Code workspace identities/aliases. + Old per-workspace captures remain preserved, readable and labelled read-only. + + Per-record Git compare-and-swap refs replace the shared lock and JSON file; + each window owns its index, pending captures, staged copies and presence. + A restore or Redo reserves one shared ref before file changes. Archives + are durable before returning, including in Restricted Mode, and cleanup + spares recent objects while another window creates its refs. Model API + turns await their running mark before hooks, edits or model calls, queued + and scheduled turns and children included; a failed mark prevents them + from running. A child outliving its parent keeps restores blocked. + Tool preimages remain until the end record persists; retry keeps the + original boundary rather than recapturing later user changes. Cancellation + stops before the next file and finalizes Redo for the files actually + changed. Errors shown in the panel are localized and storage paths are + excluded from failure logs. Stored destructive Restore/Redo is limited to + the actual attached Model API session with confirmed process safety. Every + extension-managed CLI skills/import/export or sandbox command, and + interactive/auth/MCP/installer terminal, awaits durable native admission + before process or terminal creation. Shutdown or backend disposal cancels + a held start. Terminal callbacks are awaited through command and auth + callers; terminal closure does not clear unproved descendant activity. + A workspace Muse Code serve (account probes included) publishes the same + unsafe marker before startup. + **Create AGENTS.md** applies this admission to `muse init`; refused startup + cannot choose the template fallback. The explicit pure template write holds + a file-edit lease until actual I/O settles, including in Restricted Mode + without Git, and checks shutdown/generation before writing. + Extension-managed Git worktree add/remove publish the same native marker + before normal repository hooks run, with fresh owned-cwd/lifetime checks. + Pure plan publication/stage cleanup and file-review Revert hold an activity + lease through actual I/O, preserving no-clobber and stage ownership checks. + Automatic prompt Git status/log now suppress configured fsmonitor, + signature and clean/process helpers through bounded names-only overrides, + with live trust/owner/cwd checks before actual spawn. Plain metadata leaves + file restore eligible; ordinary Git configuration remains unchanged. + The independent ACP runtime explicitly records its no-VS-Code-checkpoint + startup policy; it neither offers stored Restore nor claims the shared + window fence. Its external-editor scope is documented in the ACP guide. + After joined-main review, plans retain current trust/lifetime checks through + native staging/publication and stale-stage removal. Restore/Redo use the + existing conditional writer for final expected-file/dirty checks, including + binary/absent expectations, and recheck unlinked deletion destinations. + Executable modes and durable partial Redo are retained; the final + comparison-to-syscall residual is documented. + Shell/hook activity is tracked through its actual promise, including `!`, + background and child work; unproved descendant shutdown leaves sticky unsafe + presence rather than enabling restore. Native/old/unknown uncertainty + survives window close, PID death and age. Explicit confirmed recovery removes + only the exact stale presence marker, preserving checkpoints and history. +- **Checkpoints never touch the workspace's `.git`.** They live in a shadow + repository under the extension profile's canonical-root global storage, run with hooks, + fsmonitor, your git configuration and the workspace's filters all off, + and copy bytes as they are. A folder that is not a repository, and a + repository with no commits, get checkpoints too. +- **Ignored files the turn itself touched.** A file the Model API's edit + and write tools (or the image tools) are about to change is copied first, + so a restore brings it back. Ignored files a shell command created are + found by a bounded scan and deleted by a restore. Ones a command changed + are listed as not restorable; the restore never claims to have undone + them. +- **Limits and cleanup.** A file over 16 MiB, a link, a folder link or + junction and a nested repository are left out and named, and a workspace + with more than 50,000 files outside its ignore rules gets no checkpoints; + the panel says why. Archiving a conversation deletes its checkpoints (in + Restricted Mode its records at once, its copies once the folder is + trusted), and retention keeps the 100 + newest checkpoints and 20 redo records per conversation for 50 + conversations, within `museSpark.cleanupPeriodDays`, applied each time the + window opens. The checkpoint folder is 0700 on macOS and Linux. +- **`museSpark.turnCheckpoints`** (machine-scoped, a Preview, off by + default) turns them on. They are off in Restricted Mode, where the extension runs no git, + and without git on `PATH`; the menu says which. With Muse Code on + Windows, which cannot fork, the menu offers **Restore files to here** and + says why the conversation rewind is missing. +- **A restore keeps more of what you did since** (the third Codex review of + PR #55). A restore or Redo leaves a file whose execute bit you changed + since the turn, even with the same bytes. A subagent's turn that outlives + its parent's no longer makes the parent's own edits read as changed + outside the turns. A conversation's turns are ordered by their own count, + not the clock, so two turns in one millisecond or a clock set back never + pull an earlier turn into a restore. A repository an ignore rule hides is + left out whole like any nested repository, and no tool write inside a + nested repository is restored. The repository's `info/exclude` and your + global excludes file are read again before every capture. Switching + `museSpark.turnCheckpoints` off during a turn still keeps the tools' + copies for that turn. On a case-sensitive macOS volume, a link that + differs from its target only in letter case is refused. A window's first + message is no longer sometimes refused ("could not tell other windows …") + when the checkpoint repository was still being set up. A file you save in + the editor while a turn runs is yours: restoring that turn leaves it as + you saved it. An ignored file a tool copied but never changed (its write + was refused) is left alone, not rewritten. Unarchiving a conversation + keeps its new checkpoints even if the clock was set back. Checkpoint + storage that a link or junction puts inside the workspace is refused. + What the extension writes for you while a turn runs (Create AGENTS.md, a + Markdown export, a saved plan, a Revert, a note the Memory view creates or + trashes) is yours too. A restore leaves alone what an earlier turn of the + same conversation, still running in another window, changed, and what a + turn of a window that closed mid-turn may have changed. A file you save in + any window on the folder while a turn runs, even one with no turn of its + own and even while the turn's first checkpoint is being taken, is yours + too. + +- **Rewind code to here asks first** (M72), in the same confirmation as a + file restore. **Fork conversation and rewind code** is now one action: + the confirmation, the reverts, then the fork; declining does neither. + - **Every paid use asks first, in a popup** (M58, PLAN.md D48): **Allow once**, **Allow always in this workspace**, or **Deny**, in every permission mode, Bypass included. It covers each image (on either @@ -36,9 +503,197 @@ happened, not what was planned; superseded entries are kept. the activation bundle (`scripts/check-bundle-split.mjs`). The host-globals and third-party-notices checks cover the new bundle, `npm run cycles` follows it, and the `.vsix` ships it. +- **Build: the ACP agent's budget is 850 KiB** (PLAN.md D6). `dist/acp.js` + is 713.2 KiB now that it loads the Model API backend from + `dist/modelApi.js` (it was 874.1 KiB, over its 800 KiB budget, before + M57 reached it); the budget is that plus about 15 %. The agent is + installed once and never loaded by VS Code, so the size is a download, + not a start-up cost. No other budget changed. +- **Development: test infrastructure.** On macOS and Linux the integration + tests use a short user-data folder under the temporary folder only when the + default would not fit a Unix socket path (macOS caps it at 104 bytes; + `scripts/lib/vscodeTestProfile.mjs`), so a rig's long checkout path no longer + stops VS Code with `listen EINVAL`; a path that fits, CI's included, and + Windows are untouched. Vitest's macOS worker cap is typed so that + `npm run typecheck:host` passes (it failed on TS2769). ### Fixed +- **A very long log line no longer stalls the extension.** Hiding + credentials in a log line took time that grew with the square of a long + run of dotted or dashed words with no URL in it (3.5 s for 40,000 + characters); it is now linear. +- **Memory notes keep restore copies, and the Memory view cannot overlap + another window's restore** (M72). The Model API's memory tools and the + Memory view now copy an ignored project note and its `MEMORY.md` before + they change them, a new note's creation takes the same copy first, and a + note is copied before the view moves it to the trash, so a checkpoint + restore brings them back. The view's new note, delete and index line hold + the restore lease until they finish (personal notes outside the workspace + take none), and so does a conversation export you save inside the + workspace, whichever way the folder is spelled (a link, a junction or a + mapped drive included): another window's Restore or Redo is refused while + one runs. + +- **A workspace opened through a link, a junction or a mapped drive keeps its + conversations** (M72). Sessions, the CLI’s working directory and memory are + keyed by the folder as VS Code spells it, as before; only the checkpoint store + uses the canonical path. + +- **A restore no longer reverts another window’s edit made at the same + instant** (M72). Turns in two windows that touch at one clock tick count as + overlapping, so the other window’s edit is protected. + +- **A file named `..something` is inside the workspace** (M72, and the ACP + agent’s mention links). It was treated as outside because its name begins with + two dots, so it got no restore copy. + +- **A completed restore keeps its Redo when its lease cannot be released** + (M72). The release is tried again and logged instead of replacing the result + with a failure; the window’s next restore takes over a lease it still holds. + +- **Restore and rewind stops when any file is left behind** (M72). A file with no + earlier copy, one that was never in the checkpoint, or ignored files that + could not all be put back now keep the conversation from being rewound, as the + confirmation says. + +- **The checkpoint storage is never the model’s to edit** (M72). Checkpoints + refuse a workspace that holds their storage (a profile folder opened as a + workspace), and the tools refuse every file inside the checkpoint storage, so + no file the model writes can change the repository a turn’s end runs git on. + +- **A checkpoint copy never takes a file from outside the workspace** (M72), even + when a folder was replaced by a link after the tool’s check. + +- **Fork conversation and rewind code stops when an edit cannot be reverted** + (M72), instead of forking away from the history the code still matches. + +- **A check that never started is reported as not run** (M68/M72), not as a + failed command with hooks around it. + +- **A native start that a restore refuses no longer blocks restores** (M72); + nothing started, so the window is not marked unsafe. + +- **A file a turn made visible to git is put back from the copy kept** (M72), + when a `.gitignore` change made an ignored file show up as new. + +- **The model picker lists the models again after a new, resumed or forked + conversation** (0.9.1 regression). Every conversation change threw away the + backend’s model list, so the picker showed nothing to choose (only the pill’s + current model) until the next message, and the context meter lost the model’s + window. The list now belongs to the backend: only a backend that stops or exits, + or a sign-in change, clears it. + +- **A `!` command is checked again at its real start** (M72). On the Model + API backend, a `!` command you typed could still start after the workspace + lost trust, after your Stop, or while the window was closing, when a + checkpoint safety step was waiting in between. It now checks again just + before it starts. A command refused there says "The command did not run", + and the agent is told nothing about a command that never ran (one that did + run is still told). + +- **A shell that could not start no longer blocks Restore.** A missing + PowerShell or bash, or a command the operating system refuses to start (a + NUL byte in it, a command line past the system's length limit), no longer + leaves checkpoint Restore and Redo closed as if a command had run: nothing + launched, so nothing can still be running. A command that did launch keeps + the existing safety rule. + +- **Turn checkpoints work with long Windows paths.** Git refuses a + repository path past its own limit however `core.longpaths` is set, so a + long extension storage path could stop the first checkpoint. The + repository is now made under a short name and moved into place in one + step, and long paths are given to git in the spelling it accepts. A path + git cannot use at all (the checkpoint folder over 240 characters, or a + workspace over 258, on Windows) now says so ("the path of the workspace + or of this extension's storage folder is too long for git", in all 14 + translated languages, machine-made) instead of "the checkpoint failed". + A failed, cancelled or racing first start no longer leaves a half-made + repository in the storage folder. + +- **Checkpoint bundle budget** (M72). The checkpoint store and legacy reader + ship as `dist/checkpointStore.js`, loaded synchronously at the existing + activation construction point with the installed language. Startup safety, + maintenance, activity marks and disposal remain unchanged. A missing or + malformed store module refuses startup instead of offering unsafe work. + +- **Stop across checkpoint waits** (M68/M72). Ordinary writes, first rename + publication and late formatter writes retain captured owner admission + through preimage and atomic waits. Refused late formatting preserves the + completed edit and patch. Revoked diagnostic data is withheld, and check + commands recheck admission after checkpoint marks and native preparation; + a proven refusal starts no process and leaves no unknown activity mark. + Sequential and explicit checks retain their original admission and refusal + reasons; owned background commands preserve their separate Stop controls. + Memory reads and note/index writes retain the same original Stop, + permission-mode, trust and conversation lifetime through their waits. + Already published notes remain when a later index update is refused, + with the existing index warning. + +- Plan actions recheck current workspace trust and disposal after lookup + and confirmation; a plan may still be saved after a conversation change, + while its stale implementation is refused. No-clobber writes check their + canonical directory before mkdir and their owned stage before publishing + or cleanup. Stale plan stages that were replaced, moved or refreshed are + retained. + +- **Android/Termux regression coverage** (PR #51). Tests preserve PATH and + home-directory Muse launcher discovery, XDG credential paths, and explicit + unavailability of bundled dictation/recording helpers on Android. This is + simulated platform coverage; no Android device support claim is added. +- **Web fetch rechecks permission before every address attempt.** A permission + withdrawal while the first connection waits or fails prevents fallback + connections, stops outstanding attempts and closes late answers. +- **Standalone page-fetch proxy warnings distinguish Node transports.** + Node 24.0–24.4 can proxy Meta's `fetch` requests while HTTPS page requests + still go directly; startup now warns about that gap. The ACP package includes + the page-converter worker and its notices, and documents pinned-IP `NO_PROXY` + matching. +- **Stop still prevents a rename's first write during its final file check.** + A cancellation after approval is checked again after the awaited recheck. + Once a write starts, the remaining rename keeps its existing completion path. +- **ACP paid-grant race tests use canonical temporary paths.** Filesystem + aliases on macOS and Windows no longer leave the test waiting for a rename + under a different name. Native path resolution also expands Windows 8.3 + names. The production grant storage and timeout stay intact. +- **ACP sessions keep the newest owner while cancellation finishes.** A + concurrent reload or close waits for the old turn to stop; an older delayed + resume cannot replace the newest request. A backend exit while a turn starts + fails its prompt without an unobserved rejection terminating the agent. +- **Late ACP answers affect only their owning prompt.** Cancelled or completed + prompts reject late paid-use, approval and question answers; a stale paid + answer cannot install an Allow always grant for a later prompt. +- **ACP paid grants survive independent process updates safely.** Per-feature + revocation generations and generation-specific workspace records replace the + shared JSON map. A stale writer cannot restore revoked permission or replace + a newer explicit grant. Legacy grants ask again; storage that cannot publish + safely remembers nothing and retains only the explicit Allow once use. +- **ACP packaging works from Windows paths containing spaces.** npm runs in + the staging directory with fixed relative arguments. Host documentation now + distinguishes development-extension integration tests from VSIX installation. +- **Web fetch preserves picture fallback images** (M69). HTML conversion + now walks `` instead of discarding it, retaining the fallback + `` and its alt text under the existing safe-source rules. Source + alternatives are not selected or fetched; images inside inert templates + or embedded media remain excluded. +- The Model API verify loop hears about workspace writes before they write + or format, in every live conversation and subagent. A parent's, child's + or sibling's cached check grant cannot silently run a changed script + while its writer waits for formatting. Pending writes survive a new user + message, reach newly opened sessions, and release on failure; checks + that start during a write cannot certify its completed state. Project + memory notes and their index also invalidate checks that name them. +- Symbol renames notify the verify loop before their rechecks, release pending + notices after Stop or failure, and check only the files actually written. + ACP conversations opened through native aliases share workspace notices + without changing their saved-session folder identity. + +- **Windows plan-store test stability** (M79). The complete numeric suffix + range is checked through the existing in-memory file port, with exact + attempts, unchanged occupied files, the last free name and same-plan reuse + at that boundary. This avoids 100 staged file flushes in one test on the + hosted Windows runner; the 5-second timeout and real-file-system + publication, collision, cleanup and confinement tests are retained. - **Signing out of Muse Code finishes, and a signed-out CLI no longer reads as signed in** (PLAN.md D26). - **The cause.** `muse logout` rewrites the CLI's `auth.json` with no diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index d89337d9c..3bc9df986 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -26,10 +26,14 @@ Use this order for a candidate branch: 1. Integrate the planned milestones onto the current `main` in order. Resolve conflicts and stage the candidate with no unstaged changes. Record its `git write-tree` hash. -2. Run `npm ci` and `npm run quality` on that exact staged tree. For Windows - behavior, exercise the same tree on Windows (the maintainer uses a - Windows 11 host and VM in parallel) and record the results; collect other - platform evidence where needed. +2. Run `npm ci` and `npm run quality` on that exact staged tree. Before + pushing, test the same candidate on the maintainer's Mac mini, Kubuntu VM + and Windows 11 VM as well as the Windows host. Use isolated checkouts; + record the tree hash, OS and tool versions, commands and process exit + codes. Include full platform gates and the affected real filesystem, + process and packaging paths. Temporary-directory aliases and Windows + short names are part of those paths. An unavailable rig remains a named + blocker; hosted CI does not substitute for the missing local proof. 3. Have an independent agent review the staged diff and acceptance evidence. Scan the staged changes for secrets too: local `security:secrets` scans committed history, so it cannot see the index before commit. Fix findings, @@ -186,6 +190,15 @@ stand when it runs, never a copy taken at activation. Tests never need a proxy or a certificate: they use the failure shapes Node 24 was seen to throw (`docs/certification/m56.md`). +Web fetch (PLAN.md M69) is the one exception to `fetch`: it must connect to +the address it checked, which VS Code's patched `fetch` cannot do, so it +uses Node's `https` (`src/host/web/pinnedRequest.ts`), which VS Code patches +for its proxy and certificates too. Keep every destination check in +`src/core/web/` and test it over the fake resolver and transport in +`test/unit/webFetch.test.ts`; unit tests never reach the internet. What +only VS Code can show (the proxy asked for the pinned address) is in +`test/integration/webFetch.test.ts`, against a loopback proxy. + ## Licence By contributing you agree that your contribution is licensed under the MIT diff --git a/PLAN.md b/PLAN.md index 8a27360b2..da8cb7e61 100644 --- a/PLAN.md +++ b/PLAN.md @@ -12,18 +12,18 @@ Status legend: `[ ]` todo · `[~]` in progress · `[x]` done · `[-]` deferred. ## 1. Assumptions -| # | Assumption | Why | Reversal cost | -| --- | ------------------------------------------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------- | -| A1 | Stack: TypeScript, Node ≥ 22, npm 11, esbuild bundling, React 19 webview. | The VS Code extension API is TypeScript-first; esbuild is Microsoft's documented bundler; React is the de-facto webview framework and what the Claude Code extension appears to use. | Medium (webview components are React-specific). | -| A2 | Package manager is npm with `package-lock.json`, exact version pins (`save-exact=true`). | npm 11.19 is installed; pnpm is not. Exact pins make "latest" impossible by accident. | Trivial. | -| A3 | Extension is **unofficial** and must say so in its name, README and marketplace listing. | Meta Model API ToS / AUP forbid implying Meta endorsement; "Muse Code" and "Muse Spark" are Meta trademarks. | None. | -| A4 | Publisher id `RandyNorthrup` (confirmed 2026-09-22 on marketplace.visualstudio.com/manage: existing publisher, one extension already published). | The marketplace shows the publisher as RandyNorthrup; the URL form is lower-case. | None. | -| A5 | License: MIT. | Standard for VS Code extensions; matches `@muse-code/sdk`. | Trivial before first release. | -| A6 | Settings/command namespace `museSpark.*`, view container id `museSpark`. | Mirrors `claudeCode.*` structure users already know. | Low (rename before first release). | -| A7 | CI provider: GitHub Actions. | `gh` CLI is installed; repo will live on GitHub. | Low. | -| A8 | Target VS Code `^1.125.0` (September 2026 stable is 1.138.0; was `^1.134.0` until 0.1.1). | Needs only long-stable APIs (WebviewView, SecretStorage, `vscode.diff`, `env.openExternal`, `window.createTerminal`). `@types/vscode` publishes only some minors; 1.134.0 was taken at first as the oldest recent one on the registry; on 2026-09-22 a clean VS Code 1.130.0 (the owner's Win11 VM) refused 0.1.0, and 1.125.0 typechecks the whole tree, so the floor is 1.125.0 from 0.1.1. | Trivial. | -| A9 | Pre-commit hooks via **husky + lint-staged**, not the Python `pre-commit` tool. | `pre-commit` is not installed; a Node project should not require a Python toolchain to commit. gitleaks is invoked directly from the husky hook. | Low. | -| A10 | **Fully cross-platform (owner requirement 2026-09-22):** Windows, macOS and Linux are all first-class. Windows is the primary dev machine. | CI matrix runs the full gate set on ubuntu, windows and macos; every OS-specific path (binary discovery, process spawning, paths, line endings) has a unit test per platform branch. Meta documents the `muse` CLI for macOS and Windows; Linux users fall back to the Model API backend if the CLI is unavailable there (Q6). | None. | +| # | Assumption | Why | Reversal cost | +| --- | ------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------- | +| A1 | Stack: TypeScript, Node ≥ 22, npm 11, esbuild bundling, React 19 webview. | The VS Code extension API is TypeScript-first; esbuild is Microsoft's documented bundler; React is the de-facto webview framework and what the Claude Code extension appears to use. | Medium (webview components are React-specific). | +| A2 | Package manager is npm with `package-lock.json`, exact version pins (`save-exact=true`). | npm 11.19 is installed; pnpm is not. Exact pins make "latest" impossible by accident. | Trivial. | +| A3 | Extension is **unofficial** and must say so in its name, README and marketplace listing. | Meta Model API ToS / AUP forbid implying Meta endorsement; "Muse Code" and "Muse Spark" are Meta trademarks. | None. | +| A4 | Publisher id `RandyNorthrup` (confirmed 2026-09-22 on marketplace.visualstudio.com/manage: existing publisher, one extension already published). | The marketplace shows the publisher as RandyNorthrup; the URL form is lower-case. | None. | +| A5 | License: MIT. | Standard for VS Code extensions; matches `@muse-code/sdk`. | Trivial before first release. | +| A6 | Settings/command namespace `museSpark.*`, view container id `museSpark`. | Mirrors `claudeCode.*` structure users already know. | Low (rename before first release). | +| A7 | CI provider: GitHub Actions. | `gh` CLI is installed; repo will live on GitHub. | Low. | +| A8 | Target VS Code `^1.99.0` (September 2026 stable is 1.138.0; `^1.134.0` until 0.1.1, `^1.125.0` until M62). | Needs only long-stable APIs. M62's audit: the host typechecks against every `@types/vscode` from 1.85 on, and 1.99 is the first release on Node 20.18 and Chromium 132, which the host and webview bundles target. Tested in VSCodium 1.99.3 and code-server 4.99.4 (VS Code 1.99.3), which refused the 1.125 floor (`docs/certification/m62.md`). | Trivial. | +| A9 | Pre-commit hooks via **husky + lint-staged**, not the Python `pre-commit` tool. | `pre-commit` is not installed; a Node project should not require a Python toolchain to commit. gitleaks is invoked directly from the husky hook. | Low. | +| A10 | **Fully cross-platform (owner requirement 2026-09-22):** Windows, macOS and Linux are all first-class. Windows is the primary dev machine. | CI matrix runs the full gate set on ubuntu, windows and macos; every OS-specific path (binary discovery, process spawning, paths, line endings) has a unit test per platform branch. Meta documents the `muse` CLI for macOS and Windows; Linux users fall back to the Model API backend if the CLI is unavailable there (Q6). | None. | ## 2. Resolved decisions @@ -127,35 +127,39 @@ tested on chunk splits inside frames and inside multi-byte characters. ### D3 — Quality toolchain versions (verified against the npm registry 2026-09-21) -| Package | Pinned | Why this version | -| ----------------------------------------------------------------------- | --------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `typescript` | **6.0.3** | Registry latest is 7.0.2, but `typescript-eslint@8.70.1` declares `peerDependencies.typescript: ">=4.8.4 <6.1.0"`. TS 7 installs cleanly and silently disables every type-aware lint rule. 6.0.3 is the highest release inside the supported range. | -| `eslint` | 10.11.0 | `typescript-eslint` accepts `^10.0.0`; `eslint-plugin-unicorn@76` requires `>=10.4`. | -| `typescript-eslint` | 8.70.1 | Latest; `strictTypeChecked` + `stylisticTypeChecked`. | -| `@eslint/js` | 10.0.1 | Separate package from eslint; required by the flat config. | -| `eslint-plugin-unicorn` | 76.0.0 | Latest; needs eslint ≥ 10.4 (satisfied). | -| `eslint-plugin-react-hooks` | 7.1.1 | Declares eslint `^10.0.0`. `eslint-plugin-react` (7.37.5) and `eslint-plugin-jsx-a11y` (6.10.2) only declare up to eslint `^9`, so they are **not** installed; a11y is covered by manual checks in visual certification and revisited when the plugins add eslint 10 peers. | -| `dpdm` | 4.3.0 | Circular-import gate (`--exit-code circular:1`). `madge` is incompatible with TS 6+. `eslint-plugin-import-x` was considered and dropped: its `no-cycle` rule is known not to fire, and unresolved imports are already a hard `tsc` error (TS2307) in every project here. | -| `knip` | 6.37.0 | Unused files/exports/deps. Config is `knip.jsonc` (knip 6 rejects `"//"` pseudo-comments). Run without `--strict`: strict implies production mode, which needs `!`-suffixed entries and otherwise analyses nothing. | -| `prettier` | 3.9.8 | Formatter. | -| `stylelint` + `stylelint-config-standard` | 17.15.0 / 40.0.0 | Webview CSS gate (`--max-warnings=0`). | -| `vitest` + `@vitest/coverage-v8` | 5.0.1 | Unit tests (node env for extension code, jsdom for webview). Peer `@types/node ^22 | | >=24` satisfied. | -| `jsdom` | 30.1.0 | Webview component tests. | -| `@testing-library/react` / `dom` / `jest-dom` | 16.3.3 / 10.4.2 / 7.0.1 | Component assertions. | -| `@vscode/test-cli` + `@vscode/test-electron` + `mocha` + `@types/mocha` | 0.0.15 / 3.1.0 / 12.0.2 / 10.0.10 | Integration tests inside the Extension Development Host. `@vscode/test-electron` is an unlisted peer of test-cli, so knip ignores it explicitly. | -| `esbuild` | 0.28.2 | Bundles extension (cjs, node platform) and webview (esm/iife, browser platform). | -| `@types/vscode` | 1.125.0 | Matches `engines.vscode` (test/unit/manifest.test.ts enforces the pairing). | -| `@types/vscode-webview` | 1.57.5 | Types for `acquireVsCodeApi()` inside the webview. | -| `@types/node` | 22.20.4 | Extension host on VS Code 1.138 is Electron 42 (Node ≥ 22). Typing against 22 keeps code portable to older hosts. | -| `@vscode/vsce` | 4.0.0 | Packaging. Needs Node ≥ 22. | -| `react` / `react-dom` / `@types/react` / `@types/react-dom` | 19.3.0 | Webview UI. | -| `zod` | 4.6.5 | Runtime validation of every webview ⇄ extension message and every HTTP/MSP boundary. | -| `@muse-code/sdk` | 1.3.0 | Official MSP client. Developer Preview: "minor releases may alter APIs before 1.0" → exact pin, adapter isolated in one module, schema fingerprint checked at handshake. Installed with a one-off `--min-release-age=0` on 2026-09-22 (published 2026-09-18, inside the 7-day window); the lockfile pins it so `npm ci` is unaffected. Only its `Connection`/`spawnMspConnection` surface is used; `Connection.onNotification` holds a single handler, so the facade (`MuseClient`) is not composed. | -| `husky` / `lint-staged` | 9.1.7 / 17.5.1 | Pre-commit gates. | -| `jscpd` | 5.3.1 | Copy-paste detection. | -| `npm-run-all2` | 9.0.3 | Runs gate scripts in sequence/parallel. | -| `rimraf` | 6.1.3 | Cross-platform clean. | -| `axe-core` | 4.13.0 | The accessibility gate (M37, D32): WCAG 2.0 to 2.2, levels A and AA, run inside the harness page. MPL-2.0; a dev dependency, never bundled. | +| Package | Pinned | Why this version | +| ---------------------------------------------------------------------------------------------------- | --------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `typescript` | **6.0.3** | Registry latest is 7.0.2, but `typescript-eslint@8.70.1` declares `peerDependencies.typescript: ">=4.8.4 <6.1.0"`. TS 7 installs cleanly and silently disables every type-aware lint rule. 6.0.3 is the highest release inside the supported range. | +| `eslint` | 10.11.0 | `typescript-eslint` accepts `^10.0.0`; `eslint-plugin-unicorn@76` requires `>=10.4`. | +| `typescript-eslint` | 8.70.1 | Latest; `strictTypeChecked` + `stylisticTypeChecked`. | +| `@eslint/js` | 10.0.1 | Separate package from eslint; required by the flat config. | +| `eslint-plugin-unicorn` | 76.0.0 | Latest; needs eslint ≥ 10.4 (satisfied). | +| `eslint-plugin-react-hooks` | 7.1.1 | Declares eslint `^10.0.0`. `eslint-plugin-react` (7.37.5) and `eslint-plugin-jsx-a11y` (6.10.2) only declare up to eslint `^9`, so they are **not** installed; a11y is covered by manual checks in visual certification and revisited when the plugins add eslint 10 peers. | +| `dpdm` | 4.3.0 | Circular-import gate (`--exit-code circular:1`). `madge` is incompatible with TS 6+. `eslint-plugin-import-x` was considered and dropped: its `no-cycle` rule is known not to fire, and unresolved imports are already a hard `tsc` error (TS2307) in every project here. | +| `knip` | 6.37.0 | Unused files/exports/deps. Config is `knip.jsonc` (knip 6 rejects `"//"` pseudo-comments). Run without `--strict`: strict implies production mode, which needs `!`-suffixed entries and otherwise analyses nothing. | +| `prettier` | 3.9.8 | Formatter. | +| `stylelint` + `stylelint-config-standard` | 17.15.0 / 40.0.0 | Webview CSS gate (`--max-warnings=0`). | +| `vitest` + `@vitest/coverage-v8` | 5.0.1 | Unit tests (node env for extension code, jsdom for webview). Peer `@types/node ^22 | | >=24` satisfied. | +| `jsdom` | 30.1.0 | Webview component tests. | +| `@testing-library/react` / `dom` / `jest-dom` | 16.3.3 / 10.4.2 / 7.0.1 | Component assertions. | +| `@vscode/test-cli` + `@vscode/test-electron` + `mocha` + `@types/mocha` | 0.0.15 / 3.1.0 / 12.0.2 / 10.0.10 | Integration tests inside the Extension Development Host. `@vscode/test-electron` is an unlisted peer of test-cli, so knip ignores it explicitly. | +| `esbuild` | 0.28.2 | Bundles extension (cjs, node platform) and webview (esm/iife, browser platform). | +| `@types/vscode` | 1.99.0 | Matches `engines.vscode` (test/unit/manifest.test.ts enforces the pairing). | +| `@types/vscode-webview` | 1.57.5 | Types for `acquireVsCodeApi()` inside the webview. | +| `@types/node` | 22.20.4 | Extension host on VS Code 1.138 is Electron 42 (Node ≥ 22). Typing against 22 keeps code portable to older hosts. | +| `@vscode/vsce` | 4.0.0 | Packaging. Needs Node ≥ 22. | +| `react` / `react-dom` / `@types/react` / `@types/react-dom` | 19.3.0 | Webview UI. | +| `zod` | 4.6.5 | Runtime validation of every webview ⇄ extension message and every HTTP/MSP boundary. | +| `@muse-code/sdk` | 1.3.0 | Official MSP client. Developer Preview: "minor releases may alter APIs before 1.0" → exact pin, adapter isolated in one module, schema fingerprint checked at handshake. Installed with a one-off `--min-release-age=0` on 2026-09-22 (published 2026-09-18, inside the 7-day window); the lockfile pins it so `npm ci` is unaffected. Only its `Connection`/`spawnMspConnection` surface is used; `Connection.onNotification` holds a single handler, so the facade (`MuseClient`) is not composed. | +| `husky` / `lint-staged` | 9.1.7 / 17.5.1 | Pre-commit gates. | +| `jscpd` | 5.3.1 | Copy-paste detection. | +| `npm-run-all2` | 9.0.3 | Runs gate scripts in sequence/parallel. | +| `rimraf` | 6.1.3 | Cross-platform clean. | +| `axe-core` | 4.13.0 | The accessibility gate (M37, D32): WCAG 2.0 to 2.2, levels A and AA, run inside the harness page. MPL-2.0; a dev dependency, never bundled. | +| `parse5` | 8.0.1 | M69 (D49): web fetch parses a page with the HTML standard's own parsing algorithm (the implementation jsdom uses), after four review rounds found a hand-written tokenizer short of it. MIT; one dependency, `entities` ^8 (BSD-2-Clause, 8.1.0 locked, formerly our direct dependency); no peers; published 2026-04-19; already in the lockfile through jsdom and @vscode/vsce; `npm audit` clean. Quadratic on hostile nesting (40,000 nested lists in 73 s), so it runs only in `dist/pageWorker.js`, a worker per page (at most two at once) stopped at 10 s or 512 MiB (D6). `entities` 8 says Node ≥ 20.19 for `require(esm)`; bundled, the worker ran on Node 20.18.3. | +| `html-encoding-sniffer` | 6.0.0 | M69: the HTML standard's encoding sniffing (byte order mark, the Content-Type charset, the 1,024-byte `` prescan) for web fetch's HTML, as jsdom uses it. MIT; one dependency, `@exodus/bytes` ^1.6 (MIT; 1.15.2 locked through jsdom, published 2026-09-21, inside the 7-day window: the lockfile pins it, as for `@muse-code/sdk`; its optional `@noble/hashes` peer is not used by the `encoding-lite` entry the sniffer imports); published 2025-12-26; `npm audit` clean. Says Node ≥ 20.19 (ESM); bundled into `dist/pageWorker.js` only and run on Node 20.18.3. Ships no types: `src/core/web/html-encoding-sniffer.d.ts`. | +| `playwright-core` | 1.63.0 | The host checks' browser driver (hosts.yml, M62): code-server, Theia and JupyterLab driven in Chrome. Apache-2.0; a dev dependency, never bundled; it uses the installed Chrome, never downloads one. | +| `mdast-util-from-markdown` / `micromark-extension-gfm` / `mdast-util-gfm` / `mdast-util-to-markdown` | 2.0.3 / 3.0.0 / 3.1.0 / 2.1.2 | M79: the host reads a plan with the panel's own Markdown grammar (what react-markdown 10.1.0 and remark-gfm 4.0.1 resolve to; no peers; 0 advisories). Its own lazily loaded bundle, `dist/planMarkdown.js` (139.0 KiB with the brief writer, `character-entities` among it), so `dist/extension.js` carries none of it (464.8 KiB after merging `main`, D6). | Deprecated and avoided: `@vscode/webview-ui-toolkit` (archived; npm marks it deprecated). Webview controls are hand-built on VS Code CSS theme variables. @@ -207,13 +211,17 @@ quality`) and as a CI job. ### D6 — Bundle budgets (Phase 6) -| Artifact | Budget (minified, uncompressed) | -| ---------------------- | ---------------------------------------------------------------------------------------------------- | -| `dist/extension.js` | ≤ 600 KiB (the M7 Model API client fit without raising it; the activation bundle since M57) | -| `dist/modelApi.js` | ≤ 400 KiB (M57: the Model API backend, loaded when it first starts; 295.6 KiB when split, see below) | -| `dist/searchWorker.js` | ≤ 50 KiB | -| `dist/webview/main.js` | ≤ 900 KiB including React, the markdown renderer and highlight.js (one bundle) | -| `.vsix` | not gated; 0.8.0 is 905,941 bytes (the GitHub Release asset, §10) | +| Artifact | Budget (minified, uncompressed) | +| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | +| `dist/extension.js` | ≤ 600 KiB (the M7 Model API client fit without raising it; the activation bundle since M57) | +| `dist/modelApi.js` | ≤ 400 KiB (M57: the Model API backend, loaded when it first starts; 295.6 KiB when split, see below) | +| `dist/searchWorker.js` | ≤ 50 KiB | +| `dist/pageWorker.js` | ≤ 300 KiB (M69: web fetch's page converter, parse5 and its parts, on a worker started for each page; 212.3 KiB when split) | +| `dist/webview/main.js` | ≤ 900 KiB including React, the markdown renderer and highlight.js (one bundle) | +| `.vsix` | not gated; 0.8.0 is 905,941 bytes (the GitHub Release asset, §10) | +| `dist/acp.js` | ≤ 850 KiB (the ACP agent, installed once, never loaded by VS Code; 713.2 KiB when set, see below) | +| `dist/planMarkdown.js` | ≤ 150 KiB (M79: the plan reader, the panel's Markdown parser, loaded on the first plan action; 139.0 KiB with the brief writer) | +| `dist/checkpointStore.js` | ≤ 225 KiB (M72: synchronous checkpoint factory and legacy reader; measured 187.0 KiB plus 15%, rounded up to 25 KiB) | `npm run build` prints sizes; `scripts/check-bundle-size.mjs` holds the numbers and fails the build over budget or when a bundle is missing. This table mirrors @@ -270,6 +278,83 @@ instead. stops carrying one of them, or when a file of the folder is on neither the lazy list nor the allowed list above. +**Amendment (M79, 2026-09-28): the plan reader is a bundle of its own.** +Reading a plan with the panel's own Markdown grammar (PR #53 review) takes +`mdast-util-from-markdown`, `micromark-extension-gfm` and `mdast-util-gfm`: +114.5 KiB, which would have taken `dist/extension.js` from 448.7 to +563.2 KiB with several milestones' host code still to land. So +`src/core/plans/planMarkdown.ts` is built from `src/host/planMarkdownEntry.ts` +into `dist/planMarkdown.js` (114.7 KiB, budget 150 KiB) and required by +`planMarkdownLoader` on the first Save plan, Implement or Plans…; +`dist/extension.js` is 449.7 KiB. The reader imports no value of +`shared/constants` (that would bring the English table, 58.7 KiB): it +lists, and `planDocument.ts` cuts and caps. There is no fallback without +it: a plan action that cannot load it is refused with +`planMarkdownUnavailable`, since it is what checks for hidden text. The +split gate fails when `dist/extension.js` carries the reader's module, its +entry or any file of the parser's packages (`micromark*`, `mdast-util-*`, +`character-entities`, `decode-named-character-reference`), or when +`dist/planMarkdown.js` stops carrying the reader. Since PR #53's third review +the reader also writes the brief (`mdast-util-to-markdown`, the version +remark-gfm's writer resolves to): 139.0 KiB. +**Amendment (PR #32 joined with M57, 2026-09-27): the ACP agent loads the +same `dist/modelApi.js`.** The agent's runtime (`src/runtime/backends.ts`, +D62) builds a `ModelApiBackendManager` per folder, which since M57 needs a +bundle to require. Two ways were open: the agent requires `dist/modelApi.js` +beside `dist/acp.js`, as the extension requires it beside `extension.js`, or +the entry module is bundled into `acp.js` and handed over with `loadBundle`. +The first was taken: + +- **One build of the backend for both packages.** The file the `.vsix` ships + is the file the agent's npm package ships (`scripts/package-acp.mjs` copies + it with `acp.js` and the search worker); bundling the entry into `acp.js` + would have been a second build of the same 20 files, 300 KiB of it. +- **The manager's real path in the agent.** `loadBundle` stays what M57 made + it, a test's way to hand in the source module; the agent exercises the + `require` by path, the missing-file message and the log line, which the + extension otherwise reaches only on the Model API backend. +- **A Muse Code agent never loads the backend**, as a Muse Code user of the + extension never does. +- **Held by the gates.** `check-bundle-split.mjs` reads the agent's metafile + (`dist/meta-acp/acp.json`) too and fails when `dist/acp.js` carries a lazy + file or the entry, so the agent cannot quietly grow a second copy; + `check-host-api.mjs` lists `modelApiEntry.ts` as portable, so the bundle + the agent loads never reaches `vscode`; the agent's third-party notices + take `modelApi.js`'s packages; the stdio e2e suite (and hosts.yml, against + the installed package on three platforms) runs a Model API turn through + the package's own `dist/modelApi.js`. The identity audit of M57 applies to + the agent as it does to the extension: `src/acp` recognises a settled + prompt with `isPromptSettledError`, never `instanceof`. +- **Measured**: `dist/acp.js` went from 874.1 KiB (the joined tree before + M57) to 713.2 KiB; `dist/extension.js` is 432.5 KiB and `dist/modelApi.js` + 299.5 KiB, each under its budget. + +**Amendment (PR #32 joined with M57, 2026-09-27): the ACP agent's budget is +850 KiB.** Set at M63 to 800 KiB against 718 KiB, and outgrown when M48–M56 +reached the agent through the shared managers (874.1 KiB, PLAN.md §7 then). +With the backend in `dist/modelApi.js` the agent is 713.2 KiB; the budget is +that plus about 15 %, rounded up to a multiple of 50 KiB (820 → 850). No +other budget changes. + +- **What it holds** (the production metafile, `dist/meta-acp/acp.json`): + zod 445.2 KiB (62 %), of which 257.6 KiB are its 64 error-message locales, + exported by the classic API `@agentclientprotocol/sdk` imports and so kept + by esbuild although the agent never switches locale; zod's core and + classic API 185.0 KiB; the ACP SDK 53.7 KiB; the English table 58.6 KiB; + the engine the agent runs (the Muse Code backend and its SDK, the tool + harness, the conversation's shared code, `src/acp`, `src/runtime`) the + rest, about 155 KiB. Three of the Model API backend's files, all on + M57's allowed list, stay in it. +- **Why it is acceptable.** The agent is a process of its own that the user + installs once with npm and an editor starts; VS Code never loads it, so + its size adds nothing to the extension's activation, and it is parsed + once per agent start. What it costs is the download: 175.1 KiB of + `acp.js` gzipped, and a 597 KiB package (611,034 bytes, with + `dist/modelApi.js`, the search worker, the 14 tables and the notices). +- **Not taken**: dropping zod's locales. They come with the SDK's own + import of classic zod; removing them means aliasing or patching a + dependency's module graph, for a download a user makes once. + ### D7 — Permission modes map onto MSP approval modes; prompting modes wait for M4 `muse --help` (1.3.0) names the CLI's own modes `untrusted | on-request | @@ -564,6 +649,33 @@ the trust flag. On the Model API backend the extension is the host, so it mirrors the conventions above and no others: no invented file names, no `MUSE.md`. +**Plans (M79, 2026-09-27).** The saved-plan location is Muse Code's own, +not the extension's. It was read from `skills/plan/SKILL.md`, the `plan` +skill bundled in `muse-bin-1.4.0-R4302.1.exe`, under "Explicit File Output": + +- "If saving and no stronger convention exists, save to + `.agents/plans/YYYY-MM-DD-.md`"; +- "When creating a new dated file and the chosen file exists, add a short + numeric suffix"; +- "If you save a plan file, its content must be exactly the canonical + body". + +The extension follows all three. It does not follow the skill's +precedence for a stronger convention (an active `specs//plan.md`, +a `docs/plans/` folder), which is the model's judgement, not a fixed +name. The skill's delivery form is the other half: + +- a plan reply starts with "This is a plan, not a special mode; I haven’t + started implementation. Reply `go` to execute this plan, or tell me what + to change."; +- it ends with the second sentence repeated; +- the user's next message ("go") is the approval. + +A live capture of a Plan-mode turn on Muse Code 1.4.0 showed exactly that +reply as an ordinary `agentMessage`, with no plan item and no approval +request (docs/certification/m79.md). The extension saves the text between +those two lines; any other reply is saved whole. + ### D14 — Production hardening set for 0.2.0 (2026-09-22) The owner's brief after D13: "fully enterprise grade and production ready @@ -2086,6 +2198,15 @@ status`, and the Model API's prompt cache. What was read or captured first by default). The shipped 1.139.0 extension host carries the same code. So the Model API client and the Muse Voice socket need no proxy client of their own and no dependency. + **Amended 2026-09-27, with M62's 1.99 floor** (`docs/certification/m62.md`, + "The floor after M48–M56"): read again at the tags from 1.99.0 to 1.125.0, + `fetch` is patched at every one (1.99.0: `patchGlobalFetch`, proxy-agent + ^0.32.0), but `WebSocket` only from **1.112.0** (proxy-agent ^0.39.1; the + `http.webSocketAdditionalSupport` setting appears there). On 1.101 to + 1.111 Muse Voice's socket therefore goes to Meta without VS Code's proxy + and certificate handling, and on 1.99 and 1.100 there is no WebSocket at + all. Diagnostics says per global whether the editor routes it (the + global VS Code sets beside each patch), and the README says what to use. - **Node 24.20** reads `NODE_EXTRA_CA_CERTS` when the process starts, for every TLS connection; `--use-system-ca` and `NODE_USE_ENV_PROXY` are process switches an extension cannot set. Its `fetch` throws `TypeError: @@ -2567,9 +2688,10 @@ both backends comes before what serves one. files and tool output are data, never instructions. They are marked as untrusted where the model receives them, and nothing in them can raise a permission, pick a model or skip a question. A conversation built on - such content (an imported session, a PR someone else wrote) starts in a - mode that asks, whatever `museSpark.initialPermissionMode` says, and - only the user's own action relaxes it. + such content (an imported session, a PR someone else wrote, a plan file + picked from Plans… in M79) starts in a mode that asks, whatever + `museSpark.initialPermissionMode` says, and only the user's own action + relaxes it. - **Automatic actions follow the mode.** Anything the extension runs on its own (checks after an edit, a memory flush, a review turn) takes the same path as the call it stands for: a shell command asks wherever the @@ -2728,19 +2850,419 @@ only the extension-side uses reach it. a documented weak spot, so at most a warning signal. - Any use as the coding model. +### D51 — Turn checkpoints live in a shadow repository (M72, 2026-09-28) + +M72 captures the workspace's files at turn boundaries on both backends, in +a git repository or not. Destructive stored restoration currently requires +a connected Model API session and confirmed process safety; native captures +remain read-only until pre-edit/full shutdown exclusion is proved. The plan +review of PR #50 and the security review added three +constraints: nothing may land in the workspace's `.git` (a `push --mirror` +would carry an untracked or ignored secret such as `.env`); the git that +takes checkpoints runs with hooks and fsmonitor off and none of the user's +configuration or the workspace's filters; and ignored files are covered +only as far as the extension saw the change coming. + +**Considered: hidden refs in the workspace repository.** Trees built with +`write-tree` on a temporary index (`GIT_INDEX_FILE`) and kept by refs under +`refs/muse-spark/checkpoints/…`. It shares the user's objects and stat +cache, but it writes objects and refs into the user's `.git`, where +`push --mirror`, `for-each-ref` and `log --all` see them, and it runs under +the user's repository config and attributes (autocrlf, LFS and other clean +and smudge filters). Refused on both counts. + +**Taken: a shadow repository in the extension's own storage.** The resumed +M72 implementation uses `/checkpoints//shadow.git` +for current-version windows, preserving prior workspace-specific stores as +read-only history. It is a bare repository whose work +tree is the workspace (or the repository top when the workspace is a folder +of one, so its `.gitignore` files apply, with every path kept to the +workspace's prefix). + +- **Isolation.** Every command runs with an environment stripped of the + host's `GIT_*` variables, `GIT_CONFIG_NOSYSTEM`, an empty + `GIT_CONFIG_GLOBAL`, a `HOME` in storage, `GIT_ATTR_NOSYSTEM`, + `GIT_OPTIONAL_LOCKS=0`, `core.hooksPath` at an empty folder, + `core.fsmonitor=false` and `core.untrackedCache=false`. The shadow's + `info/attributes`, which outranks every `.gitattributes`, unsets `text`, + `eol`, `filter`, `ident` and `working-tree-encoding`, so no filter runs + and bytes are copied as they are (CRLF files, LFS files, `text=auto`). + git is found by absolute path (D24). Only the workspace's + `info/exclude` and the user's global excludes file are read, as ignore + patterns. +- **Objects.** No alternates: the shadow keeps its own copies, so the + user's `gc` can never break a checkpoint and nothing of ours is written + or freshened in `.git`. A private index keeps stat data, so a capture + hashes only what changed since the last one; the first capture of a + workspace hashes everything outside its ignore rules, within the caps. +- **Records.** Checkpoint and redo JSON is parsed with zod and stored in + a tree with its captured trees and blobs under `refs/muse-spark/record/`. + Creation, updates and deletion use Git compare-and-swap. Each window has + its own index and pending-capture refs. Trees, not commits: no author + identity is needed. +- **Ignored files.** A bounded scan of sizes and times at each turn's start + and end finds what the turn created, changed or deleted; the Model API's + write tools, the image tools and the memory tools (with the Memory view's + delete) copy a file just before they write it (`withCheckpointCopies`, + `createCheckpointedMemory`). A restore deletes created ignored files, + restores copied ones and lists the rest as not restorable; what it + overwrites or deletes is copied for its Redo. +- **Links.** Git for Windows walks into junctions (checked with git + 2.52.0.windows.1), so git's own listing cannot be trusted there: a + capture leaves out every path under a folder link or junction and names + the link, and a restore refuses any path whose canonical form is not the + canonical root plus the path. +- **Restricted Mode.** No checkpoints: the extension runs no git in an + untrusted workspace (D24), and a copy store without git would still read + and duplicate an untrusted tree's files for no gain the user asked for. + The menu says so. Archiving there drops the conversation's records at + once (no git); the next trusted window deletes their refs and copies. +- **Windows.** Independent records use CAS refs, with an index, pins, + tool copies and presence per window. File restores reserve one shared + CAS ref. Recent objects receive a prune grace period; an archive is + durable before returning (M72 Built, "Windows sharing a store"). Git's + path limits are met in the store itself: the repository is made in a short + folder and published by one rename, a long path is named relative, and a + path git cannot open is refused with a message (M72, "Long storage paths"). + +### D60 — Muse Spark Code beyond VS Code: the IDE compatibility program (2026-09-26) + +The owner (2026-09-26) handed over a plan, "Muse Spark Code — IDE +Compatibility Plan" (prepared 2026-09-25 against 0.8.0, `bdaede4`), to be +worked beside M45–M56, which the owner is building in parallel. It is kept +whole in `docs/ide-compatibility.md`: the target matrix with its sources, +the architecture, the ACP plan, the release scenarios. Its links were +not re-read here (this environment's network policy blocks them, +2026-09-26); each target's claim is re-read from its source before its +milestone starts, as M41's installers are. + +- **One product, four families.** The VS Code extension qualified in the + editors built on VS Code (VSCodium, Cursor, Kiro, Positron, Theia, + code-server, Codespaces, Che, Firebase Studio); one agent over the Agent + Client Protocol for the editors that host agents in their own chat (Zed, + JetBrains AI Assistant, Xcode 27, Qt Creator, Neovim, Emacs, Sublime, + Devin Desktop); native plugins that embed the React panel (IntelliJ with + Android Studio, Visual Studio, Eclipse, NetBeans, JupyterLab, Spyder, + RStudio); and a terminal or adjacent interface where a host allows no + more. The name stays "Muse Spark Code (Unofficial)" everywhere (rule + 11), and a port never proposes 1.0 (D44). +- **The engine and the UI are shared, the host is an adapter.** `AgentHost` + and `AgentSession` stay the backend boundary; the webview reaches its + host through one bridge; the host's services (workspace roots, documents + and their versions, selection, edits, diffs, diagnostics, terminals, + settings, secrets, persistence, notifications) become a contract with + VS Code's implementation first. VS Code stays the reference client, and + its behaviour does not change while boundaries move. +- **What already holds.** Only 11 source files import `vscode` (M60's + record, after M61 took the logger's); the React app reaches its host + only through `acquireVsCodeApi`, window messages, the text table + embedded in its HTML and 57 `--vscode-*` theme variables; the protocol + is zod-validated both ways. The work is extraction, not a + rewrite. +- **Rulings carried into every adapter.** The key never reaches a child + process, a launch argument, an environment variable, a webview message + or a general IPC field (rule 8, D1): an ACP or native build signs in + through the CLI's own login first, and its Model API backend waits for + the credential-ownership decision (Q63). Paid features stay opt in and + loud (rule 12): an adapter whose client cannot show the price and ask + first offers none. Muse's approval policy is never loosened because a + protocol allows it; a declined or cancelled request never runs by a + translation default. Editing is claimed for a host only after its + backend-specific tests (the plan's §6.1: dirty buffers, exactly-once + changes, undo) pass there. +- **Support is measured, per host.** Integration type (full Muse + interface, native agent interface, external) and release status + (Planned, Prototype, Preview, Supported) are recorded per editor, + version, backend and OS, with each feature tested, partial, + unavailable or unverified. An install is the first step, not the claim. +- **Nothing is installed or billed unasked.** Another IDE is installed + for a probe, and a dependency (the ACP SDK, a JetBrains or .NET + toolchain) is added, only on the owner's go (Q61, Q62) and with rule 9's + checks. Compatibility runs never use the subscription or a paid feature; + a live check follows CLAUDE.md. +- **Numbered from 60**, so M45–M56 keep their numbers while both are built + (M60–M66, Q60–Q64). Moves across files M45–M56 are changing (the + stylesheet, the controller, the protocol) wait until M56 merges; the + inventory and the narrow seams go first. + +### D61 — The Model API key outside VS Code (2026-09-26) + +The owner (2026-09-26): "you need to figure out the proper api key safe +storage". Inside VS Code the key stays in SecretStorage (rule 8). An agent +that another editor starts (D62), and later the native plugins (M64), run +with no VS Code, so the key needs a home of its own. + +- **Rejected**: the operating system's command-line tools as child + processes (`security`, `secret-tool`, PowerShell with DPAPI): the key + would pass through another process, and `security +add-generic-password -w` takes it as an argument, visible to `ps`. A + file encrypted with a key of our own: only as safe as the file's + permissions, and crypto invented here. An environment variable, as many + CLIs take: it invites the key into editor settings files (Zed's agent + `env`, JetBrains' `acp.json`), which rule 8 forbids. +- **Chosen**: the operating system's credential store, reached in-process + through `@napi-rs/keyring` 2.1.0 (MIT, the Node binding of the + `keyring` Rust crate; prebuilt for Windows x64, arm64 and ia32, macOS + x64 and arm64, Linux x64 and arm64 on glibc and musl, arm, riscv64, + FreeBSD; published 2026-09-13, outside the 7-day window). Windows + Credential Manager (DPAPI, per user), the macOS login Keychain, and on + Linux the Secret Service (GNOME Keyring, KWallet, KeePassXC), pinned + with `linux: { store: 'secret-service' }`: the kernel keyring the + library would otherwise fall back to forgets the key at reboot. Without + a Secret Service the Model API backend says how to get one; there is no + plaintext fallback. +- **One entry per user**: service `Muse Spark Code (Unofficial)`, account + `museSpark.modelApiKey` (the key's name in the extension's + SecretStorage), shared by every editor that runs the agent. A missing + entry reads as `null` from the binding, whatever its typings say (found + against GNOME Keyring at M63); the store turns it into `undefined`. +- **Setting it**: `muse-spark-code-acp auth set` reads the key from the + terminal with echo off (or one line from a pipe), checks its shape, + stores it and prints only that it did; `auth status` says whether a key + is stored, never any of it; `auth clear` removes it. Each ACP client is + offered a terminal sign-in that runs exactly `auth set`, so the key goes + from the keyboard to the store without passing through the editor. +- **Never**: an argument, an environment variable, a file, a log (the + redactor stays), an ACP message, or a child process: not the environment + of `muse serve` (D1), a tool or a check. +- **Credential variables** (the Codex review of `a209130`, 2026-09-28): + a `META_API_KEY` in the agent's own environment is the user's for Muse + Code, as the extension's `muse serve` inherits it (D1's amendment), and + counts as Muse Code's credential there; the agent takes it and every + other credential variable (`*_API_KEY`, the names hooks never get) out + of its own environment at start (`src/runtime/credentialVariables.ts`) + and hands them back to Muse Code's processes only, so a shell command, + a hook, git or a helper never sees one. +- **AGENTS.md rule 8 (amended 2026-09-28)** names this store as + SecretStorage's stand-in outside VS Code (it is the store SecretStorage + itself rests on), filled only through `auth set`'s standard input and + never passed to a child process. Its one named exception is M80's CI + bootstrap: GitHub hands a secret to a step only through its environment + or script, so the Action's step shell is the one environment the key is + ever in; it pipes the key to `auth set` and unsets it before `exec` + starts. +- **Later**: offering, in VS Code, to copy the key into the OS store for + the other editors needs the native module in the `.vsix`, so + per-platform packages (with M64). + +### D62 — The ACP agent, and the order the editors come in (2026-09-26) + +The owner (2026-09-26): "the top editors come first but i want them all or +as close to all as possible", and approved the ACP SDK. One agent over the +Agent Client Protocol reaches the most editors for the least code (Zed, +the JetBrains IDEs through AI Assistant, Xcode 27, Qt Creator, Neovim, +Emacs, Sublime Text, Devin Desktop), so it comes first. + +- **One executable**, `muse-spark-code-acp` (an npm package of that name; + its bin is `dist/acp.js`, Node 22 or later), speaking ACP v1 on stdio + through `@agentclientprotocol/sdk`, which parses every inbound frame + against the protocol's schema before a handler runs (rule 7). stdout is + the protocol; the log goes to stderr, redacted. +- **The panel's engine, not a second one**: the backend managers of + `src/host/backend` (portable since M61), the same `AgentHost`, + `AgentSession` and `AgentEvent`s. The ACP code lives in `src/acp` and the + process wiring in `src/runtime`, both under the M60 gate's portable + roots. +- **The backend is chosen at launch**: `--backend museCode` (the default: + the CLI signed in on its own, the subscription pays) or `--backend +modelApi` (the key of D61). There is no "auto", so the bill is never a + surprise; a user who wants both configures two agents. +- **What maps to what**: messages to `agent_message_chunk`; reasoning to + `agent_thought_chunk`; tool calls to `tool_call` and `tool_call_update` + (kind, title, locations, arguments, output, diffs for edits); a subagent + to a tool call; the todo list to a `plan`; approvals to + `session/request_permission` with the backend's own choices (allow or + deny, once or for the session: `allow_once`, `allow_always`, + `reject_once`, `reject_always`); permission modes to session modes; + model and effort to config options; skills to available commands; the + session's name to `session_info_update`; context use to `usage_update`. +- **Nothing runs by a translation default**: a permission request the + client cancels, or answers with an option it was not offered, is decided + with the backend's deny choice. A question the agent asks goes to the + client's elicitation form where it has one; otherwise the question is + shown as text and declined, so the model carries on and the user answers + in the next prompt. +- **Sessions**: new, load (the history replayed as updates), list, resume, + and fork where the backend allows it (`canEditSessions`). +- **Prompts**: text, resource links (as @mentions), embedded text + resources (as context), images (checked by their headers, as + attachments are). +- **Sign-in**: `initialize` offers the chosen backend's sign-in, "Sign in + to Muse Code" (the agent's `login`, which runs `muse login`) or "Store a + Meta Model API key" (`auth set`, D61), as a terminal method to a client + that runs them and as a command to run by hand to one that does not; + `session/new` answers `auth_required` until the chosen backend has its + credential. Muse Code's is read as the panel reads it (D26, PR #49): the + credential file's structure, and `account/read` where only the CLI can + say; `authenticate` asks afresh. +- **Load and resume run as advertised** (the Codex review of `a209130`): + a loaded or resumed session gets the permission mode, a listed model and + the effort the agent shows, or the load fails; the agent never shows a + mode stricter than the one in force. What it shows comes from the + backend's answer or an explicit set, never from a value the session's + handle merely holds (Codex on `4eb0156c`: Muse Code's resumed handle + holds the model asked for, the CLI the one last used): the model is + the one `model/list` reports active for the session where the agent + lists it, else the default, set. A session is held, and its id + answered, only once it is set up; a new, loaded or resumed session + whose setup fails is let go. A session loaded again lets the one held + go first, and one still being set up by an earlier load, as both hosts + hand the same session back; a close lets both go. A session let go + changes nothing more on the backend, sends the editor nothing more, + decides nothing on the editor's late answers (a paid use is denied), + and its running prompt ends cancelled with its turn stopped on the + backend once its start is answered (even a failed start, which past + its deadline may still begin). A reload follows the session only once + the held one's turn is stopped; the backend stopping lets go of loads + being set up too. The client's answers to the + agent's own requests (permission, elicitation) are parsed with zod, and + a form answer must be one the form allowed (its options, how many), or + the question is declined. The agent's log names a backend failure by + its kind (`failureForLog`, PR #49), never by its message. +- **Trust**: a folder's rules, skills and memory load only with + `--trust-workspace`, the flag Muse Code itself takes (D13); ACP carries + no workspace trust of its own. +- **Paid features are off in the agent** (rule 12, D60) until a + confirmation over `session/request_permission` that names the price is + built and certified. **Built 2026-09-26 (M63c):** `--web-search` and + `--image-generation` (Model API backend only) let the first prompt ask + for each, with the panel's title and price; only "Turn on" turns it on, + for the life of the process, and anything else leaves it off without + asking again. Paid rows and approvals name their price; Muse Voice has + no microphone in the agent. +- **Amended 2026-09-27 (M58 joined, D48): every paid use asks in the + editor.** The owner's rule, "anything requiring extra payment should + prompt … allow once, allow always in this workspace, or deny", holds in + the agent too, through the core's `PaidUseConsent` and the panel's own + words (`paidUseQuestion`, moved beside it so the modal and the agent say + the same): + - A feature is on only with its flag (`subagents`, scheduled prompts and + Muse Voice have none, so they are always denied); each use then asks + over `session/request_permission` in the conversation it is for: a row + naming what is about to be billed and its price, and the options + `allow_once`, `allow_always` and `reject_once`. Web search asks once per + prompt, an image before it is bought, as in the panel. The first + prompt's "Turn on" question is gone: the price is named every time + instead, and asking twice before one prompt would say the same thing + twice. + - The Model API host serves every conversation of a folder, so its + `allowsPaidUse` now names the conversation (`sessionId`, a child task's + being its parent's); the agent asks in that ACP session and denies when + it holds no such session or has no client to ask. A cancelled prompt, + an option it did not offer, or a failed request is Deny. + - "Allow always in this workspace" is offered and honoured only with + `--trust-workspace`, as the panel offers it only in a trusted workspace. + It is kept per folder in the agent's data folder (`acp/paid-uses.json.d`, + per-feature generations and workspace-hash/generation grant records, + `src/runtime/paidGrants.ts`), read at every question so other agent + processes' changes count. Independent records and atomic generation + revocation prevent cross-process stale writes from restoring a revoked + grant or replacing a newer one; the earlier JSON map is ignored and asks + again. A filesystem that cannot safely publish the generation remembers + nothing and keeps the explicit use as Allow once. The grant + lapses in every folder when the agent starts without that feature's + flag, so turning the flag on again asks again (the panel's grant + generation, D48, in the agent's terms). +- **Networks (Q66, 2026-09-27): loud, not re-routed.** VS Code's proxy and + certificate settings do not reach the agent, and Node's `fetch` uses the + environment's proxy only with `NODE_USE_ENV_PROXY=1` (Node 22.21+, 24+). + The agent does not turn that on or add a proxy agent of its own; it logs + one warning at start when `HTTPS_PROXY`, `HTTP_PROXY` (either case) is set + for the Model API backend and the switch is off or this Node lacks it, and + a request that never reaches Meta names the agent's environment + (`NODE_USE_ENV_PROXY`, `HTTPS_PROXY`, `NODE_EXTRA_CA_CERTS`, + `--use-system-ca`) rather than VS Code's `http.*` settings. Muse Code, + started by the agent, reads the proxy variables itself. +- **Tools run in the agent**, as ACP allows. Routing the Model API + backend's file reads and writes through the client (`fs/*`), so an + unsaved buffer is seen and never overwritten, is a later step, with its + own tests (the owner's plan, §6.1). +- **Where it is tested**: against the SDK's own client in-process and + over a real stdio pipe to the built `dist/acp.js`, with the fake Muse + Code CLI (`test/e2e`); in editors as each can be installed. This + container reaches npm, PyPI, Maven Central, Gradle, NuGet, Ubuntu's + archive and download.eclipse.org, and not JetBrains, Microsoft's + VS Code downloads, Open VSX, Zed's site or neovim.io (2026-09-26). +- **The order**: the most-used editors first. VS Code's family (Cursor, + Windsurf, VSCodium, Kiro, Positron, Theia, code-server, Codespaces) + through the `.vsix` and Open VSX (M62); the JetBrains IDEs, Zed, Neovim, + Emacs, Xcode 27, Qt Creator, Sublime and Devin through this agent (M63); + then Visual Studio and the JetBrains full panel (M64); Eclipse, + NetBeans, Jupyter, Spyder and RStudio (M65); and the constrained hosts + (M66). `docs/ide-compatibility/hosts.md` tracks each editor's route and + status. + +### D63 — Turn checkpoints ship as a Preview, and the restore is rebuilt on the tools' own writes (2026-10-01) + +- **What happened.** PR #55 went through seven Codex review rounds. From + the third on, every round found new P1 races of one family: the restore + undoes the difference between whole-workspace captures, so it must decide + which changes were the turn's and which were the user's, another window's, + a subagent's, a dead window's, or the clock's. Each fix closed one case and + opened ground for the next (4, 6, 5 and 6 new findings in rounds 4 to 7). + The owner's rule from 2026-09-28 (a third round means redesign, not + patching) applied from round three and was not raised; the owner raised it. +- **Decision (owner, 2026-10-01).** 0.10.0 ships with `museSpark.turnCheckpoints` + off by default and marked Preview: a user who turns it on gets M72 as + certified, with its limits recorded in `docs/certification/m72.md`. The + restore is rebuilt in M86 on the model's own tool writes, and the setting + goes back on by default only with M86. +- **The new design (M86).** Each write a model tool makes records the file's + bytes before and after (the tools already copy a file before writing it). + A restore puts a file back only when it still holds exactly the bytes the + tool left; any other content (a user's save in any window, another + window's turn, a shell command, a later tool write it does not undo) is + refused and named. Ownership is never inferred from captures, so the + multi-window, overlap, tie and clock races cannot arise. A shell command's + changes are listed, not undone, as Claude Code's own rewind does; this + narrows D51, which promised to undo them. + ## 3. Open questions (need the owner) -| # | Question | Default until answered | -| --- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------- | -| Q1 | **Resolved 2026-09-22:** owner authorised installing anything needed; Muse Code CLI 1.3.0 installed via the official installer. The owner reported Muse Code CLI device sign-in and a pay-as-you-go Model API key; M7 keeps them apart (D1 amendment). A separate current Muse Code paid entitlement or tier is unverified; the personal Muse Power/Maximum screenshot is not CLI entitlement proof. | Closed. | -| Q2 | **Resolved 2026-09-22:** publisher `RandyNorthrup` read from the signed-in marketplace management page. Display name stays "Muse Spark Code (Unofficial)" unless the owner asks otherwise. | Closed. | -| Q3 | **Resolved 2026-09-22:** owner wants both the CLI (MSP) backend and the Model API backend in the first release. M7 is required for v0.1.0. | M7 required; see §10. | -| Q4 | **Resolved 2026-09-22 (M9, superseding the M8 answer):** voice dictation ships through the operating system's own recogniser, at no API cost and with no third-party code (owner's constraints): Windows PowerShell 5.1 + `System.Speech` on Windows, a Swift helper on Apple's Speech framework on macOS (owner chose this over an `osascript` bridge), and a dimmed button with the reason on Linux (no distribution ships a recogniser; the owner may revisit). The M8 finding stands for the webview itself: Electron's Web Speech recogniser is dead, so recognition runs in a helper process. | Closed; see M9. | -| Q5 | **Resolved (M4):** `highlight.js` 11.12.0 core with a fixed language set, in the webview bundle; `shiki` was not taken (grammar weight). | Closed. | -| Q6 | **Partly answered (M55):** Meta's Muse Code overview (checked 2026-09-25) documents `curl -fsSL https://dev.meta.ai/install.sh \| sh` for macOS and Linux, and the panel offers it; `muse` itself has not been run on Linux. | Offer the installer; the Model API key remains the fallback if `muse` is absent. | -| Q7 | **Resolved 2026-09-22:** owner pressed F5 and confirmed the Muse Spark chat shell renders in the Extension Development Host (verbal confirmation; no screenshot filed). | Closed. | -| Q8 | **Resolved 2026-09-22:** owner signed in; publisher is `RandyNorthrup`. Publishing ran by hand from the CI artifact with a clipboard PAT for 0.1.0–0.5.0; since 2026-09-23 the `VSCE_PAT` repository secret lets `release.yml` publish every `v*` tag. | Closed. | -| Q9 | The Muse Code user rules file: `/rules import` writes one into the config root and the model is told "if user and project rules conflict, project rules win", but its file name is not printed by `muse --help`, `muse skills`, the settings skill or the binary's strings. The Model API backend cannot mirror what it cannot name. | Not loaded on the Model API backend; the CLI backend loads it itself. | +- **M72 native/process exclusion:** what upstream pre-edit fence and locally + owned full-descendant shutdown proof can make native/command/hook snapshots + destructively restorable? Current availability refuses unproved process + state; no wire field, primary exit, pipe drain or dead window PID supplies + that proof. Explicit confirmed removal of only a stale unsafe presence file + preserves saved checkpoints. No paid/live capture is authorized by this + implementation; investigate a bounded future proof separately. +- **M72 hooks:** should a hook get a final admission of its own? Today none: + SessionEnd hooks run on purpose while the Host closes, hook settings are + user-global and not trust gated, and Stop is honoured through the signal; the + wrapper's activity mark still fences Restore. Default: unchanged. +- **M72 long storage paths:** should a checkpoint folder over 240 characters + (Windows) be served by moving the repository? That would change the hashed + namespace. Default: refuse with the `checkpointPathTooLong` message. +- **M72 dirty buffers in other windows:** a restore refuses a file with unsaved + changes in the window that restores, but an idle peer window on the same folder + publishes only that it is open, so its unsaved buffer of a file the restore + overwrites or deletes is not known. VS Code keeps that buffer and reports the + file changed on disk at its next save (it asks before overwriting), so nothing + is silently lost. Should every window publish its unsaved paths in its + presence file, so a restore refuses them too? Default: unchanged, recorded as + a limit (Codex, `a424e526`). + +| # | Question | Default until answered | +| --- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- | +| Q1 | **Resolved 2026-09-22:** owner authorised installing anything needed; Muse Code CLI 1.3.0 installed via the official installer. The owner reported Muse Code CLI device sign-in and a pay-as-you-go Model API key; M7 keeps them apart (D1 amendment). A separate current Muse Code paid entitlement or tier is unverified; the personal Muse Power/Maximum screenshot is not CLI entitlement proof. | Closed. | +| Q2 | **Resolved 2026-09-22:** publisher `RandyNorthrup` read from the signed-in marketplace management page. Display name stays "Muse Spark Code (Unofficial)" unless the owner asks otherwise. | Closed. | +| Q3 | **Resolved 2026-09-22:** owner wants both the CLI (MSP) backend and the Model API backend in the first release. M7 is required for v0.1.0. | M7 required; see §10. | +| Q4 | **Resolved 2026-09-22 (M9, superseding the M8 answer):** voice dictation ships through the operating system's own recogniser, at no API cost and with no third-party code (owner's constraints): Windows PowerShell 5.1 + `System.Speech` on Windows, a Swift helper on Apple's Speech framework on macOS (owner chose this over an `osascript` bridge), and a dimmed button with the reason on Linux (no distribution ships a recogniser; the owner may revisit). The M8 finding stands for the webview itself: Electron's Web Speech recogniser is dead, so recognition runs in a helper process. | Closed; see M9. | +| Q5 | **Resolved (M4):** `highlight.js` 11.12.0 core with a fixed language set, in the webview bundle; `shiki` was not taken (grammar weight). | Closed. | +| Q6 | **Partly answered (M55):** Meta's Muse Code overview (checked 2026-09-25) documents `curl -fsSL https://dev.meta.ai/install.sh \| sh` for macOS and Linux, and the panel offers it; `muse` itself has not been run on Linux. | Offer the installer; the Model API key remains the fallback if `muse` is absent. | +| Q7 | **Resolved 2026-09-22:** owner pressed F5 and confirmed the Muse Spark chat shell renders in the Extension Development Host (verbal confirmation; no screenshot filed). | Closed. | +| Q8 | **Resolved 2026-09-22:** owner signed in; publisher is `RandyNorthrup`. Publishing ran by hand from the CI artifact with a clipboard PAT for 0.1.0–0.5.0; since 2026-09-23 the `VSCE_PAT` repository secret lets `release.yml` publish every `v*` tag. | Closed. | +| Q9 | The Muse Code user rules file: `/rules import` writes one into the config root and the model is told "if user and project rules conflict, project rules win", but its file name is not printed by `muse --help`, `muse skills`, the settings skill or the binary's strings. The Model API backend cannot mirror what it cannot name. | Not loaded on the Model API backend; the CLI backend loads it itself. | +| Q10 | M67's repo map on Muse Code: the plan asks for it "as an opt-in section of the system prompt", but Muse Code's instructions are its own (D13: nothing installed into its folders). It could ride as a hidden note on the first turn of a conversation (as the question-card hint does), billed to the subscription as prompt tokens. Wanted? | The `repoMap` tool only; no note in Muse Code turns. | +| Q11 | M67's prompt repo map setting: its name (`museSpark.modelApiRepoMap`), its default (off, since every request pays its tokens) and its fixed ~1,000-token budget, and whether the model should see the map by default once the M75 evaluation measures it. | Off by default, machine-scoped, 1,024 tokens, no budget setting. | +| Q60 | **Answered 2026-09-26:** the owner set up the Open VSX account: the Eclipse Publisher Agreement signed, the namespace `RandyNorthrup` created, the token in `OVSX_PAT`. The release workflow publishes there from the next tag (M62). | +| Q61 | **Resolved 2026-09-26:** the owner approved the ACP SDK. `@agentclientprotocol/sdk` 1.4.0 is pinned: 1.5.0 (2026-09-21) is inside `.npmrc`'s 7-day `min-release-age`, and 1.4.0 speaks the same ACP v1 (D62). | +| Q62 | **Resolved 2026-09-26:** "you can install whatever you need". What this container's network lets in is recorded per editor (D62); the rest is qualified in CI or on the owner's machines. | +| Q63 | **Resolved 2026-09-26:** the owner left the design to us: D61, the operating system's credential store, in-process. | +| Q64 | **Resolved 2026-09-26:** "the top editors come first but i want them all or as close to all as possible": the order is D62's. | +| Q65 | **Answered 2026-09-26:** after npm held the owner's account for suspicious activity, the owner set `NPM_TOKEN` in the `marketplace` environment. The name `muse-spark-code-acp` was free that day; the next tag publishes it, and each GitHub Release still carries the package. | +| Q66 | **Resolved 2026-09-27: loud, not re-routed.** The ACP agent's own requests (the Model API backend) use Node's `fetch`, which ignores `HTTPS_PROXY` unless `NODE_USE_ENV_PROXY=1` (Node 22.21+ or 24+; measured on seven releases). The owner: the agent does not re-route by itself or add undici. It warns once at start, in its log, when a proxy variable is set for the Model API backend and Node's switch is off or missing (`src/runtime/proxyWarning.ts`), and a request that never reaches Meta gets advice naming the agent's environment variables instead of VS Code's `http.*` settings (M56's classifier, told by the runtime which host it serves: `networkAdvice: 'agent'`). | Closed; `docs/acp.md` "Networks and proxies", `docs/certification/pr32-integration.md`. | +| Q12 | Should a shell tool session rule ("Always allow in this session" for a shell command) lapse when the model edits a file the command names or that decides what it runs, as the verify loop's rules do since M68? Today the shell tool keeps its pre-M68 behaviour: its rules are keyed on the exact command line and answer whatever the model edited. The verify loop's grants are kept apart from it (PR #54). | The shell tool's rules keep answering; only the verify loop's lapse. | ## 4. Architecture @@ -4712,7 +5234,8 @@ merged as PR #30 (`bdaede4`). ### M44b — Web fetch on the Model API backend (D36) **Status 2026-09-27: folded into M69 (D49), which also serves it to Muse -Code through the `ide` server.** Muse Code's own +Code through the `ide` server; built there on 2026-09-28 with every rule +below (see M69's status).** Muse Code's own `web_fetch` is gated off in 1.3.0, and the Model API backend has no fetch tool. The D36 inventory named the network-safety design this needs before the model may read a page: @@ -6535,6 +7058,12 @@ harness scenario, which is what the accessibility gate checks (D32). ### M67 — Code intelligence tools (D49) +**Integration review 2026-09-29:** PR #57's candidate is being joined with +PR #32 before final gates. A Stop during the last awaited check of the first +rename file could still write; the first-write boundary now rechecks cancellation. +The two held-check regression cases, independent review and final local rig +gates are tracked in `docs/certification/m67.md`; no merged status is claimed. + - **Goal.** The model finds definitions, references and symbols the way an IDE does, instead of grepping. - **Scope.** @@ -6570,9 +7099,190 @@ harness scenario, which is what the accessibility gate checks (D32). - **Tests.** A fake language-service host in unit tests. An integration test in VS Code over a TypeScript fixture. - **Size.** M. +- **As built (2026-09-28).** Decisions taken while building, each open to + the owner's review: + - **One core, two surfaces.** `src/core/codeIntel/` holds the tools + (confinement, placing, capping, the repo map, the rename plan) over a + `LanguageServiceHost` interface; `src/host/codeIntel/languageServices.ts` + implements it with VS Code's `vscode.execute…Provider`, + `vscode.prepareCallHierarchy`/`provide…Calls`, `vscode.prepareRename` + and `vscode.executeDocumentRenameProvider` commands. The Model API + offers `find_definition`, `find_references`, `workspace_symbols`, + `document_symbols`, `hover`, `call_hierarchy`, `repo_map`, + `rename_symbol`; the `ide` server offers the same as `findDefinition`, + …, `renameSymbol` (the camel case of `getDiagnostics`). The core stays + outside `src/core/backends/modelapi/**`, since the activation bundle + carries it for the `ide` tools. + - **Naming a symbol.** Path, line and column (1-based); path, line and + name (its first whole-word use on the line); path and name (its first + use in the file); or name alone, looked up among the workspace symbols + (exact name, TypeScript's `greet()` read as `greet`; the first in place + order is used and the others listed). + - **"No language service".** VS Code exposes no way to ask whether a + provider exists, and its commands answer an empty list either way. An + empty answer is therefore checked against the file's document symbols: + none means "no language service answered for (language )", + worded to allow for a file that declares nothing; some means "No + at ". Workspace symbols have no file to check, so an + empty answer says that they come from the languages' services and that + TypeScript's needs a project file open. + - **Placing results.** A result is in the workspace when its path is + (textual and canonical confinement, D24) or when its real path is under + the root's real path (a workspace opened through a link, whose files a + server reports by real path). Everything else, including virtual + documents, is left out and counted. Lines come from the disk. + - **Caps.** 100 locations, 200 symbols, 50 callers or callees with five + call sites each, 4,000 characters of hover, three outline levels; a + 20-second deadline per language-service call. + - **Rename.** Planned before the card from VS Code's rename edit: every + file must be placed in the workspace (any outside refuses the whole + rename), at most 200 files, no file operations, no unsaved changes, and + the document's text equal to the disk's (BOM aside), so the edit lands + where the service meant it. On the Model API the card is a `fileWrite` + naming up to five files and counting the rest, protected when any file + is (D24); after approval every file is confined and read again and + nothing is written unless each is unchanged; files are written with the + tools' atomic write, one patch across them (per-line hunks, which Edit + Review's revert undoes), and the fingerprints `write_file` checks are + updated. A failed write stops the rest and names what was written. + Plan refuses a rename before the language service is asked. On `ide` + the tool returns a unified diff and writes nothing (read-only). The + file tools' one-hunk patch (D27's `hunkBetween`) moved beside the + rename's hunks as `changeHunk` in `codeText.ts`, unchanged, so the two + share one implementation (the duplication gate). + - **Repo map.** Aider's idea over VS Code's services: names of three + characters or more are counted in the text of up to 1,000 listed files + (128 KiB each, read confined); the 300 names used by the most files are + looked up as workspace symbols, eight at a time, within 10 seconds (5 for + the prompt); each file scores the uses of its names by other files, + shared among a name's definers. Document symbols per file were rejected: + opening every file would make VS Code open each document for every + extension (a linter lints them all). The prompt section is opt in + (`museSpark.modelApiRepoMap`, machine-scoped since it bills prompt + tokens), made on the first turn that has it on and kept for the session + so the prompt's prefix stays cached (a Stop ends it at once, and a map + cut short that way is not kept); Muse Code's instructions are its own, + so there it is the `repoMap` tool only. + - **Annotations.** `McpTool` gained `annotations` (as M69 adds it); + `getDiagnostics` and every code intelligence tool declare + `readOnlyHint: true`, and all are listed in Restricted Mode. + - **Tests.** `codeIntelTools`, `codeText`, `renamePlan`, `repoMap`, + `modelApiCodeIntel`, `ideCodeIntelTools`, `languageServices` (the + adapter over the `vscode` mock) and `toolPresentation`; + `test/integration/codeIntel.test.ts` over `test/fixtures/workspace/ +code-intel` on VS Code stable and 1.125.0; live case19 of the Model + API sweep; the `code-intel` harness scenario. + - **Review round (after `c5c4045b`).** Three class reviews; every + finding fixed in one commit on a merge of `origin/main` (PR #50): + - A rename's edit ranges must each cover exactly the old name (the + text the edit at the position replaces, read from the text the + position came from), and that file must still be that text: an edit + the service computed on an older version is refused, never applied. + - File operations: VS Code's `WorkspaceEdit.entries()` lists only text + edits and `size` counts them (read from the extension hosts of 1.99.0, + 1.125.0 and 1.139.0), so the old check never fired. The adapter reads the + internal `_allEntries()` through zod (`_type` 2 is a text edit, 1 a + file operation); a missing or changed list is `unknown`, refused + (§8 records the undocumented member). + - Writing: every file is checked again after the card, then each once + more right before its own write; a change found partway stops with a + revertable patch of what was written. Stop before the first write + writes nothing; once writing starts the rest follow. A failed + `rename_symbol` row with a patch keeps review and rewind + (`PARTIAL_EDIT_TOOLS`, `hasLandedEdits`). + - The prompt's repo map: trusted workspaces only; only a map with text + is kept, a try that fails or comes out empty counts, three at most + (`REPO_MAP_PROMPT_TRIES`), and a Stop does not count; child tasks and + forks use the conversation's. `Limits` takes its Stop listener off, + begins no work past the budget, and holds the file listing to it; the + map says how many files it read. + - Unsaved files: a line number in one is refused; lines shown are the + editor's, and the answer says so. + - Hover: held back when every definition is outside the workspace and + outside the languages' libraries (VS Code's `appRoot` and each + extension's folder, `LanguageServiceHost.libraryRoots`). Types that + flow from outside files into workspace symbols remain (§9). + - Hooks: `rename_symbol` matches `Edit`, and a matching `PreToolUse` + hook gets the planned `files` (planned for it alone, only when one + would run and the mode allows edits). + - Wording: "no language service" allows a file that declares nothing; + an empty answer says the language may lack that provider; the same + name, a change after the plan, the header's file name. A rename named + by symbol alone has no file button. The card names protected files + first. + - Captures: live case19 again with a stand-in that skips string + literals (4 references, 4 edits, the import asserted); and Muse Code + 1.4.0-R4302.1 calling `mcp__ide__findReferences` in on-request mode + (4 model attempts): it listed our tools with annotations, asked its + own card for a `readOnlyHint` tool, and showed our text verbatim. + The README's "reads in every mode" is the Model API's alone. + - **Second review (Grok Build on `585af100`).** One P1, five P2; all + held on inspection and are fixed in one commit: + - Unsaved changes by real path: `ToolIo.unsavedFiles()` lists the + editors' files, and `unsavedDocumentPath` matches one to a file by its + own path, the service's, or its real path (a workspace opened through + a link names files by the link in the editor and by the real path in + the language service). The plan, the recheck before each write, the + lines shown and the target all use it; a target is asked and read at + the editor's own path. + - Repo map: a batch the time or a Stop cuts off is dropped whole, so + nothing it finds later reaches the map or its counts; "no language + service" only when every lookup ran and none found anything, a cut + map being partial. + - A rename planned for its `PreToolUse` hooks is the plan written; a + hook's new arguments plan afresh. + - Call hierarchy: outgoing call sites name the file of the function + asked about; of several items at a position the one declared there is + asked, and the answer counts the others. + - **Codex on PR #57.** Two P2s, both held and fixed in one commit: + - The repo map counts every line it renders against its budget (the + lead, the count of files left out, the notes, and the prompt + section's heading); `repo_map` refuses a `max_tokens` too small for + its own fixed text and names the size that would do. + - `document_symbols` opens and outlines the file at the path of an + editor holding unsaved changes to it (`openAsEdited`), as the other + tools do; no other tool opened a named file directly. +- **Status.** Built on `feature/m67-code-intel` (2026-09-28); reviewed and + fixed the same day. Drills and the live checks in + `docs/certification/m67.md`. ### M68 — Verify loop (D49) +**Resume integration, 2026-09-29 (source preparation only).** Join M68 and +its reviewed workspace-edit repair onto main `f7db5715` (M79 and PR51 included), preserving +the M67/M69 tools, required MCP cancellation signals, unsaved-file access, +ACP key isolation and the VS Code 1.99/Node 20.18 floor. Move the existing +workspace-edit registry unchanged to a core structural-observer module; +the backend manager owns and injects it before a lazy host can start, so +pending writes reach new conversations and children without loading the +lazy verify ledger in activation or ACP. The ACP runtime shares registries +by an existing canonical directory's native bigint device/inode identity, +refusing unreadable, non-directory or unusable identities. Bind actual tool +and context paths to the immutable canonical root, keep the raw cwd only for +saved-session identity, and fence the raw and canonical roots' captured native +identity before mutations, commands and final publications after awaited +staging. Cached aliases never silently rebind after retargeting. Native alias, +retargeting, directory replacement and distinct-directory +controls remain unverified until the Windows, macOS and Linux rig slot. +Rename begins notices for every +planned canonical/alias path before rechecks, records only successful +writes in the originating session, and completes all notices in finally. +Keep the current first-write Stop guard. M79's plan publication brackets each +checked create attempt with shared notices, records only a true new-file result +in its captured live Model API owner, and always completes in finally. A changed +conversation never becomes the saved plan's invented owner. Preserve M79's +owned-stage cleanup with M68's conditional writer. M77's winner-apply binding +remains a later integration seam. No verifier or certification claim is made +during preparation. The frozen behavior tree `1c3dbea9` has current host, +unit, e2e and webview types, scoped lint and the normal duplication gate all +at exit 0. Fifty-two bounded acceptance tests passed before and after 16 +deliberate production failures; every mutated source hash restored exactly. +Fifteen failures reached intended assertions, and the lazy-injection failure +reached its exact subscription-contract TypeError, labeled separately. +The host API inventory regenerated successfully. Full exact-tree quality, +independent staged review, installed ACP/VS Code and native rig controls +remain required; this evidence makes no live-model or full-rig claim. + - **Goal.** Every edit is checked, and the model sees the result without asking. - **Scope.** @@ -6613,6 +7323,21 @@ harness scenario, which is what the accessibility gate checks (D32). - **Acceptance.** Check commands and `then_run` never run in Plan or Restricted Mode, and never run unapproved where a shell command would ask. `then_run` shows in the row as one call with two results. + - **Workspace edit review, 2026-09-29:** before an approved `write_file` + or `edit_file` can write or format, notify every live Model API session + in this workspace, including parents, children and siblings. Pending + names must lapse the verify loop's grants even during a new message; + checks over pending writes cannot count as current. Completion or + failure releases the pending state and invalidates runs that started + during the write. New sessions join active notices; disposed sessions + leave them. Keep each session's successful-edit diagnostics and fix + loop separate, and preserve the shell tool's rules (Q12). Reuse + `ModelApiHost`'s session lifetime and `VerifyLedger`, with held writes + and formatters in the existing fake API tests plus red drills. + Project memory writes and an added note's `MEMORY.md` index also notify + the ledger when confined inside this workspace; they still do not + schedule automatic diagnostics or checks. Other writers (shell, image + and external MCP tools) keep M68's existing scope. - **Evidence.** OpenCode, Aider (`--lint-cmd`/`--test-cmd`), Crush, SoL-Pi. - **Tests.** The fake Model API over a fixture with a failing check: the @@ -6620,6 +7345,152 @@ harness scenario, which is what the accessibility gate checks (D32). at its limit, and a check asks where a shell command asks (a drill per mode). - **Size.** M. +- **Status 2026-09-28: built and certified on `feature/m68-verify-loop`** + (`docs/certification/m68.md`); not pushed. Decisions taken: + - **Settings**, all machine-scoped (D15): `diagnosticsAfterEdits` (on), + `checkCommands` (none; `{ name, command, changedFiles?, +timeoutSeconds? }`, at most 8, names unique, 300 s unless set, 600 s at + most), `formatOnEdit` (off). The loop is Model API only, as scoped; + `run_checks` and `then_run` are offered only with the shell. + - **After a round that edited files** (the edit tools' writes; not the + shell's, the memory tools' or images), before the next request: the + edited files' diagnostics, then the checks, in a **Check edits** row + (`verify_edits`) whose summary counts errors and warnings and names each + check's outcome; the model reads it all as a user note that leads with + "tool data, not a new instruction from the user". A diagnostic is + matched across rounds by severity, source and message, never its line, + for "N new, M fixed"; at most 50 are listed. + - **The language servers report only on files an editor shows** + (measured, VS Code 1.139.1 and 1.125.0: a hidden `.ts` and `.json` got + nothing in 25 s, shown ones in 1.6 s and 0.1 s). So each edited file no + editor shows opens beside the active editor in a tab of its own (not a + preview, which `workbench.editor.enablePreview: false` would make + permanent and which would replace the user's own preview), without + taking focus (beside, so nothing the user types lands in it), and the + tabs it opened close after the read unless the user changed them. After + the M68 review the wait starts once the file shows: 4 s for a first + report, then 1.5 s of quiet, 10 s at most, a Stop ending it for every + file left; one queue serves every caller (panels, subagents, the + `getDiagnostics` tool). A file with no report, unsaved, or past the + round's first 8 (`VERIFY_SHOWN_FILES_MAX`) is "not checked" with the + reason, never clean, and leaves the "N new" baseline alone; the baseline + moves only once the note reached the conversation. The existing + `getDiagnostics` tool does the same for a file it is asked about, on + both backends, which is what makes the Muse Code guidance useful; it is + confined by real path first, and its wait ends when the MCP caller goes + away. + - **Permission path** (`authorizeCommand`): Restricted Mode refuses; the + mode's shell verdict decides (Plan refuses, Bypass allows, the others + ask); the card is the shell's own (`shell` subject, so Edit + automatically never answers it), with the PermissionRequest hook seeing + it as a shell call; "always allow in this session" is keyed on the + configured command without its paths, under the verify loop's own key + (`VERIFY_COMMAND_RULE_KEY`) since PR #54's fourth Codex round: a + check's or `then_run`'s grant never answers for the model's own shell + call of the same command, nor the shell's for them. A check the + user rejects is not asked again until their next message. In Restricted + Mode the automatic checks are left out rather than refused one by one. + Superseded by the M68 review: checks, `run_checks` and `then_run` go + through the user's PreToolUse, PostToolUse and PostToolUseFailure hooks + as calls of the shell tool (`runVerifyCommand`: a block is "a hook + denied it" with the hook's words, `updatedInput.command` replaces the + line and the rule key, a demanded question asks even in Bypass, a + PostToolUse context or reason follows the output, and `continue: false` + ends the turn); a PermissionRequest hook's denial is told apart from the + user's Reject, whose feedback is kept. The session rule stays keyed on + the configured command, which is safe because a path can no longer + inject (below), but it does not answer after the conversation, since + the user's message, edited a file that decides what the command runs + (`canChangeWhatRuns`: `COMMAND_DEFINING_FILES`, code-loading files, a + file whose path occurs in the command's text or a plain word of it + names, or, since PR #54's fourth Codex round, any edited file when the + command holds shell syntax that makes its words uncertain: quotes, + escapes, variables, substitutions, globs, operators). The judgement is + made on the command the rule is keyed on, a hook's rewrite included. + - **Paths reaching a check** (the M68 review, P1): only edited files that + exist, or `run_checks` paths that exist in the workspace; a leading `-` + or `@`, or a control character, refuses the check, and on Windows so do + `"`, `&`, `|`, `<`, `>`, `^`, `%` and `!` (`WINDOWS_ARGUMENT_SYNTAX`): + measured, Windows PowerShell 5.1 passed `["a\"","b --inject"]` to + node.exe as `a b`, `--inject`, and a `.cmd` ran `x&echo.INJECTED` as a + second command, expanded `%OS%` and dropped `^`, while spaces, both + quotes, `$`, `;`, `,`, `=`, parentheses, braces and non-ASCII passed + intact through both. + - **Code the editor runs** (the M68 review): a linter's or formatter's + JavaScript config, `package.json`, `node_modules` + (`CODE_LOADING_FILE_PATTERNS`) is never shown or formatted, and once the + conversation writes one nothing more is shown or formatted until the + user's next message ("not checked" with the reason). + - **One budget**: the verify note, diagnostics and every check together, + is at most `VERIFY_NOTE_MAX_CHARS` (64,000), shared equally; so is a + `run_checks` result. A check the model ran since the round's last edit + (`run_checks` over the edits, or a `then_run` of its own command) is not + run again after the round, and nothing runs after the turn's last round. + - **The fix loop**: `CHECK_FIX_MAX_ROUNDS` (3) failing verdicts in a row + stop the checks until the user's next message. Since PR #54's third + review the state is one `VerifyLedger` per session: every check that + ran is recorded against the file state it ran on, a round is judged + when a run since the previous verdict is current, and it passes only + when no current run of any check failed. A user's message, once + admitted, resets it all; a steered message resets the count, the + rejections and the runs, but not what the conversation wrote; a goal's + wake, and a parent model's message to a subagent, reset nothing. + `run_checks` honours the stop and the rejections too. The model is told + to stop fixing and say what still fails; the panel shows a warning. The + diagnostics go on. + - **`then_run`** (SoL-Pi's Action Fusion, reimplemented from its public + description, no code ported, so the notices generator is unchanged and + M73 stays the first milestone that may port): after the edit (and its + format), the command takes the permission path above (a hook's forced + question carries over), then runs only if the file's SHA-256 still + equals the fingerprint the edit left; else "not run: the file changed". + The row keeps the edit's diff and adds a **Then ran** block (command, + output, exit or reason). A Stop at its card keeps the edit's result and + diff. The shell's default 120 s cap applies. + - **Format on edit**: `vscode.executeFormatDocumentProvider` over the + document once it shows what the tool wrote (2 s to catch up, else + skipped), within 5 s; edits applied to the written text (BOM kept, the + file's CRLF kept), written back by the tool, the patch and fingerprint + taken after. A formatter that fails or overlaps leaves the edit as + written and is logged; so does a write-back that fails (the M68 review), + and a formatter that does not answer in time is logged too. + - **Muse Code**: a second `` with each message, from + `MODEL_TEXT`: call `mcp__ide__getDiagnostics` on each edited file (only + when the session got the `ide` server) and fix what the edit broke, and + run the named check commands (named only in a trusted workspace). The template AGENTS.md is unchanged; no skill is + installed. Automatic checks after Muse Code's own edits need an MSP + event ("a turn's edits finished", or an edit completion hook), still to + be asked of Meta upstream: the owner's to file. + - **Evidence**: 27 unit tests over the fake Model API (a fixture whose + lint fails), 11 over the editor adapter, an integration test inside VS + Code 1.139.1 and 1.125.0 (TypeScript and JSON diagnostics, the JSON + formatter), a harness scenario `verify`, 26 red drills, and a live case + (case19: 3 requests a run, two runs, contributor model, `then_run` used + and passed, the check note accepted by Meta). After the M68 review (one + pass over three reviewers' 2 P1 and 18 P2 findings, 16 of them + distinct): 47 loop tests, 18 editor tests, 12 check-command tests (two + real round trips through Windows PowerShell 5.1, one into a `.cmd`), the + harness scenario fixed and reshot, 40 red drills, and the integration + test rerun on both versions. That run caught a regression the tab + cleanup made: the JSON server clears a file's diagnostics when its tab + closes, so `settleFile` now returns what it read while the file showed + and the diagnostics tool answers with that. Codex's review of PR #54 + added four, fixed with drills R41 to R46: a hook's stop ends the + remaining checks; a queued editor caller stops while it waits; an + already shown file's report since the write, or a shown document that + holds the disk text, counts; `run_checks` rounds without edits count for + the fix loop. Its second round added three, closed as a class: every act on + a file after an await uses the confined real path and canonical name + and checks the file (its real path, and what the edit left) just + before; a reused background tab keeps its preview state and the tab + that was in front comes back (drills R47 to R56). Its third round + (four findings) was answered by a redesign: one conditional write in + `fsAtomic` for every verify-loop writer, and one `VerifyLedger` per + session that records each check against the file state it ran on, + judges a round only by runs on the latest state, and resets on any + admitted user input, queued or steered (drills R57 to R68). + - **Open for the owner**: the side editor group, the upstream MSP ask, and + the `diagnosticsAfterEdits` default (on). ### M69 — Web fetch (D49; folds in M44b) @@ -6651,6 +7522,207 @@ harness scenario, which is what the accessibility gate checks (D32). one, and a proxy that cannot take a pinned address; a harness scenario for its row; the gate green; a certification record. - **Size.** S. +- **Status 2026-09-29: built on `feature/m69-web-fetch`, PR #52 open** + (`docs/certification/m69.md`); the resumed picture repair passes focused + tests and red drills; final review and integration gates are pending. + Integration retains M67's separate language services, binds the existing + web fetcher to the standalone ACP Model API runtime, and ships its converter + worker/notices. Independent review found missing permission checks on later + address attempts and the Node 24.0–24.4 HTTPS proxy warning gap; both have + focused regressions and intended red/restored proofs. The standalone transport + remains Node's, without automatic proxy rerouting or VS Code settings. + Built to M44b's safeguards + and the plan review's six rules: + - **Destinations** (`src/core/web/publicAddress.ts`, `pageUrl.ts`): + `https:` only, no credentials, 2,048 characters at most; local and + reserved names (`localhost`, `local`, `internal`, `home.arpa`, `test`, + `invalid`, `example`, `onion`, `alt`, and any single-label name) refused + before any lookup. IPv4 is public outside IANA's special-purpose blocks + and Azure's WireServer (so 169.254.169.254, 100.100.100.200 and + 192.0.0.192 are refused); IPv6 only inside 2000::/3 and outside + 2001::/23, 2001:db8::/32 and 3fff::/20, with IPv4-mapped, -compatible, + NAT64 and 6to4 judged by the IPv4 inside. Every DNS answer is checked: + one non-public answer refuses the name. + - **Pinning** (`src/host/web/pinnedRequest.ts`): Node's `https` connects + to the checked address, with `servername` and `Host` carrying the name, + so TLS still verifies it. VS Code's patched `fetch` cannot pin (it + replaces a caller's dispatcher with its own agent, keeping only CA and + HTTP/2 options: @vscode/proxy-agent `createFetchPatch`, read + 2026-09-27); its patched `https` can, and does so through the proxy: + the integration test shows a loopback proxy receiving + `CONNECT 203.0.113.7:443` and a ClientHello naming the host, on VS Code + 1.139.1 and 1.125.0. Only an answer that arrived over TLS is read: a proxy's + own refusal of the tunnel is reported as `Proxy response (N)`, M56's + proxy failure, never read as the page. The checked addresses are raced + in the resolver's order (ADDRCONFIG) as RFC 8305 says, never + re-resolved. + - **Redirects**: same host (host and port) followed, each hop checked, + resolved and pinned again, at most `WEB_FETCH_MAX_REDIRECTS` (5); a + redirect to another host is handed back to the model as a URL to fetch + in a new call, so each host is approved on its own; into a refused URL + it fails naming the redirect. + - **Bounds**: 5 MiB after decompression (gzip, deflate, br; another + coding refused), declared or streamed; 30 s for the whole fetch; an + allow-list of text types; the header's charset, else HTML's ``, + else UTF-8. HTML becomes Markdown in one linear pass + (`htmlToMarkdown.ts`, with `entities` for character references, the + one new dependency, D3); inline nesting, list and quote indents and + table width are capped so a hostile page stays linear. The model reads + the first 50,000 characters and is told the total. + - **Approvals**: a new `network` tool class. Bypass allows, Plan + (`denyUnmatched`) refuses (its rules allow workspace reads, not network + reads), Manual, Edit automatically and Auto ask, per host: the card + (`webFetch` subject) names the URL as it will be fetched, and "Always + allow in this session" is keyed on the host. Restricted Mode: not + offered, refused if called; a side chat (always Plan) is not offered + it. A URL refused on its face (scheme, credentials, length, a reserved + name, a non-public literal address) is refused before any card; a name + is resolved only after approval, since the lookup itself carries the + name out, so one that resolves to a private address is refused after + the card and before any connection. + - **Untrusted content**: the model's text is the header line, a notice + that the page is untrusted data, and the content between markers with + 8 random bytes the page cannot know; the instructions say the same. + - **Muse Code**: `webFetch` on the `ide` server, listed only while the + workspace is trusted and `museSpark.sandboxNetwork` is not + `restricted` (the list is read per request), with MCP annotations + `readOnlyHint: false, openWorldHint: true`, and the extension's own + modal (Allow once / Reject, naming host and URL) before every call, + whatever Muse Code's mode. Live (4 model attempts, contributor model, + empty folder): Muse Code listed and called it, asked its own approval in + on-request mode, and passed our text through verbatim as the row's + output, which the row's size line reads (AGENTS rule 13). + - **Row**: the URL beside the label, "Fetched 48.2 kB (text/html)" under + it, and what the model read in the body; harness scenario `web-fetch`. + - **Review round** (three class reviewers over `c3d7702c`, all fixed in + one commit): the `ide` call gets an `AbortSignal` aborted when Muse Code + closes the request or sends `notifications/cancelled` (both captured + from Muse Code 1.4.0 on a stopped turn, 2 model attempts), raced against + the modal, with `isOffered` checked again after it and one modal per URL + at a time; the checked addresses are raced as RFC 8305 says (250 ms); + connection failures in web fetch's own words naming the host and the + addresses (not M56's Meta advice), a proxy's refusal of the tunnel + included, the detail only as error codes (a name mismatch's message + lists the certificate's names); server text outside the markers only as short tokens, the + final URL, the title and a moved target inside them; the converter + bounded at 100,000 characters (prefix depth 4, rows unpadded); all + trailing dots stripped and empty labels refused; RFC 7050 NAT64 prefix + discovery; damaged compression is the coding's failure; the charset only + from ``, an unknown label ignored; sentences name "this tool"; + Model API rows localized for a moved page and Restricted Mode; + `sandboxNetwork` described in fifteen manifest tables. PAC and + `http.noProxy` see the pinned address, not the name (@vscode/proxy-agent + 0.45.0 `agent.js` builds the proxy URL from `opts.host`): kept, since the + name would let the proxy resolve it again; documented. + - **PR #52 review** (Codex): NAT64 discovery fails closed (only a + definite "no AAAA" from the page's own resolver means no DNS64; a + timeout, SERVFAIL or an answer without a prefix leaves it unknown, and + then no IPv6 answer is used, a name with only IPv6 answers refused as + `nat64Unknown`); a `PermissionRequest` hook's allow no longer replaces + the per-host card (it may still deny or ask). Swept: every other failed + lookup or check already refuses. + - **PR #52 second review** (Codex): what allowed a fetch is asked again + after every await: after the card or hook (the turn, trust, a mode that + now refuses), before each hop's lookup and connection (a caller's + `isStillAllowed`, ending the fetch as `withdrawn`), and once the page is + in, before the model gets it; on Muse Code the offer (trust, + `sandboxNetwork`) is that check. + - **PR #52 third review** (Codex): NAT64 absence is proven only by a DNS + query's own NXDOMAIN or NODATA (c-ares), while the system resolver's + answers still reveal a prefix; the converter hides an element a page + left open (`
  • `, cells, rows) up to where a + browser ends it, tracking what is open inside and around it, and, from + the sweep, a hidden image's alt text, a self-closed hidden element, an + unopened dialog, ruby's `rp` and `datalist`. + - **PR #52 fourth review** (Codex, a self-closed `