diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 44ebb1a1a..75e9898fa 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -15,12 +15,21 @@ jobs: steps: - uses: actions/checkout@v4 - - name: Extract version from tag + - name: Extract and validate version from tag id: version - run: echo "version=${GITHUB_REF_NAME#v}" >> "$GITHUB_OUTPUT" + run: | + set -euo pipefail + version="${GITHUB_REF_NAME#v}" + if [[ ! "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then + echo "Release tags must be vMAJOR.MINOR.PATCH; refusing ${GITHUB_REF_NAME@Q}" >&2 + exit 1 + fi + echo "version=$version" >> "$GITHUB_OUTPUT" - name: Set version in pyproject.toml - run: sed -i "s/^version = .*/version = \"${{ steps.version.outputs.version }}\"/" pyproject.toml + env: + VERSION: ${{ steps.version.outputs.version }} + run: sed -i "s/^version = .*/version = \"${VERSION}\"/" pyproject.toml - name: Install nfpm run: | diff --git a/CHANGELOG.md b/CHANGELOG.md index 43bdde6ea..bc7ca307d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,117 @@ Each GitHub release body is the matching section of this file. ## [Unreleased] +### Changed + +- Compatible and OpenRouter models budget chat, loop and agent history against + the output each request actually reserves (the profile's maximum output, + capped at 32,768 tokens), the same figure agent eligibility already used. + Long tool-using turns keep more history before summarizing (DeepSeek V4 at + the default 75%: 491,520 to 761,856 working tokens), so those turns send + more input tokens per request; models whose declared output nearly equals + their context window no longer run with no history budget and no overflow + rescue. An agent whose compatible provider rejects a payload already below + the provider-reported window now compacts instead of re-sending it. Codex is + unchanged. To retain the previous smaller compatible working set, lower + `openai_compatible.context_utilization` (about 48 for DeepSeek V4's previous + 491,520 tokens). +- One-time check, workflow and webhook schedules interrupted by a restart are + now paused as inert rather than run again from the beginning. Their reason + is shown to operators; set a new `run_at` to re-arm after checking effects. + Reminders and digests still replay, and recurring schedules are unchanged. + No action is required on upgrade. +- Documented webhook success, partial-delivery and retry behaviour for + scheduled webhook actions and inbound webhooks. +- History search finds archived conversation summaries again; the first start + after upgrading indexes existing archives in the background. Expect the + session and full-text databases to grow during this one-time migration. + +### Fixed + +- Computer-use requests refused before any input (unknown source IDs, export + on an attached desktop, Hyprland target discovery failures, and stale or + unknown session references) now report `not_dispatched` with a safe next + step instead of unknown-outcome RELEASE-ALL guidance. Such refusals no longer + cancel a running session, including another session in the same channel. + Successful computer calls are audited with a success reason code rather + than `computer_rejected`. +- `email_send` reports recipients the mail server refused (To, CC or BCC) + instead of claiming the message went to everyone; a partial send is reported, + never retried. +- `email_read` no longer shows a text attachment as the message body, and its + attachment list follows the MIME disposition (any capitalisation) instead of + a text match. +- `http_probe` sends request bodies starting with `@` literally instead of + uploading a file, and rejects header names starting with `@`. Such bodies + require curl 7.43 or newer on the probing host. +- `validate_action` process checks no longer find their own command instead of + the target process; a matching ancestor still counts. Missing `pgrep` and + unusable patterns now report errors rather than false health. +- `validate_action` log checks report an error when the journal cannot be read + or is only partly readable instead of claiming it is clean. Invalid patterns + report errors and journalctl notices no longer count as log lines. Operators + whose service user lacks journal access must grant it (for example via the + `systemd-journal` group) before relying on these checks. +- `validate_action` HTTP checks accept explicitly expected 4xx/5xx statuses; + connection failures and timeouts never count as a received status. +- The release workflow rejects tags that are not `vMAJOR.MINOR.PATCH` before + touching package metadata, and passes a validated version to shell steps as + data rather than embedding tag text in a command. +- Documentation and `scripts/monitor.sh` now point to standard output and the + systemd journal rather than a log file Odin never writes. Fresh packages since + v4.0.0 start immediately in loopback-only bootstrap mode; the computer-use + handoff and install pages now say so. +- Logging in to the WebUI without **Stay logged in** now replaces an older + saved login, so reloading after a restart retains the new session instead of + returning to the login screen. A persistent login also clears the tab's old + session-only credential. +- Compatible endpoints configured for GLM preserved thinking now count replayed + reasoning when sizing context, so history is summarized before it overflows + and overflow recovery no longer claims a fit while that reasoning is still + sent; replayed reasoning is never shortened or edited. +- Verify integrity checks every retained rotated audit file, not only the active + one, and lists each file's result (verified, predates signing, break at line N, + unreadable, missing); files are streamed in a worker thread instead of read into + memory. +- A tool that hit its time limit no longer lets the turn continue when its + ledger record could not be saved; the turn stops with an error, as every + other ledger write failure already does. +- Resuming interrupted work no longer leaves a tool call unanswered, or answers + it with an earlier call's result, when a model provider reuses tool-call IDs + across replies. Agents also accept IDs reused across separate replies while + still refusing duplicates within one reply. +- Scheduled digests report disk and memory again. They run under the schedule's + identity (the creator, or the `scheduler` system identity) and list hosts they + cannot reach as collection failures. +- Grafana-triggered schedules with an alert name now match any alert in a + notification, not just the first; each schedule still runs once per + notification. +- Browser tool descriptions state that every call starts a fresh browser session: + `browser_click` no longer suggests reading the clicked page with a later + `browser_read_page`, `browser_fill` points to `submit=true`, and + `browser_evaluate` notes that navigation it starts is not awaited. Browser + behaviour is unchanged. +- Compressed tool history labels failed calls with a short reason (for example + `run_command→ERR (blocked)`, `→ERR (timed out)` or `→ERR (disallowed host)`) + instead of incorrectly marking these failures OK; Recent Actions marks + failed calls ERROR. Successful calls with explicit outcome metadata remain OK. + +- Long replies split around code blocks keep their formatting: text after a + block no longer shows as code, no message ends with an empty code block, and + a split with a long language tag no longer creates an over-limit message. + Truncated workflow and loop posts close open code blocks before the marker. +- `parse_time` uses every part of an expression, including compound durations + (`in 1 hour 30 minutes`), a day after a time (`5pm tomorrow`), a time after a + weekday (`friday 3pm`), and `in 2 days at 9am`. Unused date or time words now + return an error instead of silently scheduling the wrong time. Hours and + minutes measure elapsed time across daylight-saving changes; days retain the + local clock time. +- A cancelled knowledge, memory, list or learned write can no longer overwrite (or corrupt) a newer save that already succeeded. +- A failed knowledge ingest no longer leaves search hits for a document that was not stored or blocks that name; leftovers from earlier failures are removed at startup. +- Knowledge ingest reports failure when its version record cannot be saved, and re-ingesting the same content repairs a missing version snapshot. +- The Sessions page User ID filter now applies to every result source; summaries + and index results that cannot be attributed are left out. + ## [4.7.0] - 2026-09-24 ### Added diff --git a/README.md b/README.md index 72008cf3c..517286709 100644 --- a/README.md +++ b/README.md @@ -72,10 +72,10 @@ sudo -u odin /opt/odin/.venv/bin/python /opt/odin/scripts/codex_login.py \ --credentials-path /var/lib/odin/codex_auth.json --device sudoedit /etc/odin/config.yml # set web.api_token; bind web.host to 127.0.0.1 unless it sits behind TLS; # review permissions.default_tier (template: admin) and tools.hosts -sudo systemctl start odin # WebUI on the configured web.port (default 3000) +sudo systemctl restart odin # WebUI on the configured web.port (default 3000) ``` -**Upcoming branch packages:** unlike the published v3.98.0 procedure above, a fresh install automatically enables and starts a restricted loopback bootstrap service. Open `http://127.0.0.1:3000/ui/` locally. For a remote install, run `ssh -L 3000:127.0.0.1:3000 user@odin-host` on your workstation, then open that same local URL. Do not publish pending setup through a reverse proxy. +**Packages since v4.0.0:** unlike the v3.98.0 procedure above, a fresh install automatically enables and starts a restricted loopback bootstrap service. Open `http://127.0.0.1:3000/ui/` locally. For a remote install, run `ssh -L 3000:127.0.0.1:3000 user@odin-host` on your workstation, then open that same local URL. Do not publish pending setup through a reverse proxy. Finish setup with a strong Web API token, then sign in as administrator. **System → Config → Web listener exposure** shows the configured host, whether it came from an explicit `web.host` key or the schema default, and the addresses the current process actually owns. Authorize beyond-loopback access there and re-enter a current admin API token. This records permission for the next start, not a live rebind. After arranging TLS and network access controls, restart manually with `sudo systemctl restart odin`. The same card can revoke authorization and narrow the next start to loopback. A Discord token alone never authenticates or widens the Web listener. @@ -234,7 +234,7 @@ The package installs: | Environment file | `/etc/odin/.env` | | Persistent data | `/var/lib/odin` | | Local command workspace | `/var/lib/odin-workspace` | -| Logs | `/var/log/odin` | +| Logs | systemd journal (`sudo journalctl -u odin`); `/var/log/odin` is created but not written | | Systemd unit | `/usr/lib/systemd/system/odin.service` | | Private computer evidence (service-owned, 0700) | `/var/lib/odin/computer` | | Precompiled root-owned Wayland guardian | `/usr/libexec/odin-computer-wayland-input` | @@ -242,7 +242,7 @@ The package installs: | Computer-use installation handoff | `/usr/share/doc/odin/computer-use/PACKAGING.md` | | Computer-use setup and recovery | `/usr/share/doc/odin/computer-use/OPERATOR.md`, `RECOVERY.md` | -The package installs the application files and systemd unit. Its post-install script creates the `odin` service account, virtual environment, SSH key, data directories, configuration links, and local command workspace. Upcoming branch packages automatically enable and start a new installation in loopback-only bootstrap mode; published v3.98.0 packages are enabled but require manual configuration and start. See the [Quick start](#quick-start) for local access and SSH forwarding. Upgrades preserve configuration and data and restart the service only if it was already running. +The package installs the application files and systemd unit. Its post-install script creates the `odin` service account, virtual environment, SSH key, data directories, configuration links, and local command workspace. Since v4.0.0, a fresh installation is enabled and started in loopback-only bootstrap mode; v3.98.0 and earlier packages were enabled but required manual configuration and start. See the [Quick start](#quick-start) for local access and SSH forwarding. Upgrades preserve configuration and data and restart the service only if it was already running. Fresh installs and upgrades install the `pdf` and `computer` Python extras and provision private computer state with symlink rejection. Computer enablement is @@ -304,7 +304,7 @@ sudoedit /etc/odin/config.yml 5. Start the service and inspect startup: ```bash -sudo systemctl start odin +sudo systemctl restart odin sudo systemctl status odin sudo journalctl -u odin -f ``` diff --git a/docs/architecture.md b/docs/architecture.md index f7de5445e..3b1b0a945 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -193,8 +193,9 @@ attempts; a healthy in-flight generation has its own transport limits. [`TurnResumeManager`][resume] rechecks the original request and current authorization before resuming suspended chat work. In-process auto-resume also requires the session not to have advanced. After restart, resume is explicit. -Ledger repair supplies stored results or explicit uncertainty for unmatched tool -calls; it does not automatically execute them again. Unresolved external effects +Ledger repair pairs calls by transcript position and generation, supplying stored +results or explicit uncertainty for unmatched tool calls; it does not automatically +execute them again. Unresolved external effects block automatic continuation. ## Managed hosts: desired state, runtime identity, access @@ -223,6 +224,10 @@ result summaries, elapsed time, errors, and optional risk/diff metadata. reflection/lifecycle hooks. These explain what happened; the durable turn ledger governs replay safety. Neither replaces the other. +Each retained audit file carries its own HMAC chain from genesis; Verify integrity +checks and reports every retained file independently. It cannot detect files +already removed by rotation or deletion from the oldest end. + Large outputs should be retrieved, not regenerated merely because a preview was short. The retention contracts are intentionally different: diff --git a/docs/computer-use/PACKAGING.md b/docs/computer-use/PACKAGING.md index 6d83f5cfb..972ebf5d2 100644 --- a/docs/computer-use/PACKAGING.md +++ b/docs/computer-use/PACKAGING.md @@ -76,7 +76,12 @@ storage safely, including for source installs; they never repair existing unsafe objects or move receipts. System Python dependencies remain distinct from the venv. Fresh installation leaves computer use disabled by default. The base Odin service -is enabled but not started until the operator completes setup. Ordinary upgrades +is enabled and started immediately in a restricted bootstrap mode: it listens on +loopback only, whatever `web.host` says, and serves guided setup at +`http://127.0.0.1:3000` (use an SSH tunnel for a remote host, never a reverse +proxy) until the operator completes it; widening beyond loopback later is an +explicit, authenticated operator choice. Packages up to v3.98.0 enabled the +service without starting it. Ordinary upgrades preserve configuration, computer enablement, data and prior service state: an already-running service is restarted; an inactive one remains inactive. No computer task, desktop capture, input action, extension activation, session-bus connection, diff --git a/docs/computer-use/RECOVERY.md b/docs/computer-use/RECOVERY.md index 0d114a5ae..805b859a0 100644 --- a/docs/computer-use/RECOVERY.md +++ b/docs/computer-use/RECOVERY.md @@ -60,6 +60,13 @@ Those measures apply only when release is unverified or outcome is unknown. inspect fresh evidence before planning a different action. No category permits replaying an action whose effect may already have occurred. +Observe, source selection, export, status, target inventory and start preflight +refusals that occur before input, focus recovery or cleanup carry +`not_dispatched` and leave an existing session running. A rejected request for +an unknown source ID or an unavailable attached-desktop export therefore does +not require RELEASE-ALL. Successful calls carry the audit reason +`computer_succeeded`, or `verified`/`executed` for successful action receipts. + For the separate alternate-input rule, see [OPERATOR.md](OPERATOR.md#alternate-input-paths). For persisted quarantine, **System > Computer** remains available even when input diff --git a/docs/configuration.md b/docs/configuration.md index 365ea2b83..55e99b0f1 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -185,6 +185,9 @@ and OpenRouter. URLs are used verbatim: Odin never appends `/v1`. Local examples are vLLM `http://127.0.0.1:8000/v1`, llama.cpp `:8080/v1`, and LM Studio `:1234/v1`. Profiles declare reasoning dialect plus reasoning-content feedback; the safe default is not to echo provider reasoning into history. +A compatible model's usable prompt budget is its total window minus the output +each request asks for (the profile's max output, capped at 32,768 tokens); +`openai_compatible.context_utilization` sets how much of that history may use. `iteration_timeout_seconds` bounds each agent LLM call. It is a backstop against a hung call, not a working limit — set it well above a legitimate @@ -248,6 +251,8 @@ browser: Leave `cdp_url` empty to launch a local headless Chromium. Set to `ws://host:port?token=secret` for remote Browserless. +Each browser tool call runs in a new, empty browser context that is closed when the call ends; cookies, storage, form input and page state never carry over between calls. + Run `playwright install chromium` after installation. ## Image Generation @@ -312,6 +317,10 @@ Runtime overrides persist in `data/permissions.json` and take precedence. ## Webhooks +For scheduled HTTP actions and inbound webhook delivery, see +[Schedules & webhooks](scheduling.md). This includes retry and partial-delivery +behaviour; outbound notifications use a separate path. + ```yaml webhook: enabled: false @@ -333,9 +342,14 @@ context: ```yaml logging: level: INFO # DEBUG, INFO, WARNING, ERROR - directory: ./data/logs + directory: ./data/logs # reserved path kept out of the command workspace; Odin writes no log files here ``` +Odin writes application logs to standard output/error: for the service use +`journalctl -u odin -f`, for Docker use `docker logs odin-bot`, and for a source +run use its terminal. `logging.level` sets verbosity. Tool executions and events +go to the audit log (`data/audit.jsonl`), which the WebUI Audit and Logs pages read. + ## File Paths (DEB install) | Purpose | Path | @@ -343,7 +357,7 @@ logging: | Config | `/etc/odin/config.yml` | | Secrets | `/etc/odin/.env` | | Data | `/var/lib/odin/` | -| Logs | `/var/log/odin/` | +| Logs | systemd journal (`sudo journalctl -u odin`); `/var/log/odin` is created but not written | | Application | `/opt/odin/` | | Systemd | `/usr/lib/systemd/system/odin.service` | diff --git a/docs/install.md b/docs/install.md index 7864d6019..44670eb2d 100644 --- a/docs/install.md +++ b/docs/install.md @@ -1,7 +1,7 @@ # Install -The WebUI-first bootstrap flow below describes this branch's upcoming release. -Published v3.98.0 packages still require the earlier manual configuration and start procedure. +The WebUI-first bootstrap flow below applies to packages since v4.0.0. +v3.98.0 and earlier packages used the manual configuration and start procedure. Odin ships as an amd64 Debian package and as a source checkout. The package path is for a long-running service; the source path is for development. @@ -27,7 +27,7 @@ The package installs a dedicated `odin` system user, a Python virtual environmen | Environment file | `/etc/odin/.env` | | Persistent data | `/var/lib/odin` | | Local command workspace | `/var/lib/odin-workspace` | -| Logs | `/var/log/odin` | +| Logs | systemd journal (`sudo journalctl -u odin`); `/var/log/odin` is created but not written | | Systemd unit | `/usr/lib/systemd/system/odin.service` | ## First-time setup diff --git a/docs/reference/api.md b/docs/reference/api.md index c37d2bb9a..7b574cd8b 100644 --- a/docs/reference/api.md +++ b/docs/reference/api.md @@ -115,19 +115,19 @@ This describes the normal authenticated deployment. With no configured tokens (i | POST | /api/grafana-alerts/rules | [src.web.api.integrations](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/integrations.py#L540) | Yes | — | | DELETE | /api/grafana-alerts/rules/{rule_id} | [src.web.api.integrations](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/integrations.py#L574) | Yes | — | | GET | /api/grafana-alerts/remediations | [src.web.api.integrations](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/integrations.py#L585) | Yes | — | -| GET | /api/knowledge | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L94) | Yes | — | -| POST | /api/knowledge | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L101) | Yes | — | -| DELETE | /api/knowledge/{source} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L126) | Yes | — | -| POST | /api/knowledge/{source}/reingest | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L137) | Yes | — | -| GET | /api/knowledge/search | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L158) | Yes | — | -| GET | /api/knowledge/{source}/chunks | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L179) | Yes | — | -| GET | /api/knowledge/duplicates | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L193) | Yes | — | -| POST | /api/knowledge/merge | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L207) | Yes | — | -| GET | /api/knowledge/{source}/versions | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L234) | Yes | — | -| GET | /api/knowledge/{source}/versions/{version:\d+} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L243) | Yes | — | -| POST | /api/knowledge/{source}/versions/{version:\d+}/restore | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L255) | Yes | — | -| GET | /api/knowledge/{source}/versions/{v1:\d+}/diff/{v2:\d+} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L278) | Yes | — | -| POST | /api/knowledge/import | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L294) | Yes | — | +| GET | /api/knowledge | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L95) | Yes | — | +| POST | /api/knowledge | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L102) | Yes | — | +| DELETE | /api/knowledge/{source} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L127) | Yes | — | +| POST | /api/knowledge/{source}/reingest | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L138) | Yes | — | +| GET | /api/knowledge/search | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L159) | Yes | — | +| GET | /api/knowledge/{source}/chunks | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L180) | Yes | — | +| GET | /api/knowledge/duplicates | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L194) | Yes | — | +| POST | /api/knowledge/merge | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L208) | Yes | — | +| GET | /api/knowledge/{source}/versions | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L235) | Yes | — | +| GET | /api/knowledge/{source}/versions/{version:\d+} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L244) | Yes | — | +| POST | /api/knowledge/{source}/versions/{version:\d+}/restore | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L256) | Yes | — | +| GET | /api/knowledge/{source}/versions/{v1:\d+}/diff/{v2:\d+} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L279) | Yes | — | +| POST | /api/knowledge/import | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L295) | Yes | — | | GET | /api/schedules/status | [src.web.api.schedules_api](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/schedules_api.py#L39) | Yes | — | | GET | /api/schedules | [src.web.api.schedules_api](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/schedules_api.py#L43) | Yes | — | | POST | /api/schedules | [src.web.api.schedules_api](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/schedules_api.py#L47) | Yes | — | @@ -159,12 +159,12 @@ This describes the normal authenticated deployment. With no configured tokens (i | GET | /api/audit/verify | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L340) | Yes | — | | GET | /api/logs/search | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L353) | Yes | — | | GET | /api/logs/stats | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L380) | Yes | — | -| GET | /api/memory | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L375) | Yes | — | -| GET | /api/memory/{scope} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L396) | Yes | Every key/value in one scope, in ONE request. | -| GET | /api/memory/{scope}/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L417) | Yes | — | -| PUT | /api/memory/{scope}/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L433) | Yes | — | -| DELETE | /api/memory/{scope}/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L457) | Yes | — | -| POST | /api/memory/bulk-delete | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L476) | Yes | — | +| GET | /api/memory | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L376) | Yes | — | +| GET | /api/memory/{scope} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L397) | Yes | Every key/value in one scope, in ONE request. | +| GET | /api/memory/{scope}/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L418) | Yes | — | +| PUT | /api/memory/{scope}/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L434) | Yes | — | +| DELETE | /api/memory/{scope}/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L458) | Yes | — | +| POST | /api/memory/bulk-delete | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L477) | Yes | — | | GET | /api/risk/stats | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L392) | Yes | — | | GET | /api/risk/recent | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L399) | Yes | — | | GET | /api/governor/stats | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L410) | Yes | — | @@ -234,9 +234,9 @@ This describes the normal authenticated deployment. With no configured tokens (i | GET | /api/freshness/stats | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L459) | Yes | — | | GET | /api/freshness/recent | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L466) | Yes | — | | GET | /api/validation/stats | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L481) | Yes | — | -| GET | /api/learned | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L524) | Yes | — | -| DELETE | /api/learned/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L530) | Yes | — | -| PUT | /api/learned/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L537) | Yes | — | +| GET | /api/learned | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L525) | Yes | — | +| DELETE | /api/learned/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L531) | Yes | — | +| PUT | /api/learned/{key} | [src.web.api.knowledge_mem](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/knowledge_mem.py#L538) | Yes | — | | GET | /api/affordances | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L495) | Yes | — | | GET | /api/compression/stats | [src.web.api.observability](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/observability.py#L507) | Yes | — | | GET | /api/startup/diagnostics | [src.web.api.config_admin](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/web/api/config_admin.py#L1214) | Yes | — | @@ -266,9 +266,9 @@ Registered by [HealthServer](https://github.com/Calmingstorm/Odin/blob/e1318eca8 | GET | /metrics | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1350) | HealthServer construction; no API authentication | Prometheus metrics endpoint. | | POST | /webhook/gitea | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1444) | webhooks.enabled; handler verifies webhook signature/shared secret | — | | POST | /webhook/grafana | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1492) | webhooks.enabled; handler verifies webhook signature/shared secret | — | -| POST | /webhook/generic | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1566) | webhooks.enabled; handler verifies webhook signature/shared secret | — | -| POST | /webhook/github | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1589) | webhooks.enabled; handler verifies webhook signature/shared secret | — | -| POST | /webhook/gitlab | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1662) | webhooks.enabled; handler verifies webhook signature/shared secret | — | +| POST | /webhook/generic | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1567) | webhooks.enabled; handler verifies webhook signature/shared secret | — | +| POST | /webhook/github | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1590) | webhooks.enabled; handler verifies webhook signature/shared secret | — | +| POST | /webhook/gitlab | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1663) | webhooks.enabled; handler verifies webhook signature/shared secret | — | | GET | / | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1051) | web.enabled + UI directory exists; no API authentication | Redirect / to /ui/. | | GET | /ui/{path:.*} | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1055) | web.enabled + UI directory exists; no API authentication | Serve static UI files, defaulting to index.html for SPA routing. | | GET | /ui | [src.health.server](https://github.com/Calmingstorm/Odin/blob/e1318eca83a40b9418d873ad85c31686ea6b57d0/src/health/server.py#L1051) | web.enabled + UI directory exists; no API authentication | Redirect / to /ui/. | diff --git a/docs/reference/tools.md b/docs/reference/tools.md index 88909d65b..408acda78 100644 --- a/docs/reference/tools.md +++ b/docs/reference/tools.md @@ -575,7 +575,7 @@ Source: [`src/tools/defs/browser_web.py`](https://github.com/Calmingstorm/Odin/b **Core:** No -
Navigates to a URL and clicks an element by CSS selector. Returns a confirmation summary after clicking. To fill forms, use browser_fill. To read page content after clicking, follow up with browser_read_page.+
Loads a URL in a fresh browser session, clicks an element by CSS selector, and returns the resulting page's title and URL. Cookies, storage and page state end with the call; a later browser_read_page reloads the URL without them. To fill and submit in one call, use browser_fill with submit=true.
[affordances: cost=high risk=high latency=seconds] (requires: browser enabled; installed Chromium or reachable configured CDP endpoint)
@@ -589,7 +589,7 @@ Source: [`src/tools/defs/browser_web.py`](https://github.com/Calmingstorm/Odin/b **Core:** No -Navigates to a URL and fills a form field by CSS selector. Optionally submits by pressing Enter. To click buttons, use browser_click.+
Loads a URL in a fresh browser session, fills one field by CSS selector, optionally presses Enter (submit=true), and returns the page's title and URL. The value is gone when the call ends, so a later browser_click cannot submit it; use submit=true, or one browser_evaluate for several fields.
[affordances: cost=high risk=high latency=seconds] (requires: browser enabled; installed Chromium or reachable configured CDP endpoint)
@@ -604,7 +604,7 @@ Source: [`src/tools/defs/browser_web.py`](https://github.com/Calmingstorm/Odin/b **Core:** No -Evaluates JavaScript on a URL and returns the result. For custom scraping or interaction. Large results have retained previews; use get_tool_output(cursor=...) without re-running the expression.+
Loads a URL in a fresh browser session, evaluates a JavaScript expression, and returns its result (a returned Promise is awaited). Use it for scraping or several interaction steps in one call; the session closes when it returns, so a navigation it starts may not complete. Large results have retained previews; use get_tool_output(cursor=...) without re-running the expression.
[affordances: cost=high risk=high latency=seconds] (requires: browser enabled; installed Chromium or reachable configured CDP endpoint)
diff --git a/docs/scheduling.md b/docs/scheduling.md new file mode 100644 index 000000000..3a948d4b1 --- /dev/null +++ b/docs/scheduling.md @@ -0,0 +1,95 @@ +# Schedules & webhooks + +Schedules can run checks, multi-step workflows, reminders, digests and HTTP +webhook actions. An inbound webhook can also trigger a schedule. These are two +different paths: an outbound notifier is a separate subsystem, not covered here. + +## Who scheduled work runs as + +Discord-created schedules run as their creator. System schedules created through +the WebUI, REST API or a skill without a requester run as `scheduler` with the +admin tool tier. Host access is *still enforced*: a Host Access entry named +`scheduler` supplies its allowed hosts and default host, if present; otherwise +the Default Policy applies. To create that entry (the WebUI Add user picker only +accepts numeric Discord IDs), use `PUT /api/host-access/user/scheduler` with +`{"allowed_hosts":["your-host"],"default_host":""}`. It then appears on +System → Host Access. Widening the Default Policy grants access to every +unlisted user, not just system schedules. Digests report denied or failed host +probes as collection failures; if every probe fails, the digest run fails. + +## Interrupted one-time runs + +A side-effecting one-time `check`, `workflow` or `webhook` run is marked as +started **before** the first effect. If Odin stops during the run, or cannot +save its completion, the schedule becomes paused and **inert** on restart or +the next scheduler tick. It is not automatically run again from step one. +The reason is shown in the schedule listing, REST response and WebUI. Check +what already happened, then set a **new `run_at`** to re-arm it. Merely +unpausing it is refused. A run that never began, including a queued run, +remains eligible; a normally recorded failure still follows its configured +retry behaviour. One-time reminders and digests have no start marker and +still replay after an interruption; recurring and trigger schedules retain +their existing cadence. An interruption may happen after a marker is saved +but before an effect starts. Quarantine chooses safety over automatic replay. + +## Scheduled webhook actions + +- Each run sends one request through a shared aiohttp session. Methods are + `GET`, `POST` (default), `PUT`, `PATCH`, `DELETE` and `HEAD`; the WebUI does + not offer `HEAD`. A dict or list body is sent as JSON, any other body as + text. JSON-looking text entered in the WebUI is sent as text unless you set + its `Content-Type` header. +- The total timeout defaults to 30 seconds and can be set up to 300 seconds. + Redirects are followed, up to ten. The request does not need an active + Discord connection. +- With `expected_status_codes` **omitted or empty (`[]`)**, *any* HTTP + response counts as success, even 4xx or 5xx. If specified, a status outside + the list fails with `Webhook returned status X, expected one of […]`. + DNS, connection, TLS, timeout and redirect failures also fail. An undecodable + response body (using its declared charset, UTF-8 by default) can mark an + already-delivered request as failed. +- History records success or failure, duration and a failure error (up to 500 + characters), along with `last_error` and `consecutive_failures`. Neither + the response status nor body is stored; success status is logged. +- Retries default to **off** (`max_retries: 0`). REST can set the retry count; + the WebUI cannot. Delays are `retry_backoff_seconds` (default 60) times + `2^(attempt−1)`, capped at 3600 seconds. A retry sends the **entire** request + again. Delivery is at least once; Odin provides no idempotency key. A cron + slot missed during a pending retry runs once immediately afterward, then + normal cadence resumes. +- A successful one-time webhook is removed. On a recorded final failure, it + remains without `next_run` until a manual run or a new `run_at`. If its + completion was **not recorded** after starting, it instead becomes inert + under the interrupted-run rule above. Verify external effects before + re-arming it. +- Every third consecutive failure sends a Discord alert, provided a channel + and connection are available. A manual run returns `{"status":"success"}` + or `{"status":"failure","error":"…"}`. Overlapping runs of the same + schedule are dropped. + +## Inbound webhooks + +Endpoints are `/webhook/gitea`, `/webhook/github`, `/webhook/gitlab`, +`/webhook/grafana` and `/webhook/generic`, registered only with +`webhook.enabled`. Gitea and GitHub use HMAC authentication; GitLab, Grafana +and generic use a shared token. Authentication fails closed with 403 when +no secret is configured. Invalid JSON receives 400. The body limit is +10 MiB, and there is no per-endpoint rate limit. + +Processing order: parse the request, run Grafana remediation where applicable, +run matching schedules **to completion**, then post the channel notification +and respond. A slow action can outlast the sender's timeout; the sender may +retry while Odin is still working. A named Grafana trigger compares its +case-insensitive alert-name substring against **every** alert in a notification, +including resolved alerts. Each schedule fires at most once per delivery, +even if several alerts match. Unnamed triggers also fire once per delivery. + +The HTTP response describes **the notification only**, not whether triggered +actions succeeded: 200 `{"status":"delivered"}` means notification posted; +500 means no channel was configured or sending failed; 503 means the bot was +not ready to send. Schedules already fired even if this final delivery fails. +Trigger machinery errors are logged without changing the response; individual +action failures appear in that schedule's history. A sender retrying after a +5xx can repeat the effects. Redeliveries are not deduplicated. While Discord +is disconnected, non-webhook triggered schedules and paused schedules are +skipped rather than queued. diff --git a/package.json b/package.json index f847d8bd2..8f1a6dea7 100644 --- a/package.json +++ b/package.json @@ -11,15 +11,17 @@ "check:templates": "node scripts/check-vue-templates.mjs", "check:codex-quota-ui": "node scripts/check-codex-quota-ui.mjs", "check:llm-config-refresh": "node --experimental-vm-modules scripts/check-llm-config-refresh.mjs", + "check:llm-provider-visibility": "node scripts/check-llm-provider-visibility-browser.mjs", "check:setup-ui": "node scripts/check-setup-ui.mjs", "check:template-bindings": "node scripts/check-template-bindings.mjs", "check:modal-race": "node scripts/check-modal-race.mjs", - "check": "npm run check:templates && npm run check:codex-quota-ui && npm run check:llm-config-refresh && npm run check:template-bindings && npm run check:permission-token-repair && npm run check:listener-consent && npm run check:w1-webui && npm run check:modal-race && npm run check:host-access-races && npm run check:host-enrollment && npm run check:socket-identity && npm run check:ws-lifecycle && npm run check:w2-freshness && npm run check:toggle-serialization && npm run check:search-races && npm run check:config-keydown && npm run check:css-tokens && npm run check:turn-state-webui && npm run check:audit-verify && npm run check:schedule-time && npm run check:schedule-report-format && npm run check:config-health && npm run check:config-center-ui2 && npm run check:mcp-webui && npm run check:tools-webui && npm run check:learning-toggle && npm run check:config-save-boundaries && npm run check:usage-activity && npm run check:output-renderer && npm run check:live-logs && npm run check:logs-preset-search-races && npm run check:discord-identity && npm run build", + "check": "npm run check:templates && npm run check:codex-quota-ui && npm run check:llm-config-refresh && npm run check:llm-provider-visibility && npm run check:template-bindings && npm run check:permission-token-repair && npm run check:listener-consent && npm run check:w1-webui && npm run check:modal-race && npm run check:host-access-races && npm run check:host-enrollment && npm run check:socket-identity && npm run check:ws-lifecycle && npm run check:login-persistence && npm run check:w2-freshness && npm run check:toggle-serialization && npm run check:search-races && npm run check:config-keydown && npm run check:css-tokens && npm run check:turn-state-webui && npm run check:audit-verify && npm run check:schedule-time && npm run check:schedule-report-format && npm run check:config-health && npm run check:config-center-ui2 && npm run check:mcp-webui && npm run check:tools-webui && npm run check:learning-toggle && npm run check:config-save-boundaries && npm run check:usage-activity && npm run check:output-renderer && npm run check:live-logs && npm run check:logs-preset-search-races && npm run check:discord-identity && npm run build", "check:listener-consent": "node scripts/check-listener-consent.mjs", "check:host-access-races": "node scripts/check-host-access-races.mjs && node scripts/check-host-access-redesign.mjs && node scripts/check-host-access-modal.mjs", "check:host-enrollment": "node scripts/check-host-enrollment-browser.mjs", "check:socket-identity": "node scripts/check-socket-identity.mjs", "check:ws-lifecycle": "node scripts/check-ws-lifecycle.mjs", + "check:login-persistence": "node scripts/check-login-persistence.mjs", "check:schedule-time": "node scripts/check-schedule-time.mjs", "check:config-health": "node scripts/check-config-health.mjs", "check:config-center-ui2": "node scripts/check-config-center-ui2.mjs", diff --git a/scripts/check-audit-verify.mjs b/scripts/check-audit-verify.mjs index 41700c26e..d79b1215b 100644 --- a/scripts/check-audit-verify.mjs +++ b/scripts/check-audit-verify.mjs @@ -71,25 +71,73 @@ async function renderState(state) { return html; } -// Valid chain with a permanent unsigned prefix: green verdict, and the -// prefix explained as history — the words "expected, not tampering" carry it. +const seg = (file, position, status, extra = {}) => ({ + file, position, status, total: 0, verified: 0, unsigned_prefix: 0, + first_bad: null, reason: null, error: null, ...extra, +}); + +// All retained files are shown, including one written before signing. { - const state = setupPage({ valid: true, availability: 'available', total: 500, verified: 420, unsigned_prefix: 80, first_bad: null }); + const state = setupPage({ + valid: true, availability: 'available', scope: 'retained_files', total: 600, verified: 420, + unsigned_prefix: 180, first_bad: null, first_bad_file: null, + segments: [ + seg('audit.jsonl', 0, 'verified', { total: 320, verified: 320 }), + seg('audit.jsonl.1', 1, 'verified', { total: 180, verified: 100, unsigned_prefix: 80 }), + seg('audit.jsonl.2', 2, 'unsigned', { total: 100, unsigned_prefix: 100 }), + ], + }); await state.verifyIntegrity(); const html = await renderState(state); - assert.match(html, /Chain valid — 420 signed entries verified/); - assert.match(html, /80 older entries predate signing/); + assert.match(html, /Chain valid — 420 signed entries verified across 3 retained files/); + assert.match(html, /180 older entries predate signing/); assert.match(html, /expected, not tampering/); + assert.match(html, /audit\.jsonl<\/span> \(current\): verified — 320 signed entries/); + assert.match(html, /audit\.jsonl\.1<\/span>: verified — 100 signed entries; 80 older entries predate signing/); + assert.match(html, /audit\.jsonl\.2<\/span>: no signatures — written before signing was enabled/); + assert.match(html, /cannot be detected/); assert.ok(!/Chain INVALID/.test(html)); } -// Broken chain arrives as a 409 with the verifier's structured verdict. +// Rotated break arrives as a 409 identifying the file and line. { - const state = setupPage({ valid: false, availability: 'available', total: 500, verified: 12, unsigned_prefix: 0, first_bad: 13, error: 'Line 13: HMAC verification failed' }, 409); + const state = setupPage({ + valid: false, availability: 'available', scope: 'retained_files', total: 500, verified: 412, + unsigned_prefix: 0, first_bad: 13, first_bad_file: 'audit.jsonl.1', + error: 'audit.jsonl.1: Line 13: HMAC verification failed (tampered or reordered)', + segments: [ + seg('audit.jsonl', 0, 'verified', { total: 400, verified: 400 }), + seg('audit.jsonl.1', 1, 'broken', { total: 13, verified: 12, first_bad: 13, reason: 'hmac_mismatch' }), + ], + }, 409); await state.verifyIntegrity(); assert.equal(state.verifyResult.value.valid, false); const html = await renderState(state); - assert.match(html, /Chain INVALID — first break at entry 13; 12 verified before it/); + assert.match(html, /Chain INVALID — problem in audit\.jsonl\.1 at line 13/); + assert.match(html, /break at line 13 — entry altered, reordered, or signed with a different key; 12 entries verified before it; later lines not checked/); + assert.match(html, /audit\.jsonl<\/span> \(current\): verified — 400 signed entries/); +} + +// Unreadable, missing and file-level signing gap are findings, never skipped. +{ + const state = setupPage({ + valid: false, availability: 'available', scope: 'retained_files', total: 3, verified: 2, + unsigned_prefix: 1, first_bad: null, first_bad_file: 'audit.jsonl.1', + error: 'audit.jsonl.1: no signatures although older files are signed', + segments: [ + seg('audit.jsonl', 0, 'verified', { total: 1, verified: 1 }), + seg('audit.jsonl.1', 1, 'broken', { total: 1, unsigned_prefix: 1, reason: 'signing_gap' }), + seg('audit.jsonl.2', 2, 'missing', { error: 'expected file not found' }), + seg('audit.jsonl.3', 3, 'unreadable', { error: 'PermissionError' }), + seg('audit.jsonl.4', 4, 'verified', { total: 1, verified: 1 }), + ], + }, 409); + await state.verifyIntegrity(); + const html = await renderState(state); + assert.match(html, /Chain INVALID — problem in audit\.jsonl\.1\./); + assert.match(html, /no signatures although older files are signed/); + assert.match(html, /audit\.jsonl\.2<\/span>: missing/); + assert.match(html, /audit\.jsonl\.3<\/span>: could not be read \(PermissionError\)/); } // Signing not enabled is a configuration fact, not an alarm. @@ -120,4 +168,4 @@ const auditSource = readFileSync(new URL('../ui/js/pages/audit.js', import.meta. assert.match(auditSource, /e\.data\.availability === 'not_enabled'/); assert.doesNotMatch(auditSource, /e\.data\.error\s*\?\s*\{[^}]*not_enabled/s); -console.log('audit-verify: all four verifier states and the honest-prefix copy pinned'); +console.log('audit-verify: every verifier state, per-file findings and honest-prefix copy pinned'); diff --git a/scripts/check-llm-config-refresh.mjs b/scripts/check-llm-config-refresh.mjs index cfd120745..0ab35799f 100644 --- a/scripts/check-llm-config-refresh.mjs +++ b/scripts/check-llm-config-refresh.mjs @@ -3,6 +3,8 @@ import fs from 'node:fs'; import vm from 'node:vm'; const source = fs.readFileSync('ui/js/pages/llm-config.js', 'utf8'); +// Run the same request-budget contract with both saved provider states. +for (const disabled of [false, true]) { const requests = []; const mounted = []; let timerCallback = null; @@ -14,9 +16,9 @@ const api = { if (path === '/api/llm/status') return { main_model: 'gpt-5.6-sol', codex: { enabled: true, model: 'gpt-5.6-sol' }, - ollama: { enabled: true, base_url: 'http://localhost:11434', model: 'local' }, + ollama: { enabled: !disabled, base_url: 'http://localhost:11434', model: 'local' }, openai_compatible: { - enabled: true, base_url: 'https://openrouter.ai/api/v1', model: 'vendor/model', + enabled: !disabled, base_url: 'https://openrouter.ai/api/v1', model: 'vendor/model', model_profiles: {}, openrouter: {}, }, }; @@ -53,6 +55,7 @@ const modules = new Map([ ['../confirm.js', synthetic('confirm', { async confirmDialog() { return false; } })], ['vue', synthetic('vue', { computed(fn) { return { get value() { return fn(); } }; }, + watch() {}, onActivated() {}, onDeactivated() {}, onMounted(fn) { mounted.push(fn); }, onUnmounted() {}, ref(value) { return { value }; }, })], @@ -76,20 +79,23 @@ await page.evaluate(); page.namespace.default.setup(); assert.equal(mounted.length, 1, 'LLM config must register one mount callback'); mounted[0](); -for (let attempt = 0; attempt < 20 && !requests.includes('/api/openrouter/catalogue'); attempt += 1) { +for (let attempt = 0; attempt < 20 && !requests.includes('/api/context/windows'); attempt += 1) { await new Promise(resolve => setTimeout(resolve, 0)); } assert.equal(typeof timerCallback, 'function', 'mount must arm the live-refresh timer'); assert.equal(timerInterval, 15000, 'live refresh must run no more often than every 15 seconds'); for (let attempt = 0; attempt < 3; attempt += 1) await new Promise(resolve => setTimeout(resolve, 0)); -for (const catalogue of ['/api/openrouter/catalogue', '/api/openai-compatible/models', '/api/ollama/models']) { +for (const catalogue of ['/api/openai-compatible/models', '/api/ollama/models']) { assert.equal(requests.filter(path => path === catalogue).length, 1, `${catalogue} must load once on mount`); } +assert.equal(requests.filter(path => path === '/api/openrouter/catalogue').length, disabled ? 0 : 1, + 'OpenRouter catalogue fetched on mount only when compatible is saved enabled'); requests.length = 0; await timerCallback(); assert.deepEqual(requests, [ '/api/llm/status', '/api/ollama/status', '/api/openai-compatible/status', '/api/agents/model', '/api/codex/status', '/api/context/windows', -], 'one live-refresh cycle must cost exactly six status/config requests'); -console.log('LLM config refresh request budget OK'); +], 'one live-refresh cycle must cost exactly six status/config requests, with no catalogue downloads'); +} +console.log('LLM config refresh request budget OK (enabled and disabled provider fixtures)'); diff --git a/scripts/check-llm-provider-visibility-browser.mjs b/scripts/check-llm-provider-visibility-browser.mjs new file mode 100644 index 000000000..4f56bc475 --- /dev/null +++ b/scripts/check-llm-provider-visibility-browser.mjs @@ -0,0 +1,294 @@ +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import { chromium } from 'playwright-core'; +import { createServer } from 'vite'; + +// Real Vue DOM against an isolated API fixture; no live settings or server calls. +const server = await createServer({ configFile:false, root:process.cwd(), appType:'custom', + resolve:{alias:{vue:'vue/dist/vue.esm-bundler.js'}}, + define:{__VUE_OPTIONS_API__:'true',__VUE_PROD_DEVTOOLS__:'false',__VUE_PROD_HYDRATION_MISMATCH_DETAILS__:'false'}, + server:{host:'127.0.0.1',port:0,watch:null} }); +const html = ``; +server.middlewares.use(async(req,res,next)=>{ + if(req.url!=='/__llm_visibility__.html')return next(); + res.setHeader('Content-Type','text/html');res.end(await server.transformIndexHtml(req.url,html)); +}); +const compat={enabled:false,configured:true,base_url:'https://openrouter.ai/api/v1',model:'vendor/alpha',model_profiles:{},openrouter:{}}; +const ollama={enabled:false,configured:true,base_url:'http://localhost:11434',model:'quartz'}; +const codex={enabled:true,configured:true,model:'gpt-6-sol',reasoning_effort:'high'}; +const entries=['compat:vendor/alpha',{model:'ollama:quartz',thinking_mode:'enabled'},'gpt-6-sol']; +const agents={model:'compat:vendor/alpha',auto_model_allowlist:entries,model_selection_hints:{'compat:vendor/alpha':'First for code'}}; +let main='compat:vendor/alpha'; +let failSave=false,holdCatalogue=null; +const writes=[]; +const catalog=()=>({ + codex:['gpt-6-astra','gpt-6-sol','gpt-6-luna','gpt-5.6-sol'].map(name=>({ref:name,name,provider:'codex',available:true,capability:'reasoning'})), + compat:[{ref:'compat:vendor/alpha',name:'vendor/alpha',provider:'compat',available:compat.enabled,unavailable_reason:compat.enabled?'':'disabled',capability:'reasoning',agent_available:true}], + ollama:[{ref:'ollama:quartz',name:'quartz',provider:'ollama',available:ollama.enabled,unavailable_reason:ollama.enabled?'':'disabled',capability:'none',agent_available:true}], +}); +const requests=[],errors=[],unexpected=[]; +let browser; +try{ + await server.listen(); + const executablePath=[process.env.CHROME_PATH,'/usr/bin/google-chrome','/usr/bin/chromium'].filter(Boolean).find(fs.existsSync); + assert.ok(executablePath,'Chromium is required'); + browser=await chromium.launch({executablePath,headless:true,args:['--no-sandbox']}); + const page=await browser.newPage();page.on('pageerror',e=>errors.push(e.message)); + await page.route('**/api/**',async route=>{ + const req=route.request(),path=new URL(req.url()).pathname; + requests.push(`${req.method()} ${path}`); + let body={}; + if(req.method()==='GET'){ + if(path==='/api/llm/status')body={main_model:main,codex:structuredClone(codex),openai_compatible:structuredClone(compat),ollama:structuredClone(ollama),auxiliary:{enabled:true,model:'ollama:quartz'},model_catalogue:catalog()}; + else if(path==='/api/ollama/status')body={configured:true,model:'quartz',health:{healthy:false}}; + else if(path==='/api/ollama/models')body={models:[{name:'quartz'}]}; + else if(path==='/api/openai-compatible/status')body={configured:true,model:'vendor/alpha',health:{healthy:false}}; + else if(path==='/api/openai-compatible/models')body={models:[{name:'vendor/alpha'}]}; + else if(path==='/api/agents/model')body=structuredClone(agents); + else if(path==='/api/codex/status')body={configured:true,accounts:[]}; + else if(path==='/api/context/windows')body={models:{}}; + else if(path==='/api/openrouter/catalogue'){ + if(holdCatalogue)await holdCatalogue; + body={models:[{id:'vendor/alpha',name:'Alpha Display Name',vendor:'vendor',supports_tools:true,supports_reasoning:true,supported_efforts:['high'],agent_eligible:true,variant:'standard',profile:{total_window_tokens:131072,max_output_tokens:8192}}]}; + } + else unexpected.push(`${req.method()} ${path}`); + }else if(req.method()==='PUT'){ + const payload=req.postDataJSON();writes.push({path,payload}); + if(failSave)body={error:'fixture save failure'}; + else if(path==='/api/openai-compatible/config')Object.assign(compat,payload); + else if(path==='/api/llm/ollama/config')Object.assign(ollama,payload); + else if(path==='/api/agents/model')Object.assign(agents,payload); + else if(path==='/api/llm/main-model')main=payload.model; + else unexpected.push(`${req.method()} ${path}`); + }else unexpected.push(`${req.method()} ${path}`); + await route.fulfill({status:failSave&&req.method()==='PUT'?503:200,contentType:'application/json',body:JSON.stringify(body)}); + }); + await page.goto(`http://127.0.0.1:${server.httpServer.address().port}/__llm_visibility__.html`); + await page.waitForFunction(()=>window.view&&!view.loading); + assert.deepEqual(errors,[],'component must render without runtime errors'); + const mainSelect=page.locator('label').filter({hasText:'Main model'}).first().locator('select').first(); + const agentSelect=page.locator('label').filter({hasText:'Agent model'}).first().locator('select').first(); + const auxSelect=page.locator('label').filter({hasText:'Auxiliary model'}).first().locator('select').first(); + const selectors=[mainSelect,agentSelect,auxSelect]; + assert.equal(await page.evaluate(()=>view.savedProviderEnabled('compat')),false); + async function checkDisabled(){ + for(const [selector,value] of [[mainSelect,'compat:vendor/alpha'],[agentSelect,'compat:vendor/alpha'],[auxSelect,'ollama:quartz']]){ + assert.equal(await selector.inputValue(),value,'saved selection survives disabling'); + const option=selector.locator(`option[value="${value}"]`); + assert.equal(await option.count(),1,'selected-disabled sentinel appears exactly once'); + assert.equal(await option.evaluate(el=>el.disabled),true,'selected sentinel itself is disabled'); + assert.match(await option.innerText(),/configured:.*provider disabled/i); + } + for(const [index,selector] of selectors.entries()){ + assert.equal(await selector.locator('optgroup[label="OpenAI-compatible"], optgroup[label="Ollama"]').count(),0); + assert.equal(await selector.locator('option[value="compat:vendor/alpha"]').count(),index===2?0:1); + assert.equal(await selector.locator('option[value="ollama:quartz"]').count(),index===2?1:0); + } + } + await checkDisabled(); + assert.equal(requests.filter(x=>x==='GET /api/openrouter/catalogue').length,0,'disabled compat must not fetch OpenRouter'); + await page.getByRole('button',{name:'Configure agent allowlist'}).click(); + const dialog=page.getByRole('dialog'); + assert.match(await dialog.innerText(),/excluded: provider disabled/i); + assert.match(await dialog.innerText(),/configured.*effective/i); + assert.deepEqual(await page.evaluate(()=>view.effectiveAllowlist),['compat:vendor/alpha','ollama:quartz','gpt-6-sol']); + await dialog.getByRole('button',{name:'Close'}).click(); + await page.evaluate(()=>{view.modelSelectorSearch='quartz';}); + await checkDisabled(); + assert.equal(await agentSelect.locator('option[value="auto"]').count(),1); + assert.equal(await agentSelect.locator('option[value=""]').count(),1); + assert.equal(await auxSelect.locator('option[value=""]').count(),1); + assert.equal(await page.getByRole('textbox',{name:'Search model catalogue'}).count(),0); + await page.evaluate(()=>{view.compatibleForm.enabled=true;view.ollamaForm.enabled=true;}); + await checkDisabled(); // unsaved drafts are not authoritative + // The plain Codex-only list must not disappear on an arbitrary stale search query. + assert.equal(await mainSelect.locator('optgroup').count(),0); + assert.equal(await mainSelect.locator('option[value="gpt-6-sol"]').count(),1); + assert.equal(await agentSelect.locator('option[value="gpt-6-astra"]').count(),1); + // Explicit successful save can change provider visibility, preserving unrelated advanced drafts. + await page.evaluate(()=>{view.compatibleForm.model_profiles={other:{supports_reasoning:true}};}); + await page.evaluate(async()=>await view.saveCompatibleConfig()); + await page.evaluate(()=>{view.modelSelectorSearch='';}); + assert.equal(compat.enabled,true); + assert.equal(await mainSelect.locator('option[value="compat:vendor/alpha"]').count(),1); + assert.deepEqual(await page.evaluate(()=>view.disableWarningRoles('compat')),['Main','Agent']); + await page.evaluate(()=>{view.agentsConfig.model='';}); + assert.deepEqual(await page.evaluate(()=>view.disableWarningRoles('compat')),['Main','Agent'], + 'an inherited Agent also depends on Main'); + await page.evaluate(()=>{view.agentsConfig.model='compat:vendor/alpha';}); + assert.deepEqual(await page.evaluate(()=>view.compatibleForm.model_profiles),{other:{supports_reasoning:true}}); + assert.equal(requests.filter(x=>x==='GET /api/openrouter/catalogue').length,1); + // A failed disable cannot hide saved-compatible choices; the operator's pending draft survives. + failSave=true; + await page.evaluate(()=>{view.compatibleForm.enabled=false;}); + const warning=page.getByRole('dialog',{name:'Disable OpenAI-compatible?'}); + const failedSave=page.evaluate(async()=>await view.saveCompatibleConfig()); + await warning.waitFor(); + assert.match(await warning.innerText(),/Main.*Agent/); + await warning.getByRole('button',{name:'Disable provider'}).click(); + await failedSave; + assert.equal(compat.enabled,true); + assert.equal(await mainSelect.locator('option[value="compat:vendor/alpha"]').count(),1); + assert.equal(await page.evaluate(()=>view.compatibleForm.enabled),false); + failSave=false; + const successfulDisable=page.evaluate(async()=>await view.saveCompatibleConfig()); + await warning.waitFor(); + await warning.getByRole('button',{name:'Disable provider'}).click(); + await successfulDisable; + await page.waitForFunction(()=>!view.savingCompatible); + assert.equal(compat.enabled,false); + await checkDisabled(); + // Empty allowlist catalogue and an unknown saved ref must not silently replace the fixed selection. + const originalEntries=structuredClone(agents.auto_model_allowlist); + agents.model='compat:future/model'; + agents.auto_model_allowlist=['compat:future/model']; + await page.evaluate(async()=>await view.fetchAll()); + assert.equal(await agentSelect.locator('option[value="compat:future/model"]').count(),1); + assert.equal(await agentSelect.inputValue(),'compat:future/model'); + assert.match(await agentSelect.locator('option[value="compat:future/model"]').innerText(),/configured:.*provider disabled/i); + assert.equal(await page.evaluate(()=>view.effectiveAutoAllowlistCount),0,'disabled-only Agent Auto has no effective choices'); + agents.model='auto'; + await page.evaluate(async()=>await view.fetchAll()); + assert.match(await page.getByText(/1 configured, 0 effective models/).first().innerText(),/1 configured, 0 effective models/); + compat.enabled=true; + await page.evaluate(async()=>await view.fetchLLMStatus()); + const autoWarning=await page.evaluate(()=>view.disableWarningRoles('compat')); + assert.ok(autoWarning.includes('Agent Auto (zero effective choices)'), + 'disabling the last Agent Auto provider warns before saving'); + assert.ok(!autoWarning.includes('Agent'),'Auto is not a fixed provider reference'); + compat.enabled=false; + await page.evaluate(async()=>await view.fetchLLMStatus()); + agents.model='compat:vendor/alpha';agents.auto_model_allowlist=originalEntries; + await page.evaluate(async()=>await view.fetchAll()); + // Enabling both providers by status (health remains unreachable) exposes their disabled options. + compat.enabled=true;ollama.enabled=true; + await page.evaluate(async()=>await view.fetchLLMStatus()); + await page.evaluate(()=>{view.modelSelectorSearch='';}); + await page.waitForFunction(()=>view.modelGroups.some(g=>g.id==='compat')&&view.modelGroups.some(g=>g.id==='ollama')); + for(const selector of selectors){ + assert.equal(await selector.locator('optgroup[label="OpenAI-compatible"]').count(),1); + assert.equal(await selector.locator('optgroup[label="Ollama"]').count(),1); + } + assert.equal(await mainSelect.locator('option[value="compat:vendor/alpha"]').count(),1); + assert.equal(await auxSelect.locator('option[value="ollama:quartz"]').count(),1); + assert.deepEqual(agents.auto_model_allowlist,entries); + // Saved provider is enabled but health is false: preserve current server-marked availability. + assert.equal(await mainSelect.locator('option[value="compat:vendor/alpha"]').evaluate(el=>el.disabled),false); + assert.equal(await page.evaluate(()=>view.savedProviderEnabled('compat')),true); + assert.equal(await page.evaluate(()=>view.savedProviderEnabled('ollama')),true); + await page.evaluate(async()=>await view.fetchAll()); + await page.waitForFunction(()=>view.openRouterCatalogue?.models?.length); + await page.evaluate(()=>{ + view.llmStatus.model_catalogue.compat[0].capability='none'; + view.llmStatus.model_catalogue.compat[0].efforts=['low']; + }); + for(const selector of selectors){ + assert.match(await selector.locator('option[value="compat:vendor/alpha"]').innerText(),/^Alpha Display Name/, + 'enabled provider retains the OpenRouter display name'); + } + assert.deepEqual(await page.evaluate(()=>{ + const model=view.modelCatalog.find(m=>m.ref==='compat:vendor/alpha'); + return [model.name,model.capability,model.efforts,model.agent_available]; + }),['Alpha Display Name','reasoning',['high'],true],'presentation metadata retains catalogue precedence'); + // Bare unknown Codex names remain Codex selections, not mistaken for provider IDs. + codex.enabled=false;agents.model='auto'; + await page.evaluate(async()=>await view.fetchAll()); + assert.equal(await agentSelect.inputValue(),'auto'); + assert.equal(await agentSelect.locator('option[value="auto"]').count(),1,'Auto has only its fixed option'); + assert.equal(await page.evaluate(()=>view.configuredDisabledSelection('agent')),false); + assert.equal(await page.getByText(/Agent model is configured on a disabled provider/).count(),0); + assert.ok(!(await page.evaluate(()=>view.disableWarningRoles('codex'))).includes('Agent'), + 'disabling Codex must not claim Agent Auto is a fixed Codex selection'); + agents.model='gpt-7-future'; + await page.evaluate(async()=>await view.fetchAll()); + const future=agentSelect.locator('option[value="gpt-7-future"]'); + assert.equal(await future.count(),1); + assert.match(await future.innerText(),/configured:.*provider disabled/i); + assert.equal(await future.evaluate(el=>el.disabled),true); + assert.equal(await agentSelect.locator('optgroup[label="Codex"]').count(),0); + codex.enabled=true;agents.model='compat:vendor/alpha'; + await page.evaluate(async()=>await view.fetchAll()); + // Health is separate from saved enablement: server-marked unreachable entries remain listed. + await page.evaluate(()=>{ + for(const provider of ['compat','ollama']){ + view.llmStatus.model_catalogue[provider][0].available=false; + view.llmStatus.model_catalogue[provider][0].unavailable_reason='unreachable'; + } + }); + assert.equal(await mainSelect.locator('optgroup[label="OpenAI-compatible"] option').count(),1); + assert.match(await mainSelect.locator('option[value="compat:vendor/alpha"]').innerText(),/unreachable/); + assert.equal(await mainSelect.locator('option[value="compat:vendor/alpha"]').evaluate(el=>el.disabled),true); + await page.waitForFunction(()=>view.openRouterCatalogue?.models?.length); + await page.evaluate(()=>{view.llmStatus.model_catalogue.compat[0].agent_available=false;}); + for(const selector of selectors){ + assert.match(await selector.locator('option[value="compat:vendor/alpha"]').innerText(),/^Alpha Display Name \(unreachable\)/, + 'display name survives a server-unavailable verdict'); + } + assert.deepEqual(await page.evaluate(()=>{ + const model=view.modelCatalog.find(m=>m.ref==='compat:vendor/alpha'); + return [model.available,model.unavailable_reason,model.agent_available,model.name,model.capability,model.efforts]; + }),[false,'unreachable',false,'Alpha Display Name','reasoning',['high']], + 'catalogue metadata cannot upgrade server availability or agent admission'); + // Cached OpenRouter metadata cannot give disabled compat an eligible model. + await page.evaluate(async()=>await view.fetchAll()); + assert.ok(await page.evaluate(()=>view.openRouterCatalogue?.models?.length)); + compat.enabled=false; + await page.evaluate(async()=>await view.fetchLLMStatus()); + assert.equal(await page.evaluate(()=>view.openRouterCatalogueActive),false); + assert.equal(await page.evaluate(()=>view.autoAllowlistModels.some(m=>m.provider==='compat')),false); + assert.equal(await page.evaluate(()=>view.openRouterResults.length),0); + assert.equal(await page.getByRole('dialog').count(),0); + assert.equal(await mainSelect.locator('option[value="compat:vendor/alpha"]').count(),1); + // A delayed response from the old endpoint is discarded after the saved endpoint changes. + compat.enabled=true; + await page.evaluate(async()=>await view.fetchLLMStatus()); + let release; + holdCatalogue=new Promise(resolve=>{release=resolve;}); + const pending=page.evaluate(async()=>await view.fetchAll()); + await page.waitForFunction(()=>view.openRouterCatalogueLoading); + compat.base_url='https://router.example.net/v1'; + await page.evaluate(async()=>await view.fetchLLMStatus()); + release();holdCatalogue=null; + await pending; + assert.equal(await page.evaluate(()=>view.openRouterCatalogue),null,'stale OpenRouter response discarded after endpoint change'); + assert.equal(await page.evaluate(()=>view.openRouterCatalogueActive),false); + assert.equal(await page.evaluate(()=>view.openRouterResults.length),0); + // Now race the same request against disabling rather than endpoint replacement. + compat.enabled=true;compat.base_url='https://openrouter.ai/api/v1'; + await page.evaluate(async()=>await view.fetchLLMStatus()); + let releaseDisabled; + holdCatalogue=new Promise(resolve=>{releaseDisabled=resolve;}); + const pendingDisable=page.evaluate(async()=>await view.fetchAll()); + await page.waitForFunction(()=>view.openRouterCatalogueLoading); + compat.enabled=false; + await page.evaluate(async()=>await view.fetchLLMStatus()); + releaseDisabled();holdCatalogue=null; + await pendingDisable; + assert.equal(await page.evaluate(()=>view.openRouterCatalogue),null,'stale response discarded after saved disable'); + assert.equal(await page.evaluate(()=>view.openRouterResults.length),0); + // An explicit Refresh updates saved provider visibility but cannot discard + // unsaved Agent model/allowlist/hints or bless them as the server snapshot. + await page.evaluate(()=>{view.agentsConfig.model='gpt-6-luna';view.agentsConfig.auto_model_allowlist=['gpt-6-luna'];view.agentsConfig.model_selection_hints={'gpt-6-luna':'Draft hint'};}); + agents.model='auto';agents.auto_model_allowlist=['gpt-6-sol']; + await page.evaluate(async()=>await view.fetchAll()); + assert.deepEqual(await page.evaluate(()=>({model:view.agentsConfig.model,allowlist:view.agentsConfig.auto_model_allowlist,hints:view.agentsConfig.model_selection_hints})), + {model:'gpt-6-luna',allowlist:['gpt-6-luna'],hints:{'gpt-6-luna':'Draft hint'}}); + // A Main save racing an explicit status refresh must keep the draft until + // the write settles. A failed write leaves the saved provider visible. + await page.evaluate(()=>{view.modelSelection.main='gpt-6-luna';}); + failSave=true; + const failedMain=page.evaluate(async()=>await view.saveMainModel()); + await page.evaluate(async()=>await view.fetchLLMStatus()); + assert.equal(await mainSelect.inputValue(),'gpt-6-luna'); + await failedMain; + assert.equal(await mainSelect.inputValue(),'gpt-6-luna'); + assert.equal(await page.evaluate(()=>view.savedProviderEnabled('codex')),true); + failSave=false; + assert.deepEqual(errors,[]);assert.deepEqual(unexpected,[]); + console.log('LLM provider visibility DOM check OK'); +}finally{await browser?.close();await server.close();} diff --git a/scripts/check-login-persistence.mjs b/scripts/check-login-persistence.mjs new file mode 100644 index 000000000..faafe7581 --- /dev/null +++ b/scripts/check-login-persistence.mjs @@ -0,0 +1,77 @@ +// Exercise the real API client against browser storage and an in-memory server. +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const here = dirname(fileURLToPath(import.meta.url)); +const source = readFileSync(join(here, '../ui/js/api.js'), 'utf8'); +const storage = (seed = {}) => { + const values = new Map(Object.entries(seed)); + return { + getItem: key => values.get(key) ?? null, + setItem: (key, value) => values.set(key, String(value)), + removeItem: key => values.delete(key), + entries: () => Object.fromEntries(values), + }; +}; +globalThis.window = globalThis; +globalThis.location = { protocol: 'http:', host: 'localhost:3002' }; +globalThis.document = { addEventListener() {}, removeEventListener() {} }; +globalThis.setInterval = () => 0; +globalThis.clearInterval = () => {}; +globalThis.localStorage = storage(); +globalThis.sessionStorage = storage(); +const { OdinAPI } = await import('data:text/javascript;base64,' + Buffer.from(`${source}\nexport { OdinAPI };`).toString('base64')); +const reload = () => new OdinAPI(); +const reset = (local = {}, session = {}) => { + globalThis.localStorage = storage(local); + globalThis.sessionStorage = storage(session); +}; +const old = { odin_persist: '1', odin_token: 'synthetic-old', odin_session_timeout: '42' }; + +reset(old); +let api = reload(); +api.setPersist(false); +api.setToken('synthetic-new', 60); +assert.deepEqual(localStorage.entries(), {}); +assert.deepEqual(sessionStorage.entries(), { odin_token: 'synthetic-new', odin_session_timeout: '60' }); +assert.equal(reload().token, 'synthetic-new'); +assert.equal(reload().sessionTimeout, 60); + +reset({}, { odin_token: 'synthetic-old', odin_session_timeout: '42' }); +api = reload(); +api.setPersist(true); +api.setToken('synthetic-new'); +assert.deepEqual(sessionStorage.entries(), {}); +assert.deepEqual(localStorage.entries(), { odin_token: 'synthetic-new', odin_persist: '1' }); +assert.equal(reload().token, 'synthetic-new'); + +const valid = new Set(); +globalThis.fetch = async (path, opts = {}) => { + if (path === '/api/auth/login') { + valid.add('synthetic-new-session'); + return { status: 200, ok: true, json: async () => ({ session_id: 'synthetic-new-session', timeout_seconds: 42 }) }; + } + const bearer = (opts.headers?.Authorization || '').replace('Bearer ', ''); + if (!valid.has(bearer)) return { status: 401, ok: false, json: async () => ({}) }; + return { status: 200, ok: true, json: async () => ({ status: 'online' }) }; +}; +reset(old); +api = reload(); +assert.equal((await api.check()).needsAuth, true); +api.setPersist(false); +await api.login('synthetic-login'); +assert.equal(reload().token, 'synthetic-new-session'); +assert.equal((await reload().check()).ok, true); +await reload().logout(); +assert.deepEqual(localStorage.entries(), {}); +assert.deepEqual(sessionStorage.entries(), {}); + +reset(old); +api = reload(); +api.setPersist(true); +api.setToken('synthetic-new'); +assert.deepEqual(localStorage.entries(), { odin_persist: '1', odin_token: 'synthetic-new' }); +assert.equal(reload().token, 'synthetic-new'); +console.log('login-persistence: 13 assertions passed'); diff --git a/scripts/incus-deploy.sh b/scripts/incus-deploy.sh index 17a355bba..2e6f4c525 100755 --- a/scripts/incus-deploy.sh +++ b/scripts/incus-deploy.sh @@ -147,8 +147,6 @@ echo " incus exec $INSTANCE -- systemctl start odin" echo "" echo "View logs:" echo " incus exec $INSTANCE -- journalctl -u odin -f" -echo " # or" -echo " incus exec $INSTANCE -- tail -f /app/data/logs/odin.log" echo "" echo "Update config:" echo " incus file push config.yml $INSTANCE/app/config.yml" diff --git a/scripts/monitor.sh b/scripts/monitor.sh index 62d9aabe6..97cd2ae41 100755 --- a/scripts/monitor.sh +++ b/scripts/monitor.sh @@ -6,6 +6,9 @@ # Supports Docker, Incus, and bare metal deployments. # Auto-detects deployment type, or set ODIN_DEPLOY=docker|incus|local # For Incus: set ODIN_INCUS_INSTANCE (default: odin) +# Odin logs to standard output/error, not a file: the systemd journal for a +# service (journalctl -u odin), docker logs for Docker, or a source-run terminal. +# Set ODIN_LOG_FILE to tail a file you explicitly redirect that output to. set -e @@ -74,17 +77,27 @@ bot_logs() { ;; incus) INSTANCE="${ODIN_INCUS_INSTANCE:-odin}" - incus exec "$INSTANCE" -- tail -"$COUNT" /app/data/logs/odin.log 2>/dev/null || \ - incus exec "$INSTANCE" -- journalctl -u odin -n "$COUNT" --no-pager 2>/dev/null || \ - echo "Could not read logs from Incus instance '$INSTANCE'" + if [ -n "${ODIN_LOG_FILE:-}" ]; then + incus exec "$INSTANCE" -- tail -"$COUNT" "$ODIN_LOG_FILE" + else + incus exec "$INSTANCE" -- journalctl -u odin -n "$COUNT" --no-pager + fi ;; local) - LOG_FILE="${ODIN_LOG_FILE:-$SCRIPT_DIR/data/logs/odin.log}" - if [ -f "$LOG_FILE" ]; then - tail -"$COUNT" "$LOG_FILE" + if [ -n "${ODIN_LOG_FILE:-}" ]; then + if [ -f "$ODIN_LOG_FILE" ]; then + tail -"$COUNT" "$ODIN_LOG_FILE" + else + echo "Log file not found: $ODIN_LOG_FILE" + echo "Set ODIN_LOG_FILE to the correct log path" + fi + elif command -v journalctl >/dev/null 2>&1 && systemctl cat odin.service >/dev/null 2>&1; then + journalctl -u odin -n "$COUNT" --no-pager else - echo "Log file not found: $LOG_FILE" - echo "Set ODIN_LOG_FILE to the correct log path" + echo "Odin logs to standard output/error, not to a file." + echo "For a source run, check the terminal where Odin was started." + echo "For a service, use journalctl -u odin; for Docker, use docker logs odin-bot." + echo "Set ODIN_LOG_FILE to a file you redirected stdout/stderr to." fi ;; esac diff --git a/site/.vitepress/config.mts b/site/.vitepress/config.mts index 106c77fdd..d4553d341 100644 --- a/site/.vitepress/config.mts +++ b/site/.vitepress/config.mts @@ -42,6 +42,7 @@ export default withMermaid(defineConfig({ { text: 'Guide', items: [ { text: 'Install', link: '/install' }, { text: 'Configuration', link: '/configuration' }, + { text: 'Schedules & webhooks', link: '/scheduling' }, { text: 'Security model', link: '/security' }, { text: 'Runtime skills', link: '/skills' }, ]}, diff --git a/src/agents/manager.py b/src/agents/manager.py index a61a3909a..681a55dd7 100644 --- a/src/agents/manager.py +++ b/src/agents/manager.py @@ -380,7 +380,6 @@ class AgentInfo: messages: list[dict] = field(default_factory=list) tools_used: list[str] = field(default_factory=list) tool_execution_count: int = 0 - _tool_call_ids: set[str] = field(default_factory=set, repr=False) iteration_count: int = 0 last_activity: float = field(default_factory=time.time) phase: str = "ready" @@ -1487,7 +1486,7 @@ def _check_lifetime() -> bool: # Transition READY → EXECUTING for LLM call agent.transition(AgentState.EXECUTING, f"iteration {iteration + 1}") - agent._tool_call_ids.update(settled_call_ids(agent.messages)) + settled_call_ids(agent.messages) agent.set_phase( "generating", time.time() + min(agent.iteration_timeout, _remaining_lifetime(agent)) ) @@ -1555,9 +1554,7 @@ def _check_lifetime() -> bool: usage_response = {**response, **usage_facts} text = content_text(response.get("text", "")) - tool_calls = normalize_tool_calls( - response.get("tool_calls", []), used_ids=agent._tool_call_ids - ) + tool_calls = normalize_tool_calls(response.get("tool_calls", [])) context_density, context_density_source, context_primary_chars = _budget_observation( generation_state ) @@ -2277,6 +2274,8 @@ async def _call_llm_with_recovery( # missing) ladder is a real terminal outcome and must not # silently widen through the unknown-model fallback. compatible_target = _compatible_overflow_target_chars(exc, plan) + if compatible_target is not None and compatible_target >= attempt_chars: + compatible_target = None active_ladder = ( (compatible_target,) if compatible_target is not None diff --git a/src/async_utils.py b/src/async_utils.py index 7c5c2814d..59dd4d43f 100644 --- a/src/async_utils.py +++ b/src/async_utils.py @@ -2,9 +2,14 @@ from __future__ import annotations import asyncio +import functools +from collections.abc import Callable +from typing import Any, TypeVar from .odin_log import get_logger +_T = TypeVar("_T") + _log = get_logger("async_utils") @@ -22,3 +27,53 @@ def _done_cb(t: asyncio.Task) -> None: task.add_done_callback(_done_cb) return task + + +async def _settle_executor_call(call: Callable[[], Any]) -> tuple[asyncio.Future, bool]: + """Run ``call`` on the default executor and wait until it has physically finished. + + The wait survives any number of caller cancellations: each one is recorded + and the executor future is awaited again through ``asyncio.shield``. An + executor future is not a ``Task``, so the shutdown drain that cancels + ``asyncio.all_tasks()`` cannot cancel it either. + Returns ``(done_future, was_cancelled)``. + """ + loop = asyncio.get_running_loop() + fut = loop.run_in_executor(None, call) + was_cancelled = False + while not fut.done(): + try: + await asyncio.shield(fut) + except asyncio.CancelledError: + was_cancelled = True + except Exception: + break # worker raised; fut.done() is now True + return fut, was_cancelled + + +async def run_persist_settled( + persist_sync: Callable[[], Any], +) -> tuple[BaseException | None, bool]: + """Run a sync write to settlement; never raise. Returns ``(exc, was_cancelled)``. + + The caller commits or restores its own state, then re-raises + cancellation when ``was_cancelled`` is true. + """ + fut, was_cancelled = await _settle_executor_call(persist_sync) + return fut.exception(), was_cancelled + + +async def to_thread_settled(func: Callable[..., _T], /, *args: Any, **kwargs: Any) -> _T: + """``asyncio.to_thread`` for work done while an ``asyncio.Lock`` is held. + + Cancelling ``asyncio.to_thread`` releases the caller's lock while the + worker keeps running. This keeps the caller (and so the lock) until the + worker has finished, then re-raises the cancellation. Without a + cancellation it returns the result or raises the worker's exception, + exactly like ``asyncio.to_thread``. + """ + fut, was_cancelled = await _settle_executor_call(functools.partial(func, *args, **kwargs)) + if was_cancelled: + fut.exception() # retrieve, so a worker error is never "never retrieved" + raise asyncio.CancelledError + return fut.result() diff --git a/src/audit/logger.py b/src/audit/logger.py index 9598abe31..c7a79f528 100644 --- a/src/audit/logger.py +++ b/src/audit/logger.py @@ -6,7 +6,7 @@ from collections.abc import AsyncIterator, Callable from datetime import UTC, datetime from pathlib import Path -from typing import BinaryIO, Literal +from typing import BinaryIO, Literal, cast import aiofiles @@ -15,7 +15,7 @@ from ..observability.failure_classes import classify_failure from ..odin_log import get_logger from ..permissions.persistence import write_private_atomic -from .signer import GENESIS_HASH, AuditSigner, verify_log +from .signer import GENESIS_HASH, AuditSigner, verify_log, verify_segment log = get_logger("audit") @@ -256,9 +256,8 @@ def _maybe_rotate(self) -> None: each append. Bounds total growth to roughly max_bytes * (max_files + 1). The HMAC chain (if enabled) starts fresh in the new current file — the signer's prev-hash is reset to GENESIS after - rotation, so verify_integrity() (which reads the current file from - genesis) stays valid instead of chaining across the rotation boundary - and failing on the first post-rotation entry.""" + rotation, so each file verifies independently from genesis instead + of chaining across the rotation boundary.""" try: if not self.path.exists() or self.path.stat().st_size < self._max_bytes: return @@ -951,12 +950,62 @@ async def initialize_chain(self) -> None: ) self._chain_initialized = True - async def verify_integrity(self) -> dict: - """Verify the HMAC chain of the audit log. + async def _open_verify_snapshot(self) -> list[dict]: + """Open bounded descriptors of every retained generation under the append lock. - Returns a dict with ``valid``, ``total``, ``verified``, ``first_bad``, - and ``error`` fields. Requires signing to be enabled. + Unlike search, verification reports unreadable positions and interior + gaps. Scan outside the lock; descriptor identity survives rotation. """ + rows: list[dict] = [] + seen: set[tuple[int, int]] = set() + try: + async with self._persist_lock: + present: list[int] = [] + for index in range(self._max_files + 1): + path = self.path if index == 0 else self.path.with_name( + self.path.name + f".{index}") + row = {"file": path.name, "position": index, "handle": None, "size": 0, + "open_error": None, "absent": False} + try: + handle = open(path, "rb") + except FileNotFoundError: + row["absent"] = True + rows.append(row) + continue + except OSError as exc: + row["open_error"] = type(exc).__name__ + rows.append(row) + present.append(index) + continue + try: + stat = os.fstat(handle.fileno()) + except OSError as exc: + handle.close() + row["open_error"] = type(exc).__name__ + rows.append(row) + present.append(index) + continue + identity = (stat.st_dev, stat.st_ino) + if identity in seen: + handle.close() + continue + seen.add(identity) + row.update(handle=handle, size=stat.st_size) + rows.append(row) + present.append(index) + highest = max((index for index in present if index > 0), default=0) + return [ + row for row in rows + if not row["absent"] or row["position"] == 0 or row["position"] < highest + ] + except BaseException: + for row in rows: + if row["handle"] is not None: + cast(BinaryIO, row["handle"]).close() + raise + + async def verify_integrity(self) -> dict: + """Check each retained file's independent HMAC chain without blocking appends.""" if not self._signer: return { "valid": False, @@ -964,16 +1013,71 @@ async def verify_integrity(self) -> dict: "verified": 0, "unsigned_prefix": 0, "first_bad": None, + "first_bad_file": None, "availability": "not_enabled", "error": "Signing not enabled (no hmac_key configured)", + "segments": [], } - result = await verify_log(self.path, self._signer._key.decode()) - # Availability is distinct from verdict. A configured verifier that - # returns valid=False is a failure even when its diagnostic has an - # error string; only the explicit not_enabled shape is soft copy. - result["availability"] = "available" - result["durability"] = ( - "repair_required" if self.repair_required - else "degraded" if self.durability_degraded or not result["valid"] else "durable" - ) - return result + rows = await self._open_verify_snapshot() + segments: list[dict] = [] + try: + for row in rows: + seg = {"file": row["file"], "position": row["position"], "total": 0, + "verified": 0, "unsigned_prefix": 0, "first_bad": None, + "reason": None, "error": None} + if row["absent"]: + if row["position"] == 0: + seg["status"] = "verified" # No active entries yet. + else: + seg.update(status="missing", error="expected file not found") + elif row["open_error"]: + seg.update(status="unreadable", error=row["open_error"]) + else: + try: + result = await asyncio.to_thread( + verify_segment, row["handle"], row["size"], self._signer._key) + except OSError as exc: + seg.update(status="unreadable", error=type(exc).__name__) + else: + for field in ("total", "verified", "unsigned_prefix", + "first_bad", "reason", "error"): + seg[field] = result[field] + seg["status"] = ( + "broken" if not result["valid"] + else "unsigned" if result["verified"] == 0 and result["total"] > 0 + else "verified" + ) + segments.append(seg) + finally: + for row in rows: + if row["handle"] is not None: + row["handle"].close() + # Files are independent chains, but a wholly unsigned generation + # newer than a signed one is still a signing gap. + signed_seen = False + for seg in reversed(segments): + if seg["status"] == "unsigned" and signed_seen: + seg.update(status="broken", reason="signing_gap", + error="no signatures although older files are signed") + if seg["verified"] > 0: + signed_seen = True + problems = [seg for seg in segments if seg["status"] in ("broken", "unreadable", "missing")] + first = problems[0] if problems else None + active = segments[0] if segments else {"status": "verified"} + return { + "valid": not problems, + "availability": "available", + "scope": "retained_files", + "total": sum(seg["total"] for seg in segments), + "verified": sum(seg["verified"] for seg in segments), + "unsigned_prefix": sum(seg["unsigned_prefix"] for seg in segments), + "first_bad": first["first_bad"] if first else None, + "first_bad_file": first["file"] if first else None, + "error": f"{first['file']}: {first['error']}" if first else None, + "durability": ( + "repair_required" if self.repair_required + else "degraded" if self.durability_degraded or active["status"] == "broken" + else "durable" + ), + "segments": segments, + } diff --git a/src/audit/signer.py b/src/audit/signer.py index e087fcece..2b79c1aff 100644 --- a/src/audit/signer.py +++ b/src/audit/signer.py @@ -19,7 +19,7 @@ class AuditSigner: file from top to bottom and checks that every link in the chain is valid. """ - def __init__(self, key: str) -> None: + def __init__(self, key: str | bytes) -> None: self._key = key.encode() if isinstance(key, str) else key self._prev_hmac: str = GENESIS_HASH @@ -70,6 +70,74 @@ def _canonical(entry: dict) -> str: return json.dumps(filtered, sort_keys=True, default=str, separators=(",", ":")) +REASON_INVALID_JSON = "invalid_json" +REASON_NON_OBJECT = "non_object" +REASON_UNSIGNED_AFTER_SIGNED = "unsigned_after_signed" +REASON_HMAC_MISMATCH = "hmac_mismatch" + + +def verify_segment(handle, size: int, key: str | bytes) -> dict: + """Verify a single bounded file snapshot from GENESIS, one line at a time. + + The bound is captured while no append is in flight, so a later append is + neither scanned nor mistaken for a torn audit entry. Physical line numbers + include blank lines. This synchronous function belongs in a worker thread. + """ + signer = AuditSigner(key) + prev = GENESIS_HASH + total = verified = unsigned_prefix = 0 + remaining = size + lineno = 0 + + def failure(line: int, reason: str, message: str) -> dict: + return { + "valid": False, "total": total, "verified": verified, + "unsigned_prefix": unsigned_prefix, "first_bad": line, + "reason": reason, "error": f"Line {line}: {message}", + } + + while remaining > 0: + raw = handle.readline(remaining) + if not raw: + break + remaining -= len(raw) + lineno += 1 + if not raw.strip(): + continue + total += 1 + try: + entry = json.loads(raw) + except ValueError: # Includes invalid UTF-8 from bytes input. + return failure(lineno, REASON_INVALID_JSON, "invalid JSON") + if not isinstance(entry, dict): + return failure(lineno, REASON_NON_OBJECT, "non-object audit entry") + if "_hmac" not in entry: + # verified == 0 ⟺ still in the pre-enablement prefix. + if verified == 0: + unsigned_prefix += 1 + continue + return failure( + lineno, REASON_UNSIGNED_AFTER_SIGNED, + "missing _hmac field (unsigned entry after chain began)", + ) + try: + valid = signer.verify_entry(entry, prev) + except (TypeError, ValueError): + valid = False + if not valid: + return failure( + lineno, REASON_HMAC_MISMATCH, + "HMAC verification failed (tampered or reordered)", + ) + prev = entry["_hmac"] + verified += 1 + return { + "valid": True, "total": total, "verified": verified, + "unsigned_prefix": unsigned_prefix, "first_bad": None, + "reason": None, "error": None, + } + + async def verify_log(path, key: str) -> dict: """Verify the full HMAC chain of an audit log file. @@ -86,10 +154,10 @@ async def verify_log(path, key: str) -> dict: (signed entries verified, int), ``unsigned_prefix`` (int), ``first_bad`` (int or None — 1-indexed line number), and ``error`` (str or None). """ + import asyncio + import os from pathlib import Path - import aiofiles - p = Path(path) if not p.exists(): return { @@ -101,88 +169,16 @@ async def verify_log(path, key: str) -> dict: "error": None, } - signer = AuditSigner(key) - prev = GENESIS_HASH - total = 0 - verified = 0 - unsigned_prefix = 0 + def _verify_file() -> dict: + with open(p, "rb") as handle: + return verify_segment(handle, os.fstat(handle.fileno()).st_size, key) try: - async with aiofiles.open(p) as f: - lines = await f.readlines() + result = await asyncio.to_thread(_verify_file) except Exception as exc: return { - "valid": False, - "total": 0, - "verified": 0, - "unsigned_prefix": 0, - "first_bad": None, - "error": str(exc), + "valid": False, "total": 0, "verified": 0, + "unsigned_prefix": 0, "first_bad": None, "error": str(exc), } - - for i, line in enumerate(lines, 1): - line = line.strip() - if not line: - continue - total += 1 - - try: - entry = json.loads(line) - except json.JSONDecodeError: - return { - "valid": False, - "total": total, - "verified": verified, - "unsigned_prefix": unsigned_prefix, - "first_bad": i, - "error": f"Line {i}: invalid JSON", - } - - if not isinstance(entry, dict): - return { - "valid": False, "total": total, "verified": verified, - "unsigned_prefix": unsigned_prefix, "first_bad": i, - "error": f"Line {i}: non-object audit entry", - } - - if "_hmac" not in entry: - # verified == 0 ⟺ no signed entry seen yet (a signed entry that - # fails returns immediately), i.e. still in the pre-enablement - # prefix. - if verified == 0: - unsigned_prefix += 1 - continue - return { - "valid": False, - "total": total, - "verified": verified, - "unsigned_prefix": unsigned_prefix, - "first_bad": i, - "error": f"Line {i}: missing _hmac field (unsigned entry after chain began)", - } - - try: - valid = signer.verify_entry(entry, prev) - except (TypeError, ValueError): - valid = False - if not valid: - return { - "valid": False, - "total": total, - "verified": verified, - "unsigned_prefix": unsigned_prefix, - "first_bad": i, - "error": f"Line {i}: HMAC verification failed (tampered or reordered)", - } - - prev = entry["_hmac"] - verified += 1 - - return { - "valid": True, - "total": total, - "verified": verified, - "unsigned_prefix": unsigned_prefix, - "first_bad": None, - "error": None, - } + result.pop("reason", None) + return result diff --git a/src/computer/controller.py b/src/computer/controller.py index db75b953b..9304d5272 100644 --- a/src/computer/controller.py +++ b/src/computer/controller.py @@ -1148,6 +1148,21 @@ def _input_status(self, live, grant): } async def session(self, context: RequestContext, inp: dict) -> dict: + """Attach non-dispatch evidence only before lifecycle, cleanup or input steps.""" + from .error_guidance import InputBoundaryError + + boundary: dict[str, bool | str] = {"dispatched": False, "state": "unstarted"} + try: + return await self._session(context, inp, boundary) + except ComputerError as exc: + if boundary["dispatched"] or isinstance(exc, InputBoundaryError): + raise + raise InputBoundaryError( + exc.code, execution={"injected": False, "sent": False}, + state=str(boundary["state"]), + ) from exc + + async def _session(self, context: RequestContext, inp: dict, boundary) -> dict: exact_keys( inp, { @@ -1197,7 +1212,13 @@ async def session(self, context: RequestContext, inp: dict) -> dict: "supported_next_step": "start", } try: - result = await inventory() + from .runtime.hyprland_discovery import HyprlandDiscoveryError + + try: + result = await inventory() + except HyprlandDiscoveryError as exc: + # Discovery rejected a request, not a session or input transition. + raise ComputerError(exc.code) from None await self._auth(context) if ( type(result) is not dict @@ -1322,6 +1343,7 @@ async def session(self, context: RequestContext, inp: dict) -> dict: selected, selection_proof = self._selection_binding( context, selected, with_proof=True ) + boundary["dispatched"] = True grant = self.store.create_session( context, app, @@ -1467,15 +1489,18 @@ async def session(self, context: RequestContext, inp: dict) -> dict: grant = self._grant( context, inp, same_turn=False, generation=operation in {"resume", "export", "reconcile"} ) + boundary["state"] = grant.state if operation == "status": return self._public_session(grant) if operation in {"stop", "cancel", "close"}: if grant.state in {"cancelled", "closed"} and grant.session_id not in self._live: return self._public_session(grant) + boundary["dispatched"] = True return await self._stop( grant.session_id, "closed" if operation == "close" else "cancelled" ) if operation == "pause": + boundary["dispatched"] = True return await self._pause(grant.session_id) if operation == "resume": if grant.state != "paused" or grant.session_id not in self._live: @@ -1505,6 +1530,8 @@ async def session(self, context: RequestContext, inp: dict) -> dict: if grant.platform == "wayland" else STOP_TIMEOUT_SECONDS ) + # Re-arming input has its own rollback/stop path below. + boundary["dispatched"] = True try: await _bounded(resume(consent_generation=grant.consent_generation), timeout) measured = getattr(live.backend, "capabilities", None) @@ -1572,6 +1599,8 @@ async def session(self, context: RequestContext, inp: dict) -> dict: raise return self._public_session(self.store.get_session(grant.session_id)) if operation == "reconcile": + # Owner reconciliation can release input; observe owns its own boundary. + boundary["dispatched"] = True if grant.state in {"quarantined", "closed"} and grant.session_id not in self._live: return await self.reconcile_hyprland_owner( context, grant.session_id, grant.generation) @@ -1629,7 +1658,8 @@ async def _pause(self, sid): return await self._stop(sid, "cancelled") return self._public_session(self.store.get_session(sid)) - async def _capture(self, grant, *, acknowledge_modal=False, crop=None, strict_binding=False): + async def _capture(self, grant, *, acknowledge_modal=False, crop=None, strict_binding=False, + boundary=None): from .vision import FrameCrop, FrameMetadata, _validate_png live = self._active(grant) @@ -1649,6 +1679,9 @@ async def _capture(self, grant, *, acknowledge_modal=False, crop=None, strict_bi live.observations.clear() self._delivered_observations.pop(grant.session_id, None) if self._hyprland_continuity_failure(live, exc): + if boundary is not None: + # Quarantine cleanup may release input. + boundary["dispatched"] = True await self._quarantine_hyprland(grant, live, phase="native_continuity_lost") if self.store.get_session(grant.session_id).state == "active": raise ComputerError("hyprland_recovered_fresh_observation_required") from None @@ -1753,6 +1786,21 @@ async def _capture(self, grant, *, acknowledge_modal=False, crop=None, strict_bi return obs, raw.image_bytes async def observe(self, context, inp): + """Attach non-dispatch evidence only before recovery or cleanup can act.""" + from .error_guidance import InputBoundaryError + + boundary: dict[str, bool | str] = {"dispatched": False, "state": "unstarted"} + try: + return await self._observe(context, inp, boundary) + except ComputerError as exc: + if boundary["dispatched"] or isinstance(exc, InputBoundaryError): + raise + raise InputBoundaryError( + exc.code, execution={"injected": False, "sent": False}, + state=str(boundary["state"]), + ) from exc + + async def _observe(self, context, inp, boundary): exact_keys( inp, {"session_id", "generation", "source_id", "crop", "task_context"}, @@ -1764,6 +1812,7 @@ async def observe(self, context, inp): crop = crop_arguments(crop) await self._auth(context) grant = self._grant(context, inp) + boundary["state"] = grant.state async with self._actions: live = self._active(grant) if (hints is not None and live.capabilities is not None @@ -1793,7 +1842,7 @@ async def observe(self, context, inp): self._active(grant) await self._auth(context) try: - obs, image = await self._capture(grant, crop=crop) + obs, image = await self._capture(grant, crop=crop, boundary=boundary) except ComputerError as exc: # A requested observation can be the first proof that a human # changed focus or the scoped native geometry. With no action @@ -1801,21 +1850,24 @@ async def observe(self, context, inp): # candidate and produce entirely fresh pixels. It never adopts # a different output/application or resumes input. live = self._live.get(grant.session_id) - if ( + recover = ( grant.environment == "existing_session" and live is not None and exc.code in {"stale_source_binding", "input_focus_unavailable"} and self._no_input_pending(grant.session_id) - and await self._recover_focus( - context, - grant, - live, - getattr(live.backend, "application_provenance", None), - ) + ) + if recover and live is not None and self._recovery_enabled(live): + # Native focus recovery may send input; do not claim otherwise. + boundary["dispatched"] = True + if recover and live is not None and await self._recover_focus( + context, + grant, + live, + getattr(live.backend, "application_provenance", None), ): await self._auth(context) self._active(grant) - obs, image = await self._capture(grant, crop=crop) + obs, image = await self._capture(grant, crop=crop, boundary=boundary) else: raise await self._auth(context) diff --git a/src/computer/error_guidance.py b/src/computer/error_guidance.py index 0043935ee..98b1588c7 100644 --- a/src/computer/error_guidance.py +++ b/src/computer/error_guidance.py @@ -91,6 +91,15 @@ def __init__(self, code: str, *, execution: dict, state: str): "unsupported_operation", "target_inventory_unavailable", "inventory_targets_unsupported", + "hyprland_discovery_deadline", + "hyprland_discovery_runtime_untrusted", + "hyprland_discovery_policy_invalid", + "hyprland_discovery_runtime_unavailable", + "hyprland_discovery_candidate_limit", + "hyprland_discovery_hint_invalid", + "hyprland_discovery_not_found", + "hyprland_discovery_ambiguous", + "hyprland_discovery_unavailable", "operation_unavailable", "focus_preflight_refused", "focus_candidate_unavailable", @@ -108,6 +117,20 @@ def __init__(self, code: str, *, execution: dict, state: str): _AUDIT_REASON_CODES = frozenset( { "computer_rejected", + "computer_succeeded", + "verified", + "executed", + "capture_source_not_granted", + "capture_not_active", + "hyprland_discovery_deadline", + "hyprland_discovery_runtime_untrusted", + "hyprland_discovery_policy_invalid", + "hyprland_discovery_runtime_unavailable", + "hyprland_discovery_candidate_limit", + "hyprland_discovery_hint_invalid", + "hyprland_discovery_not_found", + "hyprland_discovery_ambiguous", + "hyprland_discovery_unavailable", "computer_not_satisfied", "outcome_unknown", "permission_denied", @@ -852,6 +875,8 @@ def safety_terminal(result: dict) -> bool: or "state" in item and item["state"] not in ( + # No session/grant resolved yet; no held input possible. + "unstarted", "starting", "active", "paused", @@ -955,6 +980,23 @@ def audit_reason_code(reason: object) -> str: return "computer_rejected" +# Receipt statuses the foreground tool loop publishes as failed calls. +FAILED_RECEIPT_STATUSES = frozenset( + {"unavailable", "not_satisfied", "rejected", "failed", "unknown", "interrupted"} +) + + +def audit_outcome_code(result: object, *, succeeded: bool) -> str: + """Return fixed audit code for a result without labelling successes rejections.""" + result = result if isinstance(result, dict) else {} + evidence = result.get("verification") + evidence = evidence if isinstance(evidence, dict) else {} + value = result.get("reason") or evidence.get("reason") or result.get("status") + if succeeded and not (isinstance(value, str) and value in _AUDIT_REASON_CODES): + return "computer_succeeded" + return audit_reason_code(value) + + def failure_guidance(result: dict, *, terminal: bool = False) -> dict: """Keep native reasons and receipts intact, adding a conservative summary.""" evidence = result.get("verification") diff --git a/src/computer/integration.py b/src/computer/integration.py index 413b5a2eb..56e995818 100644 --- a/src/computer/integration.py +++ b/src/computer/integration.py @@ -16,6 +16,8 @@ from ..tools.output_authorization import tool_scope_allows from ..tools.result_validator import ToolResult from .error_guidance import ( + FAILED_RECEIPT_STATUSES, + audit_outcome_code, audit_reason_code, exception_reason, failure_guidance, @@ -304,11 +306,9 @@ async def _tool(self, name, values): image["__computer_audit_metadata__"] = { "computer_call_id": grant.call_id, "computer_turn_id": grant.context.turn_id, - "computer_reason_code": audit_reason_code( - receipt.get("reason") or ( - receipt.get("verification", {}).get("reason") - if isinstance(receipt.get("verification"), dict) else None - ) or receipt.get("status") + "computer_reason_code": audit_outcome_code( + receipt, + succeeded=receipt.get("status") not in FAILED_RECEIPT_STATUSES, ), "computer_input_outcome": input_outcome(receipt), } @@ -341,13 +341,7 @@ async def _tool(self, name, values): "failed", } rejected = rejected and not capability_refusal - safe_reason = audit_reason_code( - (result.get("reason") or ( - result.get("verification", {}).get("reason") - if isinstance(result.get("verification"), dict) else None - ) or result.get("status")) - if isinstance(result, dict) else None - ) + safe_reason = audit_outcome_code(result, succeeded=not (unknown or rejected)) return ToolResult( json.dumps(result, ensure_ascii=True, separators=(",", ":")), ok=not (unknown or rejected), diff --git a/src/computer/runtime/x11_attached.py b/src/computer/runtime/x11_attached.py index f118d522a..5a2b53c99 100644 --- a/src/computer/runtime/x11_attached.py +++ b/src/computer/runtime/x11_attached.py @@ -918,9 +918,9 @@ def _select_source(self, source_id): async def select_source(self, source_id): async with self._lock: if self._closed or self._paused or not self._started: - raise AttachedFailure("capture_not_active") + raise ComputerError("capture_not_active") if source_id not in self._sources: - raise AttachedFailure("capture_source_not_granted") + raise ComputerError("capture_source_not_granted") self._select_source(source_id) return {"selected_source": source_id, "capture_only": not self._input_enabled} @@ -1706,7 +1706,7 @@ async def revoke(): await revoke() async def export(self, name): - raise AttachedFailure("existing_session_export_not_granted") + raise ComputerError("existing_session_export_not_granted") async def pause(self): if self._pause_job is None or self._pause_job.done(): diff --git a/src/discord/channel_logger.py b/src/discord/channel_logger.py index 65b672861..c4aa8a2df 100644 --- a/src/discord/channel_logger.py +++ b/src/discord/channel_logger.py @@ -201,7 +201,9 @@ def _index_batch(self, path: Path, cursor: str | None) -> list[dict]: return self._index_batch(path, None) return batch - def search(self, query: str, limit: int = 20, channel_id: str | None = None) -> list[dict]: + def search(self, query: str, limit: int = 20, channel_id: str | None = None, + *, author_id: str | None = None, + accept=None) -> list[dict]: """Keyword search on JSONL files (fallback when FTS is unavailable). Returns dicts with content, author, channel_id, timestamp, type="channel". @@ -230,6 +232,10 @@ def search(self, query: str, limit: int = 20, channel_id: str | None = None) -> record = json.loads(line) except json.JSONDecodeError: continue + if author_id and str(record.get("author_id", "")) != author_id: + continue + if accept is not None and not accept(record): + continue content = record.get("content", "") if query_lower in content.lower(): results.append({ diff --git a/src/discord/channel_state.py b/src/discord/channel_state.py index 80978c0d6..66637b4d3 100644 --- a/src/discord/channel_state.py +++ b/src/discord/channel_state.py @@ -334,6 +334,8 @@ def track_action( result_preview: str, elapsed_ms: int, channel_id: str | None = None, + *, + failed: bool = False, ) -> None: """Record a tool execution for conversational context injection. @@ -354,7 +356,7 @@ def track_action( inp_summary = ", ".join(f"{k}={v}" for k, v in safe_input.items() if isinstance(v, str)) if len(inp_summary) > 100: inp_summary = inp_summary[:100] + "..." - status = "OK" if "error" not in result_preview.lower()[:50] else "ERROR" + status = "ERROR" if failed or "error" in result_preview.lower()[:50] else "OK" entry = f"- [{ts}] `{tool_name}`({inp_summary}) → {status} ({elapsed_ms}ms)" self.track_recent_action(channel_id, entry) diff --git a/src/discord/client.py b/src/discord/client.py index 621a37129..5ee133139 100644 --- a/src/discord/client.py +++ b/src/discord/client.py @@ -316,11 +316,15 @@ async def start_application(self) -> None: self.scheduled_events._on_scheduled_task, self.scheduled_events._on_schedule_failure, ) + if getattr(self, "_knowledge_store", None) and getattr(self, "_fts_index", None): + fire_and_forget(self._reconcile_knowledge_fts(), name="reconcile_knowledge_fts") try: await self.computer.start() except Exception: log.exception("Computer startup failed; desktop tools remain unavailable") + if getattr(self, "_vector_store", None): + fire_and_forget(self._backfill_archives(), name="backfill_archives") self._application_started = True async def close(self) -> None: @@ -455,8 +459,6 @@ async def on_ready(self, *, expected_generation: int | None = None) -> None: self.scheduled_events._on_scheduled_task, self.scheduled_events._on_schedule_failure, ) - if self._vector_store: - fire_and_forget(self._backfill_archives(), name="backfill_archives") await self.delivery.set_status(None, task_end=True) async def on_guild_join(self, guild: discord.Guild) -> None: @@ -530,22 +532,32 @@ async def _reconcile_application_commands(self, guilds=None) -> None: async def _backfill_archives(self) -> None: """Backfill semantic search index and FTS5 with existing archive files.""" try: + vector_store = getattr(self, "_vector_store", None) + embedder = getattr(self, "_embedder", None) + if vector_store is None or embedder is None: + return archive_dir = self.sessions.persist_dir / "archive" - count = await self._vector_store.backfill(archive_dir, self._embedder) # type: ignore[union-attr, arg-type] # built together under search.enabled + count = await vector_store.backfill(archive_dir, embedder) + if hasattr(vector_store, "backfill_segments"): + segment_count = await vector_store.backfill_segments(archive_dir, embedder) + if segment_count: + log.info("Backfilled segments for %d archives", segment_count) if count: log.info("Backfilled %d archive sessions into vector store", count) else: log.info("Vector store up to date") - # Backfill knowledge FTS from existing data - if self._knowledge_store and self._fts_index: - import asyncio - - kb_count = await asyncio.to_thread(self._knowledge_store.backfill_fts) - if kb_count: - log.info("Backfilled %d knowledge chunks into FTS index", kb_count) except Exception as e: log.error("Archive backfill failed: %s", e) + async def _reconcile_knowledge_fts(self) -> None: + """Reconcile knowledge FTS even on API-only installs.""" + try: + count = await self._knowledge_store.backfill_fts_async() # type: ignore[union-attr] + if count: + log.info("Backfilled %d knowledge chunks into FTS index", count) + except Exception: + log.exception("Knowledge FTS reconciliation failed") + async def on_message(self, message: discord.Message) -> None: """Intake gating chain — owned by intake_pipeline.MessageIntake.""" await self.intake.handle(message) diff --git a/src/discord/delivery.py b/src/discord/delivery.py index 82a996651..1a0067b78 100644 --- a/src/discord/delivery.py +++ b/src/discord/delivery.py @@ -225,6 +225,17 @@ def _close_generated_fallback_file(file: discord.File) -> None: } +def close_open_fence(text: str) -> str: + """Close a code block that a cut left open in a Discord-only copy. + + Uses the chunker's fence rule (a line starting with three backticks + toggles a block); text with every block closed comes back unchanged. + """ + if sum(1 for line in text.split("\n") if line.startswith("```")) % 2: + return text + "\n```" + return text + + class ResponseDelivery: STATUS_DEBOUNCE: float = 5.0 @@ -410,6 +421,9 @@ async def send_chunked(self, message, text: str) -> None: current = "" in_code_block = False code_block_lang = "" + # Where the open block's own fence line starts in ``current`` while + # that block has no content in this chunk yet; None otherwise. + opener_at: int | None = None # Pre-split any lines longer than the chunk limit so the chunker # never encounters a single line that can't fit in one chunk. @@ -422,23 +436,53 @@ async def send_chunked(self, message, text: str) -> None: lines.append(raw_line) for line in lines: - # Track code block state (toggle on ``` lines) - if line.startswith("```"): + is_fence = line.startswith("```") + # Decide the split with the fence state BEFORE this line: the + # outgoing chunk must be closed/reopened for the block it is in. + if len(current) + len(line) + 1 > DISCORD_MAX_LEN - 10: + if ( + in_code_block + and is_fence + and len(current) + len(line) + 1 <= DISCORD_MAX_LEN + ): + # The closing fence fits the reserve kept for closing: + # it ends this chunk instead of opening the next one. + chunks.append(current + line + "\n") + current = "" + in_code_block = False + code_block_lang = "" + opener_at = None + continue + if in_code_block and opener_at is not None: + # The block has no content here yet: move its opening + # fence to the next chunk instead of sending an empty block. + head, current = current[:opener_at], current[opener_at:] + if head.strip(): + chunks.append(head) + else: + if in_code_block: + current += "\n```" + if current.strip(): + chunks.append(current) + current = "" + if in_code_block: + current = f"```{code_block_lang}\n" + if in_code_block and len(current) + len(line) + 1 + 4 > DISCORD_MAX_LEN: + # A long language tag would push this chunk past the + # limit once closed: continue the block without it. + current = "```\n" + opener_at = None + if is_fence: if in_code_block: in_code_block = False code_block_lang = "" + opener_at = None else: in_code_block = True code_block_lang = line[3:].strip() - - if len(current) + len(line) + 1 > DISCORD_MAX_LEN - 10: - if in_code_block: - current += "\n```" - if current.strip(): - chunks.append(current) - current = "" - if in_code_block: - current = f"```{code_block_lang}\n" + opener_at = len(current) + elif in_code_block: + opener_at = None current += line + "\n" if current.strip(): diff --git a/src/discord/llm_gateway.py b/src/discord/llm_gateway.py index c78623387..db169c1fb 100644 --- a/src/discord/llm_gateway.py +++ b/src/discord/llm_gateway.py @@ -626,18 +626,9 @@ async def run_persist_settled(self, persist_sync): or exactly-restores, then re-raises cancellation, once state is coherent. """ - loop = asyncio.get_running_loop() - fut = loop.run_in_executor(None, persist_sync) - was_cancelled = False - while not fut.done(): - try: - await asyncio.shield(fut) - except asyncio.CancelledError: - was_cancelled = True - except Exception: - break # worker raised; fut.done() is now True - exc = fut.exception() - return exc, was_cancelled + from ..async_utils import run_persist_settled + + return await run_persist_settled(persist_sync) def _snapshot_aux_config(self) -> dict: aux_cfg = self.get_config().openai_codex.auxiliary diff --git a/src/discord/scheduled_events.py b/src/discord/scheduled_events.py index 1feec487d..399df0a91 100644 --- a/src/discord/scheduled_events.py +++ b/src/discord/scheduled_events.py @@ -24,6 +24,7 @@ from ..odin_log import get_logger from ..scheduler.scheduler import NonRetryableScheduleError from ..tools import ToolResult +from .delivery import close_open_fence from .mcp_dispatch import uncertain_outcome as mcp_uncertain_outcome from .response_guards import scrub_response_secrets from .tool_loop import _LoopMessageProxy @@ -81,7 +82,7 @@ async def _on_scheduled_digest(self, schedule: dict) -> None: log.info("Running daily digest for channel %s", channel_id) try: - raw = await self._format_digest_raw() + raw, failed, total = await self._format_digest_raw(schedule, channel) except Exception as e: log.error("Digest data collection failed: %s", e) try: @@ -97,6 +98,15 @@ async def _on_scheduled_digest(self, schedule: dict) -> None: ) from send_error raise RuntimeError(f"Digest data collection failed: {e}") from e + if total and len(failed) == total: + await channel.send( + scrub_response_secrets( + "**Daily Infrastructure Digest**\n\n" + f"Collection failed for every check ({total} of {total}).\n\n{raw[:1500]}" + ) + ) + raise RuntimeError(f"Digest collected no data: all {total} checks failed") + # Summarize the digest — prefer Codex (free), fall back to raw truncation digest_messages = [ { @@ -119,24 +129,39 @@ async def _on_scheduled_digest(self, schedule: dict) -> None: log.warning("Digest summary failed, using raw: %s", e) summary = raw[:3000] + if failed: + labels = ", ".join(failed[:10]) + if len(failed) > 10: + labels += ", …" + summary += f"\n\nCollection failed for {len(failed)} of {total} checks: {labels}" + await channel.send(scrub_response_secrets(f"**Daily Infrastructure Digest**\n\n{summary}")) # Audit log the digest - await self._audit.log_execution( - user_id="system", - user_name="scheduler", - channel_id=channel_id, - tool_name="digest", - tool_input={"schedule_id": schedule.get("id")}, - approved=True, - result_summary=summary, - execution_time_ms=0, - ) + try: + await self._audit.log_execution( + user_id="system", + user_name="scheduler", + channel_id=channel_id, + tool_name="digest", + tool_input={"schedule_id": schedule.get("id")}, + approved=True, + result_summary=summary, + execution_time_ms=0, + ) + except Exception: + # Discord delivery already succeeded. Raising here would make the + # scheduler retry and deliver the same digest a second time. + log.exception("Failed to audit delivered scheduled digest") - async def _format_digest_raw(self) -> str: + async def _format_digest_raw( + self, schedule: dict, channel: discord.abc.Messageable + ) -> tuple[str, list[str], int]: """Collect raw infrastructure data for the digest.""" tasks = [] labels = [] + req_id = schedule.get("requester_id") or None + req_name = schedule.get("requester") or schedule.get("created_by") or "scheduler" # Disk + memory checks on all hosts via run_command aliases = ( @@ -146,33 +171,48 @@ async def _format_digest_raw(self) -> str: ) for host_alias in aliases: tasks.append( - self._tool_executor.execute( + self._execute_scheduled_tool( "run_command", { "host": host_alias, "command": "df -h --exclude-type=tmpfs --exclude-type=devtmpfs", }, + channel, + req_id, + req_name, ) ) labels.append(f"Disk ({host_alias})") tasks.append( - self._tool_executor.execute( + self._execute_scheduled_tool( "run_command", {"host": host_alias, "command": "free -h"}, + channel, + req_id, + req_name, ) ) labels.append(f"Memory ({host_alias})") + if not labels: + raise RuntimeError("No configured hosts available for digest checks") + results = await asyncio.gather(*tasks, return_exceptions=True) sections = [] + failed = [] for label, result in zip(labels, results): - if isinstance(result, Exception): - sections.append(f"### {label}\nERROR: {result}") + if isinstance(result, (Exception, asyncio.CancelledError)): + failed.append(label) + sections.append(f"### {label}\nCollection failed: {str(result)[:300]}") + elif isinstance(result, ToolResult) and not result.ok: + failed.append(label) + sections.append(f"### {label}\nCollection failed: {str(result)[:300]}") else: - sections.append(f"### {label}\n{str(result)[:800]}") + output = result.output if isinstance(result, ToolResult) else str(result) + sections.append(f"### {label}\n{output[:800]}") - return "\n\n".join(sections) + return "\n\n".join(sections), failed, len(labels) def _resolve_mentions(self, text: str) -> str: """Replace @username with proper Discord <@ID> mentions.""" @@ -369,7 +409,7 @@ async def _run_scheduled_workflow( summary = "\n".join(results) text = f"**Workflow: {desc}**\n{summary}" if len(text) > 1900: - text = text[:1900] + "\n... (truncated)" + text = close_open_fence(text[:1900]) + "\n... (truncated)" try: await channel.send(scrub_response_secrets(text)) diff --git a/src/discord/tool_loop.py b/src/discord/tool_loop.py index 626757166..29dc885ba 100644 --- a/src/discord/tool_loop.py +++ b/src/discord/tool_loop.py @@ -2968,6 +2968,7 @@ async def _run_one_tool_captured(self, st: _ChatTurn, block) -> dict: result[:200], elapsed_ms, channel_id=str(st.message.channel.id), + failed=error is not None, ) except Exception: pass # Non-critical tracking @@ -3163,10 +3164,12 @@ async def _run_one_tool_with_timeout( # WI-3 (interrupted): wait_for cancelled _run_one_tool before its # own settle. The persisted effect class decides whether this is a # definite non-effect failure or an unknown external outcome. + settle_error = None try: await st.durability.after_tool_interrupted(block, error_msg) - except Exception: + except Exception as exc: log.exception("Ledger settle failed for timed-out %s", block.name) + settle_error = exc try: await self._audit.log_execution( user_id=str(st.message.author.id), @@ -3183,6 +3186,8 @@ async def _run_one_tool_with_timeout( ) except Exception: pass + if settle_error is not None: + raise settle_error return { "type": "tool_result", "tool_use_id": block.id, diff --git a/src/discord/turn_resume.py b/src/discord/turn_resume.py index 8d3d29b88..0aeedbbe1 100644 --- a/src/discord/turn_resume.py +++ b/src/discord/turn_resume.py @@ -508,7 +508,8 @@ async def _validate_and_rebuild(self, key: TurnKey, row: dict): tools = self._tool_catalog.merged_definitions() tools = self._permissions.filter_tools(str(original.author.id), tools) self._repair_unmatched_tool_use( - fields["messages"], row.get("operations") or [] + fields["messages"], row.get("operations") or [], + generation_seq=payload["generation_seq"], ) except Exception: log.exception("Checkpoint reconstruction failed — rejecting") @@ -562,16 +563,17 @@ async def _validate_and_rebuild(self, key: TurnKey, row: dict): return st, original, None @staticmethod - def _repair_unmatched_tool_use(messages: list, operations: list[dict]) -> None: + def _repair_unmatched_tool_use( + messages: list, operations: list[dict], *, generation_seq: int + ) -> None: """Guarantee matched tool_use/tool_result blocks after a crash. Missing results are synthesized from the ledger: APPLIED replays the stored result; anything else states the truth (unknown / never ran). Nothing is re-executed. """ - seen_results: set[str] = set() - use_blocks: dict[str, str] = {} - for msg in messages: + open_uses: list[tuple[int, str]] = [] + for message_index, msg in enumerate(messages): content = msg.get("content") if not isinstance(content, list): continue @@ -579,27 +581,39 @@ def _repair_unmatched_tool_use(messages: list, operations: list[dict]) -> None: if not isinstance(block, dict): continue if block.get("type") == "tool_use" and block.get("id"): - use_blocks[block["id"]] = block.get("name", "tool") + open_uses.append((message_index, block["id"])) elif block.get("type") == "tool_result" and block.get("tool_use_id"): - seen_results.add(block["tool_use_id"]) - missing = [cid for cid in use_blocks if cid not in seen_results] - if not missing: + for position, (_index, cid) in enumerate(open_uses): + if cid == block["tool_use_id"]: + open_uses.pop(position) + break + if not open_uses: return - ops_by_id = {op["tool_call_id"]: op for op in operations} + if any(index != len(messages) - 1 for index, _cid in open_uses): + raise ValueError("unmatched tool_use outside the checkpoint's final message") + ops_by_id = {} + for op in operations: + identity = (op["generation_seq"], op["tool_call_id"]) + if identity in ops_by_id: + # A dict-comprehension silently selected the last ledger row. + # Duplicate durable identities are corrupt/ambiguous; never + # guess which outcome belongs to this transcript block. + raise ValueError(f"duplicate operation identity: {identity!r}") + ops_by_id[identity] = op repaired = [] - for cid in missing: - op = ops_by_id.get(cid) - if op is not None and op["state"] in ( + for _index, cid in open_uses: + matched_op = ops_by_id.get((generation_seq, cid)) + if matched_op is not None and matched_op["state"] in ( OpState.APPLIED, OpState.RECONCILED_APPLIED, ): - content = op.get("result") or "[completed; result recorded]" - elif op is None: + content = matched_op.get("result") or "[completed; result recorded]" + elif matched_op is None: content = ( "[Interrupted before execution — this call never ran; " "re-issue it if still needed.]" ) - elif op.get("effect_class") == ToolEffectClass.EFFECT_FREE_OBSERVATION: + elif matched_op.get("effect_class") == ToolEffectClass.EFFECT_FREE_OBSERVATION: content = ( "[Interrupted observation — no external effect was left " "unresolved; repeat the observation if it is still needed.]" diff --git a/src/health/server.py b/src/health/server.py index f72807f82..328dc2c11 100644 --- a/src/health/server.py +++ b/src/health/server.py @@ -1551,6 +1551,7 @@ async def _webhook_grafana(self, request: web.Request) -> web.Response: event_data: dict = { "event": "alert", "alert_name": alert_name, + "alert_names": [a.alert_name for a in parsed_alerts] or [alert_name], "alert_count": len(parsed_alerts), "firing_count": sum(1 for a in parsed_alerts if a.status == "firing"), "resolved_count": sum(1 for a in parsed_alerts if a.status == "resolved"), diff --git a/src/knowledge/store.py b/src/knowledge/store.py index 1329063ad..5ea3ef441 100644 --- a/src/knowledge/store.py +++ b/src/knowledge/store.py @@ -14,6 +14,7 @@ from datetime import UTC, datetime from typing import TYPE_CHECKING, Literal +from ..async_utils import to_thread_settled from ..odin_log import get_logger from ..search.errors import SearchExecutionError, validate_search_query from ..search.hybrid import reciprocal_rank_fusion @@ -300,9 +301,10 @@ async def _ingest_locked( return outcome # Existing rows remain searchable until replacement is verified. - old_content = await asyncio.to_thread(self.get_source_content, source) - is_update = old_content is not None - indexed = await asyncio.to_thread( + old_content = await to_thread_settled(self.get_source_content, source) + action = "update" if old_content is not None else "create" + diff_summary = self._make_diff_summary(old_content, content) + indexed = await to_thread_settled( self._write_chunks_sync, chunks, vectors, @@ -311,23 +313,9 @@ async def _ingest_locked( now, uploader, doc_content_hash, + version=(content, action, diff_summary), ) - # A version record is a success claim, not a pre-write intention. - if indexed == len(chunks): - action = "update" if is_update else "create" - diff_summary = self._make_diff_summary(old_content, content) - await asyncio.to_thread( - self._record_version, - source, - doc_content_hash, - content, - indexed, - uploader, - action, - diff_summary, - ) - log.info("Ingested '%s': %d/%d chunks indexed", source, indexed, len(chunks)) if indexed == len(chunks): return IngestOutcome(indexed, INGEST_STORED) @@ -342,21 +330,26 @@ async def _dedup_outcome( log_non_durable: bool = True, ) -> IngestOutcome | None: """Check exact and near duplicates while the caller holds the lock.""" - existing = await asyncio.to_thread(self._find_by_doc_hash, doc_content_hash) + existing = await to_thread_settled(self._find_by_doc_hash, doc_content_hash) if existing: existing_source, count = existing - if existing_source == source and await asyncio.to_thread( + same_durable = existing_source == source and await to_thread_settled( self.source_is_durable, source, count, - ): + ) + if same_durable and await to_thread_settled( + self.get_source_snapshot, source, + ) is not None: log.info( "Skipping ingest of '%s': content unchanged (hash=%s)", source, doc_content_hash[:12], ) return IngestOutcome(count, INGEST_UNCHANGED, source) - if existing_source != source and await asyncio.to_thread( + if same_durable: + log.warning("Re-ingesting '%s': missing version snapshot; repairing it", source) + elif existing_source != source and await to_thread_settled( self.source_is_durable, existing_source, count, @@ -374,16 +367,16 @@ async def _dedup_outcome( existing_source, source, ) - elif log_non_durable: + elif log_non_durable and not same_durable: log.warning( "Ignoring non-durable duplicate source '%s' while re-ingesting it", existing_source, ) hashes = [self._content_hash(chunk) for chunk in chunks] - near_dup = await asyncio.to_thread(self._find_near_duplicate, hashes, source) + near_dup = await to_thread_settled(self._find_near_duplicate, hashes, source) if near_dup: - if await asyncio.to_thread(self.source_is_durable, near_dup[0]): + if await to_thread_settled(self.source_is_durable, near_dup[0]): log.warning( "Skipping ingest of '%s': %.0f%% chunk overlap with existing source '%s'", source, @@ -408,55 +401,54 @@ def _write_chunks_sync( now: str, uploader: str, doc_content_hash: str = "", + *, + version: tuple[str, str, str] | None = None, ) -> int: - """Install a complete DB/FTS document before retiring its old rows.""" + """Publish one source with chunks, vectors, snapshot and FTS as a unit.""" conn = self._conn assert conn is not None - old_ids = { - str(row[0]) + before_rows = [ + (str(row[0]), str(row[1]), int(row[2])) for row in conn.execute( - "SELECT chunk_id FROM knowledge_chunks WHERE source = ?", + "SELECT chunk_id, content, chunk_index FROM knowledge_chunks WHERE source = ?", (source,), ).fetchall() - } + ] + old_ids = {row[0] for row in before_rows} desired_rows: list[tuple[str, str, int]] = [] - all_writes_ok = True - for i, chunk in enumerate(chunks): - # Legacy IDs are retained wherever they belong to this source. - # A short source-hash collision must never REPLACE another owner's - # DB, FTS or vector row, even for restore and dedup=False imports. - base_id = f"{doc_hash}_{i}_{doc_content_hash[:12]}" - chunk_id = base_id - suffix = hashlib.sha256(source.encode()).hexdigest() - fallback_id = f"{base_id}_{suffix}" - # Once a source has a fallback ID, keep that identity even if - # another source later releases the original short ID. - prior = conn.execute( - "SELECT source FROM knowledge_chunks WHERE chunk_id = ?", - (fallback_id,), - ).fetchone() - if prior is not None and prior[0] == source: - chunk_id = fallback_id - owner = conn.execute( - "SELECT source FROM knowledge_chunks WHERE chunk_id = ?", - (chunk_id,), - ).fetchone() - if owner is not None and owner[0] != source: - chunk_id = fallback_id - counter = 0 - while True: - owner = conn.execute( - "SELECT source FROM knowledge_chunks WHERE chunk_id = ?", - (chunk_id,), - ).fetchone() - if owner is None or owner[0] == source: - break - counter += 1 - chunk_id = f"{base_id}_{suffix}_{counter}" - log.warning("Chunk ID collision for '%s', using source-specific ID", source) - desired_rows.append((chunk_id, chunk, i)) - chunk_hash = self._content_hash(chunk) - try: + fts_touched = False + try: + for i, chunk in enumerate(chunks): + # Legacy IDs are retained wherever they belong to this source. + # A short source-hash collision must never REPLACE another owner's + # DB, FTS or vector row, even for restore and dedup=False imports. + base_id = f"{doc_hash}_{i}_{doc_content_hash[:12]}" + chunk_id = base_id + suffix = hashlib.sha256(source.encode()).hexdigest() + fallback_id = f"{base_id}_{suffix}" + # Once a source has a fallback ID, keep that identity even if + # another source later releases the original short ID. + prior = conn.execute( + "SELECT source FROM knowledge_chunks WHERE chunk_id = ?", (fallback_id,), + ).fetchone() + if prior is not None and prior[0] == source: + chunk_id = fallback_id + owner = conn.execute( + "SELECT source FROM knowledge_chunks WHERE chunk_id = ?", (chunk_id,), + ).fetchone() + if owner is not None and owner[0] != source: + chunk_id = fallback_id + counter = 0 + while True: + owner = conn.execute( + "SELECT source FROM knowledge_chunks WHERE chunk_id = ?", (chunk_id,), + ).fetchone() + if owner is None or owner[0] == source: + break + counter += 1 + chunk_id = f"{base_id}_{suffix}_{counter}" + log.warning("Chunk ID collision for '%s', using source-specific ID", source) + desired_rows.append((chunk_id, chunk, i)) conn.execute( "INSERT OR REPLACE INTO knowledge_chunks " "(chunk_id, content, source, chunk_index, total_chunks, " @@ -470,18 +462,10 @@ def _write_chunks_sync( len(chunks), uploader, now, - chunk_hash, + self._content_hash(chunk), doc_content_hash, ), ) - if self._fts and not self._fts.index_knowledge_chunk( - chunk_id, - chunk, - source, - i, - ): - all_writes_ok = False - log.error("Failed to index chunk %d of '%s' in FTS", i, source) if vectors[i] is not None: vec_bytes = serialize_vector(vectors[i]) # type: ignore[arg-type] conn.execute( @@ -492,40 +476,11 @@ def _write_chunks_sync( "INSERT INTO knowledge_vec (chunk_id, embedding) VALUES (?, ?)", (chunk_id, vec_bytes), ) - except Exception as exc: - all_writes_ok = False - log.error("Failed to index chunk %d of '%s': %s", i, source, exc) - try: - if all_writes_ok: - conn.commit() - else: - conn.rollback() - except Exception as exc: - conn.rollback() - log.error("Failed to commit replacement for '%s': %s", source, exc) - return 0 - - if not all_writes_ok or not self._source_contains_rows(source, desired_rows): - log.error("Failed durable DB/FTS verification for '%s' after ingest", source) - return 0 - - desired_ids = {row[0] for row in desired_rows} - obsolete_ids = old_ids - desired_ids - if obsolete_ids: - obsolete_fts_ids: set[str] = set() - if self._fts: - obsolete_fts_ids = { - chunk_id for chunk_id in obsolete_ids if self._fts.has_knowledge_chunk(chunk_id) - } - if obsolete_fts_ids: - removed_fts = self._fts.delete_knowledge_chunks(obsolete_fts_ids) - if removed_fts != len(obsolete_fts_ids) or any( - self._fts.has_knowledge_chunk(chunk_id) for chunk_id in obsolete_fts_ids - ): - log.error("Failed to retire old FTS rows for '%s'", source) - return 0 - placeholders = ",".join("?" for _ in obsolete_ids) - try: + elif self._has_vec: + conn.execute("DELETE FROM knowledge_vec WHERE chunk_id = ?", (chunk_id,)) + obsolete_ids = old_ids - {row[0] for row in desired_rows} + if obsolete_ids: + placeholders = ",".join("?" for _ in obsolete_ids) if self._has_vec: conn.execute( f"DELETE FROM knowledge_vec WHERE chunk_id IN ({placeholders})", @@ -535,11 +490,54 @@ def _write_chunks_sync( f"DELETE FROM knowledge_chunks WHERE chunk_id IN ({placeholders})", tuple(obsolete_ids), ) - conn.commit() - except Exception as exc: + if version is not None: + content, action, diff_summary = version + self._insert_version_row( + source, doc_content_hash, content, len(chunks), uploader, action, diff_summary, + ) + staged = sorted( + (str(row[0]), str(row[1]), int(row[2])) + for row in conn.execute( + "SELECT chunk_id, content, chunk_index FROM knowledge_chunks WHERE source = ?", + (source,), + ).fetchall() + ) + if staged != sorted(desired_rows): + raise RuntimeError("staged DB rows do not match the replacement") + if self._has_vec: + expected_vector_ids = { + desired_rows[i][0] for i, vector in enumerate(vectors) if vector is not None + } + vector_ids = { + str(row[0]) + for row in conn.execute( + "SELECT v.chunk_id FROM knowledge_vec v " + "JOIN knowledge_chunks c ON c.chunk_id = v.chunk_id " + "WHERE c.source = ?", (source,), + ).fetchall() + } + if vector_ids != expected_vector_ids: + raise RuntimeError("staged vectors do not match the replacement") + if self._fts is not None: + if not self._fts.available: + raise RuntimeError("FTS store is unavailable") + if not self._fts.replace_knowledge_source(source, desired_rows): + raise RuntimeError("FTS replacement failed") + fts_touched = True + fts_rows = self._fts.get_knowledge_source_rows(source) + if fts_rows is None or sorted(fts_rows) != sorted(desired_rows): + raise RuntimeError("FTS rows do not match the replacement") + conn.commit() + except Exception as exc: + try: conn.rollback() - log.error("Failed to retire old DB rows for '%s': %s", source, exc) - return 0 + except Exception as rollback_exc: + log.error("Rollback failed for '%s': %s", source, rollback_exc) + if fts_touched and self._fts is not None: + if not self._fts.replace_knowledge_source(source, before_rows): + log.error("FTS compensation failed for '%s'", source) + log.error("Failed to publish replacement for '%s': %s", source, exc) + return 0 if not self.source_is_durable( source, @@ -548,51 +546,8 @@ def _write_chunks_sync( ): log.error("Final durable DB/FTS verification failed for '%s'", source) return 0 - if self._has_vec: - expected_vector_ids = { - desired_rows[i][0] for i, vector in enumerate(vectors) if vector is not None - } - vector_ids = { - str(row[0]) - for row in conn.execute( - "SELECT v.chunk_id FROM knowledge_vec v " - "JOIN knowledge_chunks c ON c.chunk_id = v.chunk_id " - "WHERE c.source = ?", - (source,), - ).fetchall() - } - if vector_ids != expected_vector_ids: - log.error("Final vector verification failed for '%s'", source) - return 0 return len(chunks) - def _source_contains_rows( - self, - source: str, - expected_rows: list[tuple[str, str, int]], - ) -> bool: - """Verify expected chunk rows exist in both DB and configured FTS.""" - if not self.available: - return False - try: - db_rows = self._conn.execute( # type: ignore[union-attr] - "SELECT chunk_id, content, chunk_index FROM knowledge_chunks WHERE source = ?", - (source,), - ).fetchall() - expected = set(expected_rows) - db_set = {(str(row[0]), str(row[1]), int(row[2])) for row in db_rows} - if not expected.issubset(db_set): - return False - if self._fts is None: - return True - if not self._fts.available: - return False - fts_rows = self._fts.get_knowledge_source_rows(source) - return fts_rows is not None and expected.issubset(set(fts_rows)) - except Exception as exc: - log.error("Chunk-row verification failed for '%s': %s", source, exc) - return False - async def search( self, query: str, @@ -890,6 +845,10 @@ def delete_source(self, source: str, *, _record_version: bool = True) -> int: log.info("Deleted %d chunks for source '%s'", len(ids), source) return len(ids) except Exception as e: + try: + self._conn.rollback() # type: ignore[union-attr] + except Exception as rollback_exc: + log.error("Delete rollback failed for '%s': %s", source, rollback_exc) log.error("Failed to delete source '%s': %s", source, e) return 0 @@ -925,7 +884,15 @@ def delete_source_confirmed( (source,), ).fetchall() if not rows: - return 0 + orphans = self._fts.count_knowledge_source(source) + if not orphans: + return 0 + self._fts.delete_knowledge_source(source) + if self._fts.has_knowledge_source(source): + log.error("Failed to remove %d orphan FTS rows for '%s'", orphans, source) + return 0 + log.warning("Removed %d orphan FTS rows for '%s'", orphans, source) + return orphans content = self.get_source_content(source) if _record_version else None content_hash = self._content_hash(content) if content else "" @@ -943,20 +910,22 @@ def delete_source_confirmed( ) return 0 if self._fts: - if not self.source_is_durable( - source, - expected_chunks=len(rows), - require_fts=True, - expected_content_hash=expected_source_hash, + if expected_source_hash is not None and any( + str(row[0] or "") != expected_source_hash + for row in conn.execute( + "SELECT doc_content_hash FROM knowledge_chunks WHERE source = ?", + (source,), + ).fetchall() ): log.error( - "Confirmed delete refused for '%s': DB/FTS rows do not match", + "Confirmed delete refused for '%s': stored content is not expected", source, ) return 0 + fts_before = self._fts.count_knowledge_source(source) removed_fts = self._fts.delete_knowledge_source(source) fts_remains = self._fts.has_knowledge_source(source) - if removed_fts != len(rows) or fts_remains: + if removed_fts != fts_before or fts_remains: # delete_knowledge_source may commit and then report/fail; # restore from the still-authoritative DB snapshot before # refusing the migration. @@ -1073,7 +1042,7 @@ async def delete_source_confirmed_async( ) -> int: """Run migration-safe confirmed deletion under the async write lock.""" async with self._write_lock: - return await asyncio.to_thread( + return await to_thread_settled( self.delete_source_confirmed, source, survivor_source=survivor_source, @@ -1091,12 +1060,12 @@ async def delete_source_async(self, source: str) -> int: could commit half-written chunks. Callers in async contexts must use this wrapper (it also moves the blocking DB work to a thread).""" async with self._write_lock: - return await asyncio.to_thread(self.delete_source, source) + return await to_thread_settled(self.delete_source, source) async def merge_sources_async(self, keep_source: str, remove_source: str) -> int: """Merge sources under the write lock, off the event loop (see delete_source_async).""" async with self._write_lock: - return await asyncio.to_thread(self.merge_sources, keep_source, remove_source) + return await to_thread_settled(self.merge_sources, keep_source, remove_source) async def search_hybrid( self, @@ -1139,7 +1108,7 @@ async def search_hybrid( ) def backfill_fts(self) -> int: - """Index existing knowledge chunks into FTS5. Returns count indexed.""" + """Reconcile FTS against DB, removing orphan rows and indexing missing rows.""" if not self._fts or not self.available: return 0 try: @@ -1149,6 +1118,18 @@ def backfill_fts(self) -> int: except Exception: return 0 + db_ids = {str(row[0]) for row in rows} + inventory = self._fts.knowledge_chunk_sources() + if inventory is None: + return 0 + orphan_ids = {chunk_id for chunk_id, _source in inventory if chunk_id not in db_ids} + if orphan_ids: + removed = self._fts.delete_knowledge_chunks(orphan_ids) + log.warning( + "Removed %d orphan knowledge FTS row(s) for %d chunk id(s) no document owns", + removed, len(orphan_ids), + ) + count = 0 for row in rows: chunk_id, content, source, chunk_index = row @@ -1165,6 +1146,11 @@ def backfill_fts(self) -> int: count += 1 return count + async def backfill_fts_async(self) -> int: + """Run reconciliation under the write lock, settled through cancellation.""" + async with self._write_lock: + return await to_thread_settled(self.backfill_fts) + # ------------------------------------------------------------------ # Deduplication helpers # ------------------------------------------------------------------ @@ -1332,6 +1318,30 @@ def _next_version(self, source: str) -> int: ).fetchone() return (row[0] or 0) + 1 + def _insert_version_row( + self, + source: str, + content_hash: str, + content: str | None, + chunk_count: int, + uploader: str, + action: str, + diff_summary: str = "", + ) -> int: + """Stage a version row in the caller's open transaction.""" + version = self._next_version(source) + self._conn.execute( # type: ignore[union-attr] + "INSERT INTO knowledge_versions " + "(source, version, content_hash, content, chunk_count, " + "uploader, action, created_at, diff_summary) " + "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)", + ( + source, version, content_hash, content, chunk_count, + uploader, action, datetime.now(UTC).isoformat(), diff_summary, + ), + ) + return version + def _record_version( self, source: str, @@ -1342,32 +1352,20 @@ def _record_version( action: str, diff_summary: str = "", ) -> int: - """Record a version entry. Returns the version number.""" + """Record a version entry. Returns the version number (0 on failure).""" if not self._conn: return 0 try: - version = self._next_version(source) - now = datetime.now(UTC).isoformat() - self._conn.execute( - "INSERT INTO knowledge_versions " - "(source, version, content_hash, content, chunk_count, " - "uploader, action, created_at, diff_summary) " - "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)", - ( - source, - version, - content_hash, - content, - chunk_count, - uploader, - action, - now, - diff_summary, - ), + version = self._insert_version_row( + source, content_hash, content, chunk_count, uploader, action, diff_summary, ) self._conn.commit() return version except Exception as e: + try: + self._conn.rollback() + except Exception as rollback_exc: + log.error("Version rollback failed for '%s': %s", source, rollback_exc) log.error("Failed to record version for '%s': %s", source, e) return 0 diff --git a/src/learning/reflector.py b/src/learning/reflector.py index 0a7b822bf..bcb8554a3 100644 --- a/src/learning/reflector.py +++ b/src/learning/reflector.py @@ -7,6 +7,7 @@ from pathlib import Path from typing import TYPE_CHECKING +from ..async_utils import to_thread_settled from ..odin_log import get_logger if TYPE_CHECKING: @@ -439,7 +440,7 @@ async def delete_entry_async(self, key: str) -> bool: """Delete an entry under the reflection lock, so it can't be resurrected by a concurrent reflection/consolidation writing a stale snapshot.""" async with self._lock: - result = await asyncio.to_thread(self.delete_entry, key) + result = await to_thread_settled(self.delete_entry, key) self.invalidate_cache() return result @@ -483,7 +484,7 @@ async def update_entry_async( ) -> dict | None: """Update an entry under the reflection lock (see delete_entry_async).""" async with self._lock: - result = await asyncio.to_thread(self.update_entry, key, content, category) + result = await to_thread_settled(self.update_entry, key, content, category) self.invalidate_cache() return result @@ -801,7 +802,7 @@ async def reflect_on_operation( if not self._policy_allows(policy_token): return try: - data = await asyncio.to_thread(self._load_for_write) + data = await to_thread_settled(self._load_for_write) except StoreCorruptError as exc: log.error( "Skipping reflection merge — learned.json corrupt " @@ -893,7 +894,7 @@ async def _reflect( from ..json_store import StoreCorruptError try: - data = await asyncio.to_thread(self._load_for_write) + data = await to_thread_settled(self._load_for_write) except StoreCorruptError as exc: log.error( "Skipping reflection — learned.json corrupt (backup preserved): %s", exc diff --git a/src/llm/context_budget.py b/src/llm/context_budget.py index 9be9e3da9..6a2571b10 100644 --- a/src/llm/context_budget.py +++ b/src/llm/context_budget.py @@ -71,8 +71,8 @@ FIXED_ENVELOPE_RESERVE_TOKENS = 42_000 # Compatible requests never ask an endpoint for more output than this. Profile -# declarations may be much larger, but reserving unreachable output would make -# otherwise viable agent models fail admission. +# declarations may be much larger; every budget reserves exactly what the +# request asks for. COMPATIBLE_REQUEST_OUTPUT_CEILING = 32_768 #: Default chars-per-token, expressed in MILLICHARS per token so derivations @@ -468,7 +468,6 @@ def compatible_agent_unavailable_reason( model, compatible_config, max_context_chars=None, - output_reserve_ceiling=COMPATIBLE_REQUEST_OUTPUT_CEILING, ) if snapshot.working_budget < COMPATIBLE_RESCUE_MIN_USABLE_TOKENS: return ( @@ -478,9 +477,13 @@ def compatible_agent_unavailable_reason( return None -def compatible_usable_input_tokens( - profile: object | None, *, output_reserve_ceiling: int | None = None -) -> int | None: +def compatible_request_output_tokens(profile: object | None) -> int | None: + """One output-reserve rule shared by request, admission, runtime snapshots and rescue.""" + cap = getattr(profile, "max_output_tokens", None) + return min(cap, COMPATIBLE_REQUEST_OUTPUT_CEILING) if type(cap) is int and cap > 0 else None + + +def compatible_usable_input_tokens(profile: object | None) -> int | None: """Derive usable prompt space from a profile's total window and output. Legacy profile-shaped test/config objects exposing only @@ -493,9 +496,9 @@ def compatible_usable_input_tokens( output = getattr(profile, "max_output_tokens", None) if total is not None and output is not None: try: - reserve = int(output) - if output_reserve_ceiling is not None: - reserve = min(reserve, output_reserve_ceiling) + reserve = compatible_request_output_tokens(profile) + if reserve is None: + return None return max(0, int(total) - reserve) except (TypeError, ValueError): return None @@ -510,16 +513,13 @@ def snapshot_for_compatible_profile( compatible_config: object, *, max_context_chars: int | None, - output_reserve_ceiling: int | None = None, ) -> ContextBudgetSnapshot: """Resolve compatible budgets post-utilization, without Codex policy floors.""" canonical = canonical_compatible_model(model) profile = compatible_model_profile(canonical, compatible_config) # Unknown compatible models have no claimed window. Keep ordinary history # compaction total, but do not qualify them for rescue. - usable = compatible_usable_input_tokens( - profile, output_reserve_ceiling=output_reserve_ceiling - ) or 0 + usable = compatible_usable_input_tokens(profile) or 0 source = "compatible_profile" if profile is not None else "unknown_compatible" utilization = int(getattr(compatible_config, "context_utilization", 100)) utilization = max(0, min(100, utilization)) diff --git a/src/llm/context_compressor.py b/src/llm/context_compressor.py index 59a045b5b..1b9d9651c 100644 --- a/src/llm/context_compressor.py +++ b/src/llm/context_compressor.py @@ -271,6 +271,10 @@ def estimate_message_images(messages: list[dict]) -> int: return total +_COUNTED_BLOCK_KEYS = ("text", "content", "input", "arguments", "reasoning_content") +_ELIDABLE_BLOCK_KEYS = ("text", "content", "input", "arguments") + + def estimate_message_chars(messages: list[dict]) -> int: """Estimate total character payload across a message list.""" total = 0 @@ -282,7 +286,15 @@ def estimate_message_chars(messages: list[dict]) -> int: for block in content: if not isinstance(block, dict): continue - for key in ("text", "content", "input", "arguments"): + # OpenAI-compatible chat serializes list-valued tool results + # with json.dumps as the tool message's content. Charge the + # same payload here; otherwise large structured results are + # invisible to context budgeting even though they reach wire. + if block.get("type") == "tool_result" and isinstance( + block.get("content"), list + ): + total += len(json.dumps(block["content"], default=str)) + for key in _COUNTED_BLOCK_KEYS: val = block.get(key) if val is None: continue @@ -309,12 +321,26 @@ def estimate_message_chars(messages: list[dict]) -> int: ) +def _failure_label(result: str) -> str: + from ..tools.tool_text import failure_reason + + return f"ERR ({failure_reason(result) or 'error'})" + + +def _outcome_from_text(result: str) -> str: + from ..tools.tool_text import failure_reason + + if failure_reason(result) or result.startswith(_ERROR_PREFIXES): + return _failure_label(result) + return "OK" + + def summarize_iteration(iteration: list[dict]) -> str: """Produce a compact ``tool_name→OK/ERR`` summary for one iteration.""" tool_names: list[str] = [] call_ids: list[str | None] = [] outcomes_by_id: dict[str, str] = {} - outcomes: list[str] = [] + unkeyed_outcomes: list[str] = [] for msg in iteration: if _is_injected_directive(msg): @@ -342,12 +368,13 @@ def summarize_iteration(iteration: list[dict]) -> str: if block.get("status"): outcome = str(block["status"]) elif "is_error" in block: - outcome = "ERR" if block["is_error"] else "OK" + outcome = _failure_label(result) if block["is_error"] else "OK" else: - outcome = "ERR" if result.startswith(_ERROR_PREFIXES) else "OK" - outcomes.append(outcome) + outcome = _outcome_from_text(result) if block.get("tool_use_id"): outcomes_by_id[block["tool_use_id"]] = outcome + else: + unkeyed_outcomes.append(outcome) # Agent-style string tool results: "[Tool result: tool_name]\n..." elif isinstance(content, str) and content.startswith("[Tool result:"): # Extract tool name from "[Tool result: tool_name]" @@ -357,17 +384,20 @@ def summarize_iteration(iteration: list[dict]) -> str: tool_names.append(name) call_ids.append(None) result_body = content[end + 1 :].strip() if end > 0 else content - if result_body.startswith(_ERROR_PREFIXES): - outcomes.append("ERR") - else: - outcomes.append("OK") + unkeyed_outcomes.append(_outcome_from_text(result_body)) parts = [] - for i, name in enumerate(tool_names): - outcome = outcomes[i] if i < len(outcomes) else "?" - call_id = call_ids[i] + unkeyed_index = 0 + for name, call_id in zip(tool_names, call_ids, strict=True): if call_id: - outcome = outcomes_by_id.get(call_id, outcome) + outcome = outcomes_by_id.get(call_id, "?") + else: + outcome = ( + unkeyed_outcomes[unkeyed_index] + if unkeyed_index < len(unkeyed_outcomes) + else "?" + ) + unkeyed_index += 1 parts.append(f"{name}\u2192{outcome}") summary = ", ".join(parts) @@ -560,7 +590,8 @@ def _truncate_iteration(iteration: list[dict], max_chars: int) -> tuple[list[dic for block in content: if not isinstance(block, dict) or block.get("type") == "tool_use": continue - for key in ("text", "content", "input", "arguments"): + # Preserved reasoning is counted, but only its whole iteration may be removed. + for key in _ELIDABLE_BLOCK_KEYS: value = block.get(key) if isinstance(value, str): strings.append((block, key, value)) @@ -588,6 +619,18 @@ def _truncate_iteration(iteration: list[dict], max_chars: int) -> tuple[list[dic return work, max(0, original_chars - compressed_chars) +def _preserved_reasoning_chars(iteration: list[dict]) -> int: + return sum( + len(block["reasoning_content"]) + for msg in iteration + if isinstance(msg.get("content"), list) + for block in msg["content"] + if isinstance(block, dict) + and block.get("type") == "reasoning_content" + and isinstance(block.get("reasoning_content"), str) + ) + + def _is_emergency_summary(msg: dict) -> bool: """Identify the exact user-message shape emitted by this compressor.""" content = msg.get("content") @@ -945,8 +988,9 @@ def _assemble( for idx in range(len(iterations) - 1, -1, -1): iteration = iterations[idx] size = _iteration_chars(iteration) - if size > single_cap: - iteration, elided = _truncate_iteration(iteration, single_cap) + iteration_cap = single_cap + _preserved_reasoning_chars(iteration) + if size > iteration_cap: + iteration, elided = _truncate_iteration(iteration, iteration_cap) if elided: report["results_truncated"] += 1 report["chars_elided"] += elided @@ -1005,7 +1049,9 @@ def _assemble( report["compressed_chars"] = compressed_chars report["fits"] = compressed_chars <= target_chars if not report["fits"]: - report["unfit_reason"] = "immutable newest call or control messages exceed target" + report["unfit_reason"] = ( + "immutable newest call, preserved reasoning or control messages exceed target" + ) return messages, report if stats: stats.compressions += 1 diff --git a/src/llm/openai_compatible.py b/src/llm/openai_compatible.py index fc5b63d48..4d49a7af8 100644 --- a/src/llm/openai_compatible.py +++ b/src/llm/openai_compatible.py @@ -14,7 +14,7 @@ from .backoff import DEFAULT_BASE_DELAY, DEFAULT_MAX_DELAY, DEFAULT_MAX_RETRIES, compute_backoff from .circuit_breaker import CircuitBreaker from .client_lifecycle import leased_call -from .context_budget import COMPATIBLE_REQUEST_OUTPUT_CEILING +from .context_budget import canonical_compatible_model, compatible_request_output_tokens from .errors import LLMContextLengthError, LLMRateLimitError, LLMRequestError, LLMTransportError from .progress import GenerationProgress, GenerationProgressObserver, emit_progress from .provider import LLMProvider @@ -328,17 +328,13 @@ def _request_max_tokens( """Resolve the per-request output cap without treating zero as unset.""" if type(requested) is int and requested > 0: return requested - from .context_budget import canonical_compatible_model - canonical = canonical_compatible_model(model or self.model) profile = self.model_profiles.get(canonical) if profile is None and self.openrouter_routing is not None: derived = getattr(self.openrouter_routing, "catalogue_profiles", {}) or {} profile = derived.get(canonical) - profile_cap = getattr(profile, "max_output_tokens", None) - if type(profile_cap) is int and profile_cap > 0: - return min(profile_cap, COMPATIBLE_REQUEST_OUTPUT_CEILING) - return self.max_tokens + cap = compatible_request_output_tokens(profile) + return cap if cap is not None else self.max_tokens def _preserves_reasoning_content(self) -> bool: """Whether this endpoint explicitly requires preserved-thinking replay.""" diff --git a/src/llm/tool_history.py b/src/llm/tool_history.py index 529336b0a..7b6347676 100644 --- a/src/llm/tool_history.py +++ b/src/llm/tool_history.py @@ -26,7 +26,8 @@ def normalize_tool_calls(calls, *, used_ids: set[str] | None = None) -> list[dic A duplicate identity receives a fresh correlation ID and a paired parse error, not permission to execute an ambiguous call. Legacy callbacks with - no ID remain supported. The caller owns the lifetime identity set. + no ID remain supported. The caller owns any supplied identity set; agents + scope that set to one provider reply, not their entire transcript. """ seen = used_ids if used_ids is not None else set() normalized = [] @@ -95,6 +96,10 @@ def settled_call_ids(messages: list[dict]) -> set[str]: continue if message.get("role") == "assistant" and pending: raise ValueError("Unresolved native tool calls before assistant generation") + if message.get("role") == "assistant": + # A provider can reuse an ID in its next reply. Reject ambiguity + # within one reply, not across already-settled generations. + seen.clear() for block in content: if not isinstance(block, dict): continue diff --git a/src/scheduler/scheduler.py b/src/scheduler/scheduler.py index 725aa2edc..085b43afc 100644 --- a/src/scheduler/scheduler.py +++ b/src/scheduler/scheduler.py @@ -195,6 +195,41 @@ def __init__(self, data_path: str, history_path: str | None = None) -> None: self._http_session: aiohttp.ClientSession | None = None self._load() self._degrade_removed_trigger_sources() + for schedule in self._schedules: + self._resolve_interrupted_run(schedule) + + REPLAY_SAFE_ONE_TIME_ACTIONS = frozenset({"reminder", "digest"}) + + @classmethod + def _tracks_run_start(cls, schedule: dict) -> bool: + return ( + bool(schedule.get("one_time")) + and schedule.get("action") not in cls.REPLAY_SAFE_ONE_TIME_ACTIONS + ) + + def _resolve_interrupted_run(self, schedule: dict) -> bool: + started = schedule.pop("run_started_at", None) + if not started or not self._tracks_run_start(schedule): + return False + self._quarantine_schedule( + schedule, + f"One-time schedule started at {started!r} but its completion was never recorded " + "(Odin stopped or could not save the result); it may have partly run, so it was not " + "run again. Check what it did, then set a new run_at to re-arm it", + ) + return True + + async def _mark_run_started(self, schedule: dict) -> None: + started = datetime.now(UTC).isoformat() + sid = schedule.get("id") + async with self._lock: + candidate = copy.deepcopy(self._schedules) + for current in candidate: + if current.get("id") == sid: + current["run_started_at"] = started + await self._publish(candidate) + schedule["run_started_at"] = started + return def _load(self) -> None: if self.data_path.exists(): @@ -789,7 +824,8 @@ def _trigger_matches(trigger: dict, source: str, event_data: dict) -> bool: - source: exact match (required if specified) - event: exact match against event_data["event"] - repo: case-insensitive substring match against event_data["repo"] - - alert_name: case-insensitive substring match against event_data["alert_name"] + - alert_name: case-insensitive substring match against any string in + event_data["alert_names"], falling back to event_data["alert_name"] All specified fields must match (AND logic). """ @@ -803,8 +839,16 @@ def _trigger_matches(trigger: dict, source: str, event_data: dict) -> bool: if trigger["repo"].lower() not in repo.lower(): return False if trigger.get("alert_name"): - alert = event_data.get("alert_name", "") - if trigger["alert_name"].lower() not in alert.lower(): + alert_names = event_data.get("alert_names") + if alert_names is None: + alert_names = [event_data.get("alert_name", "")] + elif not isinstance(alert_names, (list, tuple)): + alert_names = [] + if not any( + isinstance(name, str) + and trigger["alert_name"].lower() in name.lower() + for name in alert_names + ): return False return True @@ -1009,6 +1053,7 @@ async def update( # retry_at that caused this schedule to be quarantined. target.pop("retry_at", None) target.pop("inert_reason", None) + target.pop("run_started_at", None) if trigger is not None: self._validate_trigger(trigger) @@ -1105,10 +1150,15 @@ async def run_now(self, schedule_id: str) -> dict: """ schedule: dict | None = None async with self._lock: + resolved = False for s in self._schedules: if s["id"] == schedule_id: + if schedule_id not in self._in_flight: + resolved = self._resolve_interrupted_run(s) schedule = copy.deepcopy(s) break + if resolved: + await self._publish(copy.deepcopy(self._schedules)) if schedule is None: raise ValueError(f"Schedule '{schedule_id}' not found") if schedule.get("inert_reason"): @@ -1305,6 +1355,8 @@ async def _execute_and_record( ): await self._restore_unstarted_reservation(schedule, reservation) raise ScheduleConnectionUnavailableError(snapshot) + if self._tracks_run_start(schedule): + await self._mark_run_started(schedule) identity = self._execution_identity(schedule) nonce = uuid.uuid4().hex self._active_execution_nonces.add(nonce) @@ -1317,8 +1369,13 @@ async def _execute_and_record( async with self._lock: candidate = copy.deepcopy(self._schedules) for current in candidate: - if current["id"] != sid or self._execution_identity(current) != identity: + if current["id"] != sid: continue + if current.get("run_started_at") == schedule.get("run_started_at"): + current.pop("run_started_at", None) + if self._execution_identity(current) != identity: + await self._publish(candidate) + break for key in ("last_run", "consecutive_failures", "retry_count", "last_error", "last_error_at", "retry_at"): if key in schedule: @@ -1546,6 +1603,13 @@ async def _tick(self) -> None: for schedule in candidate: if schedule.get("paused"): continue + if ( + schedule.get("run_started_at") + and schedule.get("id") not in self._in_flight + ): + if self._resolve_interrupted_run(schedule): + quarantined = True + continue # Do not persist/rollback a retry reservation on every tick # while a manual run already owns this schedule. if schedule.get("id") in self._in_flight and schedule.get("retry_at"): diff --git a/src/search/fts.py b/src/search/fts.py index 99855c4ee..f6bb8669c 100644 --- a/src/search/fts.py +++ b/src/search/fts.py @@ -218,6 +218,46 @@ def index_knowledge_chunk( log.error("FTS knowledge index failed for %s: %s", chunk_id, e) return False + def replace_knowledge_source( + self, source: str, rows: Iterable[tuple[str, str, int]], + ) -> bool: + """Replace every FTS row for a source in one transaction.""" + if not self._conn: + return False + try: + with self._write_lock: + try: + self._conn.execute("DELETE FROM knowledge_fts WHERE source = ?", (source,)) + for chunk_id, content, chunk_index in rows: + self._conn.execute( + "DELETE FROM knowledge_fts WHERE chunk_id = ?", (chunk_id,), + ) + self._conn.execute( + "INSERT INTO knowledge_fts (chunk_id, content, source, chunk_index) " + "VALUES (?, ?, ?, ?)", + (chunk_id, content, source, str(chunk_index)), + ) + self._conn.commit() + except Exception: + self._rollback_after_failure() + raise + return True + except Exception as exc: + log.error("FTS knowledge replace failed for '%s': %s", source, exc) + return False + + def knowledge_chunk_sources(self) -> list[tuple[str, str]] | None: + """Inventory all FTS chunk owners; None means unreadable, not empty.""" + if not self._conn: + return None + try: + with self._write_lock: + rows = self._conn.execute("SELECT chunk_id, source FROM knowledge_fts").fetchall() + return [(str(row[0]), str(row[1])) for row in rows] + except Exception as exc: + log.error("FTS knowledge inventory failed: %s", exc) + return None + def search_knowledge(self, query: str, limit: int = 20) -> list[dict]: if not self._conn: raise SearchExecutionError("full-text search is unavailable") diff --git a/src/search/vectorstore.py b/src/search/vectorstore.py index a67fbdb6f..69a19ff2e 100644 --- a/src/search/vectorstore.py +++ b/src/search/vectorstore.py @@ -6,6 +6,7 @@ from __future__ import annotations import asyncio +import hashlib import json import sqlite3 import threading @@ -28,6 +29,13 @@ VECTOR_DIM = 384 # must match LocalEmbedder.DIMENSIONS +def summary_segment_doc_id(channel_id: str, seg: dict) -> str: + """Stable identity across restored archives containing the same segment.""" + payload = json.dumps([channel_id, seg.get("start_ts"), seg.get("end_ts"), + seg.get("summary", "")]) + return "seg:" + hashlib.sha256(payload.encode()).hexdigest()[:24] + + class SessionVectorStore: """Semantic + FTS search over archived session conversations.""" @@ -47,6 +55,7 @@ def __init__(self, db_path: str, fts_index: FullTextIndex | None = None) -> None # wrap in ``asyncio.to_thread``. WAL mode still allows concurrent # reads, so reads stay unlocked. self._write_lock = threading.Lock() + self._segment_backfill_lock = asyncio.Lock() try: conn = sqlite3.connect(db_path, check_same_thread=False) conn.execute("PRAGMA journal_mode=WAL") @@ -70,6 +79,8 @@ def __init__(self, db_path: str, fts_index: FullTextIndex | None = None) -> None conn.execute( "CREATE INDEX IF NOT EXISTS idx_session_channel ON session_archives(channel_id)" ) + conn.execute("CREATE TABLE IF NOT EXISTS segment_index_state (" + "archive_doc_id TEXT PRIMARY KEY, indexed_at REAL NOT NULL)") # Vector table (only if sqlite-vec loaded) if self._has_vec: conn.execute(f""" @@ -100,8 +111,10 @@ async def index_session(self, archive_path: Path, embedder: LocalEmbedder) -> bo doc_id = archive_path.stem # e.g. "channelid_timestamp" doc_text = self._build_document_text(data) + # A segment-only archive has no legacy document, but still needs its + # independent, searchable segment documents. if not doc_text: - return False + return await self._index_segments(data, doc_id, embedder) channel_id = str(data.get("channel_id", "")) last_active = float(data.get("last_active", 0)) @@ -120,12 +133,85 @@ async def index_session(self, archive_path: Path, embedder: LocalEmbedder) -> bo self._write_session_sync, doc_id, doc_text, channel_id, last_active, message_count, vector, ) + if not await self._index_segments(data, doc_id, embedder): + return False log.info("Indexed session %s for search", doc_id) return True except Exception as e: log.error("Session index failed for %s: %s", doc_id, e) return False + async def _index_segments(self, data: dict, archive_doc_id: str, + embedder: LocalEmbedder | None) -> bool: + """Write missing segments, then acknowledge the entire archive. + + FTS has its own database: the state row is committed only after each + FTS write acknowledges. An interrupted run repairs missing FTS rows. + """ + try: + channel_id = str(data.get("channel_id", "")) + missing: list[tuple[str, str, str, float, int, list[float] | None]] = [] + for seg in data.get("summary_segments", []): + text = seg.get("summary", "") + if not text: + continue + doc_id = summary_segment_doc_id(channel_id, seg) + exists = await asyncio.to_thread(self._segment_exists_sync, doc_id) + indexed = (await asyncio.to_thread(self._fts.has_session, doc_id) + if self._fts else True) + if exists and indexed: + continue + vector = None + if not exists and self._has_vec and embedder: + vector = await embedder.embed(text) + missing.append((doc_id, text, channel_id, float(seg.get("end_ts", 0)), + int(seg.get("source_count", 0)), vector)) + await asyncio.to_thread(self._write_segment_archive_sync, archive_doc_id, missing) + return True + except Exception as e: + log.error("Segment indexing failed for %s: %s", archive_doc_id, e) + return False + + def _segment_exists_sync(self, doc_id: str) -> bool: + return self._conn.execute( # type: ignore[union-attr] + "SELECT 1 FROM session_archives WHERE doc_id = ?", (doc_id,), + ).fetchone() is not None + + def _write_segment_archive_sync( + self, archive_doc_id: str, + missing: list[tuple[str, str, str, float, int, list[float] | None]], + ) -> None: + """Commit all missing metadata, vectors, and completion in one transaction.""" + with self._write_lock: + try: + for doc_id, text, channel_id, end_ts, source_count, vector in missing: + self._conn.execute( # type: ignore[union-attr] + "INSERT OR REPLACE INTO session_archives " + "(doc_id, content, channel_id, last_active, message_count) " + "VALUES (?, ?, ?, ?, ?)", + (doc_id, text, channel_id, end_ts, source_count), + ) + if self._fts and not self._fts.index_session( + doc_id, text, channel_id, end_ts): + raise RuntimeError("FTS did not acknowledge segment indexing") + if vector is not None: + self._conn.execute( # type: ignore[union-attr] + "DELETE FROM session_vec WHERE doc_id = ?", (doc_id,), + ) + self._conn.execute( # type: ignore[union-attr] + "INSERT INTO session_vec (doc_id, embedding) VALUES (?, ?)", + (doc_id, serialize_vector(vector)), + ) + self._conn.execute( # type: ignore[union-attr] + "INSERT OR REPLACE INTO segment_index_state " + "(archive_doc_id, indexed_at) VALUES (?, strftime('%s','now'))", + (archive_doc_id,), + ) + self._conn.commit() # type: ignore[union-attr] + except BaseException: + self._conn.rollback() # type: ignore[union-attr] + raise + @staticmethod def _read_archive_sync(archive_path: Path) -> dict: """Read and parse archive JSON (sync, for use with asyncio.to_thread).""" @@ -266,6 +352,52 @@ async def backfill(self, archive_dir: Path, embedder: LocalEmbedder) -> int: return count + async def backfill_segments(self, archive_dir: Path, embedder: LocalEmbedder) -> int: + """Bounded, single-flight legacy segment migration, one archive at a time.""" + if not self.available or not archive_dir.exists(): + return 0 + async with self._segment_backfill_lock: + # The FTS table lives in another SQLite file. Its rows can go + # missing even when the metadata/state transaction was committed + # (e.g. restoring only one of the two databases). Repair from the + # stored segment text, without rereading every archive on startup. + await asyncio.to_thread(self._repair_segment_fts_sync) + existing = await asyncio.to_thread(self._get_segment_state_sync) + paths = await asyncio.to_thread(lambda: sorted(archive_dir.glob("*.json"))[:10000]) + count = 0 + bytes_read = 0 + for path in paths: + if path.stem in existing: + continue + try: + size = await asyncio.to_thread(lambda: path.stat().st_size) + if bytes_read + size > 2_000_000_000: + break + bytes_read += size + data = await asyncio.to_thread(self._read_archive_sync, path) + if await self._index_segments(data, path.stem, embedder): + count += 1 + except Exception as e: + log.warning("Skipping unreadable archive %s: %s", path, e) + await asyncio.sleep(0) + return count + + def _get_segment_state_sync(self) -> set[str]: + return {row[0] for row in self._conn.execute( # type: ignore[union-attr] + "SELECT archive_doc_id FROM segment_index_state")} + + def _repair_segment_fts_sync(self) -> None: + if not self._fts: + return + rows = self._conn.execute( # type: ignore[union-attr] + "SELECT doc_id, content, channel_id, last_active " + "FROM session_archives WHERE doc_id LIKE 'seg:%'", + ).fetchall() + for doc_id, content, channel_id, end_ts in rows: + if not self._fts.has_session(doc_id) and not self._fts.index_session( + doc_id, content, channel_id, end_ts): + raise RuntimeError(f"FTS repair did not acknowledge {doc_id}") + def _get_indexed_ids_sync(self) -> set[str]: """Get set of already-indexed doc IDs (sync).""" # Caller (backfill) checks self.available first. diff --git a/src/sessions/manager.py b/src/sessions/manager.py index 028b408d0..6491b98ee 100644 --- a/src/sessions/manager.py +++ b/src/sessions/manager.py @@ -21,6 +21,11 @@ from ..relevance import rank as relevance_rank from ..relevance import score as relevance_score from ..search.errors import validate_search_query +from ..search.vectorstore import summary_segment_doc_id + + +def _segment_visible_to(seg: dict, user_id: str | None) -> bool: + return not user_id or user_id in seg.get("participants", []) if TYPE_CHECKING: from ..learning.reflector import ConversationReflector @@ -1377,10 +1382,12 @@ def _search_archives( user_id: str | None = None, after: float | None = None, before: float | None = None, + seen_segments: set[str] | None = None, ) -> list[dict]: """Search archived session files for keyword matches (sync, for use in thread).""" results: list[dict] = [] archive_dir = self.persist_dir / "archive" + seen_segments = seen_segments if seen_segments is not None else set() if not archive_dir.exists(): return results for path in sorted(archive_dir.glob("*.json"), reverse=True): @@ -1393,7 +1400,7 @@ def _search_archives( # it must not resurface via search, mirroring restore. reset_epoch = self._reset_epochs.get(arch_cid, 0.0) summary = data.get("summary", "") - if summary and query_lower in summary.lower(): + if not user_id and summary and query_lower in summary.lower(): ts = data.get("last_active", 0) if (after and ts < after) or (before and ts > before) or ts <= reset_epoch: pass @@ -1404,6 +1411,22 @@ def _search_archives( "timestamp": ts, "channel_id": arch_cid, }) + for seg in data.get("summary_segments", []): + if not _segment_visible_to(seg, user_id): + continue + text = seg.get("summary", "") + if not text or query_lower not in text.lower(): + continue + seg_id = summary_segment_doc_id(arch_cid, seg) + ts = seg.get("end_ts", 0) + if (seg_id in seen_segments or (after and ts < after) or + (before and ts > before) or ts <= reset_epoch): + continue + seen_segments.add(seg_id) + results.append({"type": "summary", "content": text[:500], + "timestamp": ts, "channel_id": arch_cid}) + if len(results) >= limit: + return results for msg in reversed(data.get("messages", [])): if user_id and msg.get("user_id") != user_id: continue @@ -1449,6 +1472,7 @@ async def search_history( validate_search_query(query) query_lower = query.lower() results: list[dict] = [] + seen_segments: set[str] = set() def _ts_ok(ts: float) -> bool: if after and ts < after: @@ -1464,7 +1488,7 @@ def _ts_ok(ts: float) -> bool: sessions_iter = [s] if s else [] for session in sessions_iter: - if session.summary and query_lower in session.summary.lower(): + if not user_id and session.summary and query_lower in session.summary.lower(): if _ts_ok(session.last_active): results.append({ "type": "summary", @@ -1473,10 +1497,14 @@ def _ts_ok(ts: float) -> bool: "channel_id": session.channel_id, }) for seg in session.summary_segments: + if not _segment_visible_to(seg, user_id): + continue seg_text = seg.get("summary", "") if seg_text and query_lower in seg_text.lower(): ts = seg.get("end_ts", session.last_active) - if _ts_ok(ts): + seg_id = summary_segment_doc_id(session.channel_id, seg) + if _ts_ok(ts) and seg_id not in seen_segments: + seen_segments.add(seg_id) results.append({ "type": "summary", "content": seg_text[:500], @@ -1502,7 +1530,7 @@ def _ts_ok(ts: float) -> bool: # Step 2: keyword search on archives (most recent first) archive_results = await asyncio.to_thread( self._search_archives, query_lower, limit - len(results), - channel_id, user_id, after, before, + channel_id, user_id, after, before, seen_segments, ) results.extend(archive_results) if len(results) >= limit: @@ -1515,14 +1543,27 @@ async def _append_channel_matches() -> None: remaining = limit - len(results) fts = self._fts_index channel_results = [] - if fts and hasattr(fts, "search_channel_logs"): - channel_results = fts.search_channel_logs( - query, limit=remaining, channel_id=channel_id, - ) - if not channel_results and hasattr(self._channel_logger, "search"): + if not user_id and fts and hasattr(fts, "search_channel_logs"): channel_results = await asyncio.to_thread( - self._channel_logger.search, query, remaining, + fts.search_channel_logs, query, limit=remaining, + channel_id=channel_id, ) + if not channel_results and hasattr(self._channel_logger, "search"): + if user_id: + channel_results = await asyncio.to_thread( + self._channel_logger.search, query, remaining, channel_id, + author_id=user_id, + accept=lambda record: ( + _ts_ok(record.get("ts", 0)) and + record.get("ts", 0) > self._reset_epochs.get( + record.get("channel_id", ""), 0.0)), + ) + else: + # Preserve compatibility with legacy search(query, limit) + # implementations; channel scope is filtered below. + channel_results = await asyncio.to_thread( + self._channel_logger.search, query, remaining, + ) seen = {(r.get("channel_id", ""), r.get("timestamp", 0)) for r in results} for cr in channel_results: ts = cr.get("timestamp", 0) @@ -1549,12 +1590,14 @@ async def _append_channel_matches() -> None: # Step 4: hybrid search (FTS5 + semantic) fills remaining slots. # Backend failures propagate to the caller rather than masquerading as # a keyword-only result set. - if len(results) < limit and self._vector_store: + if not user_id and len(results) < limit and self._vector_store: hybrid_results = await self._vector_store.search_hybrid( query, self._embedder, limit=limit, ) seen = {(r["channel_id"], r.get("timestamp", 0)) for r in results} for hr in hybrid_results: + if hr.get("doc_id") in seen_segments: + continue if channel_id and hr.get("channel_id") != channel_id: continue ts = hr.get("timestamp", 0) diff --git a/src/tools/autonomous_loop.py b/src/tools/autonomous_loop.py index a0b2e49c0..15a54e675 100644 --- a/src/tools/autonomous_loop.py +++ b/src/tools/autonomous_loop.py @@ -565,7 +565,9 @@ async def _post_response( try: text = response if len(text) > 2000: - text = text[:1950] + "\n... (truncated)" + from ..discord.delivery import close_open_fence + + text = close_open_fence(text[:1950]) + "\n... (truncated)" await channel.send(text) except Exception as e: log.warning("Loop %s: failed to post response: %s", info.id, e) diff --git a/src/tools/defs/browser_web.py b/src/tools/defs/browser_web.py index 903860410..8081ab3a0 100644 --- a/src/tools/defs/browser_web.py +++ b/src/tools/defs/browser_web.py @@ -100,9 +100,10 @@ { "name": "browser_click", "description": ( - "Navigates to a URL and clicks an element by CSS selector. Returns a confirmation " - "summary after clicking. To fill forms, use browser_fill. To read page content after " - "clicking, follow up with browser_read_page." + "Loads a URL in a fresh browser session, clicks an element by CSS selector, and " + "returns the resulting page's title and URL. Cookies, storage and page state end " + "with the call; a later browser_read_page reloads the URL without them. To fill and " + "submit in one call, use browser_fill with submit=true." ), "input_schema": { "type": "object", @@ -126,8 +127,10 @@ { "name": "browser_fill", "description": ( - "Navigates to a URL and fills a form field by CSS selector. Optionally submits by " - "pressing Enter. To click buttons, use browser_click." + "Loads a URL in a fresh browser session, fills one field by CSS selector, optionally " + "presses Enter (submit=true), and returns the page's title and URL. The value is gone " + "when the call ends, so a later browser_click cannot submit it; use submit=true, or " + "one browser_evaluate for several fields." ), "input_schema": { "type": "object", @@ -157,8 +160,11 @@ { "name": "browser_evaluate", "description": ( - "Evaluates JavaScript on a URL and returns the result. For custom scraping or " - "interaction. Large results have retained previews; use get_tool_output(cursor=...) " + "Loads a URL in a fresh browser session, evaluates a JavaScript expression, and " + "returns its result (a returned Promise is awaited). Use it for scraping or several " + "interaction " + "steps in one call; the session closes when it returns, so a navigation it starts may " + "not complete. Large results have retained previews; use get_tool_output(cursor=...) " "without re-running the expression." ), "input_schema": { diff --git a/src/tools/email_client.py b/src/tools/email_client.py index c82b907b0..db1469767 100644 --- a/src/tools/email_client.py +++ b/src/tools/email_client.py @@ -49,11 +49,32 @@ def _decode_header_value(raw: str | None) -> str: return " ".join(decoded) +def _is_attachment(part: email_lib.message.Message) -> bool: + return part.get_content_disposition() == "attachment" + + +def _body_parts(msg: email_lib.message.Message): + """Yield body leaves, never text nested within an attached message.""" + + def walk(part: email_lib.message.Message, inside_attachment: bool): + inside_attachment |= _is_attachment(part) + if part.is_multipart(): + children = part.get_payload() + if isinstance(children, list): + for child in children: + if isinstance(child, email_lib.message.Message): + yield from walk(child, inside_attachment) + elif not inside_attachment: + yield part + + yield from walk(msg, False) + + def _extract_body(msg: email_lib.message.Message, max_chars: int) -> str: if msg.is_multipart(): - for part in msg.walk(): + for part in _body_parts(msg): ct = part.get_content_type() - if ct == "text/plain" and part.get("Content-Disposition") != "attachment": + if ct == "text/plain": payload = part.get_payload(decode=True) if payload: charset = part.get_content_charset() or "utf-8" @@ -64,9 +85,9 @@ def _extract_body(msg: email_lib.message.Message, max_chars: int) -> str: + f"\n\n[truncated at {max_chars} chars, original {len(text)}]" ) return text - for part in msg.walk(): + for part in _body_parts(msg): ct = part.get_content_type() - if ct == "text/html" and part.get("Content-Disposition") != "attachment": + if ct == "text/html": payload = part.get_payload(decode=True) if payload: charset = part.get_content_charset() or "utf-8" @@ -91,8 +112,7 @@ def _extract_body(msg: email_lib.message.Message, max_chars: int) -> str: def _attachment_metadata(msg: email_lib.message.Message) -> list[dict]: attachments = [] for part in msg.walk(): - disp = part.get("Content-Disposition", "") - if "attachment" in disp: + if _is_attachment(part): filename = part.get_filename() or "(unnamed)" filename = _decode_header_value(filename) size = len(part.get_payload(decode=True) or b"") @@ -114,9 +134,7 @@ def _message_summary(msg: email_lib.message.Message, uid: str = "") -> dict: "subject": _decode_header_value(msg.get("Subject")), "date": msg.get("Date", ""), "message_id": msg.get("Message-ID", ""), - "has_attachments": any( - "attachment" in (p.get("Content-Disposition") or "") for p in msg.walk() - ), + "has_attachments": any(_is_attachment(p) for p in msg.walk()), } @@ -194,24 +212,74 @@ def send_email( with smtplib.SMTP(smtp_host, smtp_port, timeout=timeout) as server: server.starttls() server.login(username, password) - server.sendmail(from_address, all_recipients, msg.as_string()) + refused_map = server.sendmail(from_address, all_recipients, msg.as_string()) + except smtplib.SMTPRecipientsRefused as e: + refusals = _refusals(e.recipients, to, cc, bcc, password) + raise RuntimeError( + f"SMTP send failed: no message was sent. Refused: {_format_refusals(refusals)}" + ) from None except Exception as e: raise RuntimeError(f"SMTP send failed: {_safe_error(e, password)}") from None + refusals = _refusals(refused_map, to, cc, bcc, password) + accepted = [address for address in dict.fromkeys(all_recipients) if address not in refused_map] log.info( - "Email sent to %s, subject=%r, message_id=%s", all_recipients, subject, msg["Message-ID"] + "Email accepted by SMTP for %s, refused for %s, subject=%r, message_id=%s", + accepted, + [entry["address"] for entry in refusals], + subject, + msg["Message-ID"], ) return { - "status": "sent", + "status": "partial" if refusals else "sent", "message_id": msg["Message-ID"], "to": to, "cc": cc or [], "subject": subject, "attachments": [Path(p).name for p in (attachments or [])], + "accepted_count": len(accepted), + "refused": refusals, } +def _refusals( + refused_map: dict, + to: list[str], + cc: list[str] | None, + bcc: list[str] | None, + password: str | None, +) -> list[dict]: + """Label and sanitize SMTP refusals, including addresses hidden in BCC.""" + result: list[dict] = [] + seen: set[str] = set() + for field, addresses in (("To", to), ("CC", cc or []), ("BCC", bcc or [])): + for address in addresses: + if address in seen or address not in refused_map: + continue + seen.add(address) + code, response = refused_map[address] + text = ( + response.decode("utf-8", "replace") + if isinstance(response, bytes) + else str(response) + ) + reason = " ".join(text.split()) + if password and password in reason: + reason = reason.replace(password, "[REDACTED]") + result.append( + {"address": address, "field": field, "code": int(code), "reason": reason[:200]} + ) + return result + + +def _format_refusals(refusals: list[dict]) -> str: + return "; ".join( + f"{entry['address']} ({entry['field']}): {entry['code']} {entry['reason']}".rstrip() + for entry in refusals + ) + + def search_email( *, imap_host: str, diff --git a/src/tools/handlers/comms.py b/src/tools/handlers/comms.py index c3770be42..e1c57b489 100644 --- a/src/tools/handlers/comms.py +++ b/src/tools/handlers/comms.py @@ -54,6 +54,33 @@ async def _handle_email_send(self, inp: dict) -> str: max_attachment_bytes=cfg.max_attachment_bytes, timeout=cfg.connect_timeout_seconds, ) + refused = result.get("refused") or [] + if refused: + # Accepted recipients already have the message; do not mark + # this as an error and invite an unsafe automatic retry. + from ..email_client import _format_refusals + + rejected = {entry["address"] for entry in refused} + accepted_to = [address for address in result["to"] if address not in rejected] + accepted_cc = [ + address for address in result.get("cc") or [] if address not in rejected + ] + partial = [ + "Email partially sent: the SMTP server accepted it for " + f"{result.get('accepted_count', 0)} recipient(s) and refused " + f"{len(refused)}. Not re-sent: a retry would duplicate it for the " + "accepted recipients.", + f"Message-ID: {result['message_id']}", + ] + if accepted_to: + partial.append(f"To: {', '.join(accepted_to)}") + partial.append(f"Subject: {result['subject']}") + if accepted_cc: + partial.append(f"CC: {', '.join(accepted_cc)}") + if result.get("attachments"): + partial.append(f"Attachments: {', '.join(result['attachments'])}") + partial.append(f"Refused: {_format_refusals(refused)}") + return "\n".join(partial) parts = [ "Email sent successfully.", f"Message-ID: {result['message_id']}", diff --git a/src/tools/handlers/state.py b/src/tools/handlers/state.py index b889dd967..2b47b12aa 100644 --- a/src/tools/handlers/state.py +++ b/src/tools/handlers/state.py @@ -11,10 +11,10 @@ from __future__ import annotations -import asyncio import json from pathlib import Path +from ...async_utils import to_thread_settled from .deps import HandlerBase, HandlerDeps # Max working-memory notes retained per section (global / per-user). The full @@ -58,7 +58,7 @@ async def _handle_memory_manage(self, inp: dict, *, user_id: str | None = None) if not key: return "'key' is required for get." try: - all_mem = await asyncio.to_thread(self._load_all_memory) + all_mem = await to_thread_settled(self._load_all_memory) except StoreCorruptError as exc: return ( "Memory store is currently unreadable or corrupt (a backup copy was " @@ -73,7 +73,7 @@ async def _handle_memory_manage(self, inp: dict, *, user_id: str | None = None) if action == "list": try: - all_mem = await asyncio.to_thread(self._load_all_memory) + all_mem = await to_thread_settled(self._load_all_memory) except StoreCorruptError as exc: return ( "Memory store is currently unreadable or corrupt (a backup copy was " @@ -96,7 +96,7 @@ async def _handle_memory_manage(self, inp: dict, *, user_id: str | None = None) if not key or not value: return "Both 'key' and 'value' are required for save." try: - all_mem = await asyncio.to_thread(self._load_all_memory) + all_mem = await to_thread_settled(self._load_all_memory) except StoreCorruptError as exc: return ( "Memory store is currently unreadable or corrupt (a backup copy was " @@ -122,7 +122,7 @@ async def _handle_memory_manage(self, inp: dict, *, user_id: str | None = None) break del section_map[oldest] evicted += 1 - await asyncio.to_thread(self._save_all_memory, all_mem) + await to_thread_settled(self._save_all_memory, all_mem) scope_label = "global" if section == "global" else "personal" suffix = ( f" (evicted {evicted} oldest note(s) at cap {MEMORY_MAX_KEYS_PER_SECTION})" @@ -136,7 +136,7 @@ async def _handle_memory_manage(self, inp: dict, *, user_id: str | None = None) if not key: return "'key' is required for delete." try: - all_mem = await asyncio.to_thread(self._load_all_memory) + all_mem = await to_thread_settled(self._load_all_memory) except StoreCorruptError as exc: return ( "Memory store is currently unreadable or corrupt (a backup copy was " @@ -145,11 +145,11 @@ async def _handle_memory_manage(self, inp: dict, *, user_id: str | None = None) user_key = f"user_{user_id}" if user_id else None if user_key and key in all_mem.get(user_key, {}): del all_mem[user_key][key] - await asyncio.to_thread(self._save_all_memory, all_mem) + await to_thread_settled(self._save_all_memory, all_mem) return f"Deleted personal note '{key}'." elif key in all_mem.get("global", {}): del all_mem["global"][key] - await asyncio.to_thread(self._save_all_memory, all_mem) + await to_thread_settled(self._save_all_memory, all_mem) return f"Deleted global note '{key}'." return f"No note found with key '{key}'." @@ -242,7 +242,7 @@ async def _manage_list_locked(self, action, list_name, raw_items, owner_pref, us from ...json_store import StoreCorruptError try: - lists = await asyncio.to_thread(self._load_lists_for_write) + lists = await to_thread_settled(self._load_lists_for_write) except StoreCorruptError as exc: # Reads degrade to an empty view; mutations refuse — never # overwrite a corrupt file, which would wipe the lists. @@ -290,7 +290,7 @@ async def _manage_list_locked(self, action, list_name, raw_items, owner_pref, us return f"The '{list_name}' list is already empty." count = len(lst["items"]) lst["items"] = [] - await asyncio.to_thread(self._save_lists, lists) + await to_thread_settled(self._save_lists, lists) return f"Cleared {count} item(s) from the '{list_name}' list." if action == "add": @@ -318,7 +318,7 @@ async def _manage_list_locked(self, action, list_name, raw_items, owner_pref, us } ) added.append(name) - await asyncio.to_thread(self._save_lists, lists) + await to_thread_settled(self._save_lists, lists) parts = [] if added: parts.append(f"Added to '{list_name}': {', '.join(added)}") @@ -346,7 +346,7 @@ async def _manage_list_locked(self, action, list_name, raw_items, owner_pref, us removed.append(lst["items"].pop(idx)["name"]) else: not_found.append(name) - await asyncio.to_thread(self._save_lists, lists) + await to_thread_settled(self._save_lists, lists) parts = [] if removed: parts.append(f"Removed from '{list_name}': {', '.join(removed)}") @@ -377,7 +377,7 @@ async def _manage_list_locked(self, action, list_name, raw_items, owner_pref, us break if not found: not_found.append(name.strip()) - await asyncio.to_thread(self._save_lists, lists) + await to_thread_settled(self._save_lists, lists) parts = [] if marked: parts.append(f"Marked done: {', '.join(marked)}") @@ -405,7 +405,7 @@ async def _manage_list_locked(self, action, list_name, raw_items, owner_pref, us break if not found: not_found.append(name.strip()) - await asyncio.to_thread(self._save_lists, lists) + await to_thread_settled(self._save_lists, lists) parts = [] if marked: parts.append(f"Marked undone: {', '.join(marked)}") diff --git a/src/tools/http_probe_ops.py b/src/tools/http_probe_ops.py index 4efc28a56..6f0cda4ed 100644 --- a/src/tools/http_probe_ops.py +++ b/src/tools/http_probe_ops.py @@ -172,6 +172,11 @@ def build_http_probe_command(params: dict) -> str: headers = params.get("headers") if isinstance(headers, dict): for name, value in headers.items(): + # curl interprets -H @file as a request to read local file contents. + if str(name).startswith("@"): + raise ValueError( + f"Invalid header name {str(name)!r}: header names cannot start with '@'" + ) header_str = f"{name}: {value}" parts.append(f"-H {_sq(header_str)}") @@ -191,7 +196,9 @@ def build_http_probe_command(params: dict) -> str: raise ValueError( f"Request body is {body_bytes} bytes, over the {MAX_BODY_SIZE}-byte limit" ) - parts.append(f"-d {_sq(body)}") + # -d @path reads a local file; --data-raw sends the literal bytes. + flag = "--data-raw" if body.startswith("@") else "-d" + parts.append(f"{flag} {_sq(body)}") # URL (always last) parts.append(_sq(url)) diff --git a/src/tools/post_validation.py b/src/tools/post_validation.py index 0b0290254..07e4c2523 100644 --- a/src/tools/post_validation.py +++ b/src/tools/post_validation.py @@ -261,6 +261,38 @@ def _default_compare_for(check_type: str) -> str: }.get(check_type, "equals") +# Filter the probe's own command line, not ancestors of the probe: a real +# target can itself be an ancestor. Concurrent process probes carry this same +# marker and are filtered as well. procps and BSD/macOS both print command +# lines with -a -l -f; the inner POSIX shell works under other login shells. +_PROCESS_PROBE_PGREP = "pgrep -a -l -f --" +_PROCESS_PROBE_SCRIPT = ( + f'o=$({_PROCESS_PROBE_PGREP} "$1"); r=$?; ' + 'if [ "$r" -gt 1 ]; then echo "PROCESS_CHECK_ERROR pgrep exit $r"; exit 0; fi; ' + f'm=$(printf "%s\\n" "$o" | grep -v -F -e "{_PROCESS_PROBE_PGREP} "); ' + 'if [ -n "$m" ]; then echo PRESENT; else echo ABSENT; fi' +) + + +# Validate the ERE first, then check journal visibility without -q; the data +# run uses -q to keep journalctl's own status lines out of the matches. +_LOG_PROBE_SCRIPT = ( + 'p="$3"; ' + 'printf "" | grep -E -e "$p" >/dev/null 2>&1; ' + 'if [ "$?" -gt 1 ]; then echo "LOG_CHECK_ERROR invalid pattern"; ' + 'printf "" | grep -E -e "$p" 2>&1 | head -n 2; exit 0; fi; ' + 'if [ -n "$1" ]; then set -- -u "$1" --since "$2 seconds ago"; ' + 'else set -- --since "$2 seconds ago"; fi; ' + 'e=$(journalctl "$@" --no-pager -n 1 2>&1 >/dev/null); r=$?; ' + 'if [ "$r" -ne 0 ]; then echo "LOG_CHECK_ERROR journalctl exit $r"; ' + 'printf "%s\\n" "$e" | tail -n 1; exit 0; fi; ' + 'journalctl "$@" --no-pager -q 2>/dev/null | grep -E -e "$p" | head -n 20; ' + 'case "$e" in *"not seeing messages from"*|*"insufficient permissions"*' + '|*"No journal files were found"*) echo LOG_READ_PARTIAL;; *) echo LOG_READ_OK;; esac' +) +_LOG_STATUS_LINES = frozenset({"LOG_READ_OK", "LOG_READ_PARTIAL"}) + + def _build_command(check: Check) -> str | None: """Build the shell command the check will execute on the target host. @@ -271,10 +303,10 @@ def _build_command(check: Check) -> str | None: timeout = check.timeout_seconds if t == "http": - # curl with redirect following; print status on stderr-safe line + # HTTP errors are received responses; only transport errors fail curl. url = shlex.quote(tgt) return ( - f"curl -fsS -o /dev/null -w '%{{http_code}}' " + f"curl -sS -o /dev/null -w '%{{http_code}}' " f"--max-time {timeout} -L {url} || echo FAILED_$?" ) if t == "port": @@ -298,36 +330,45 @@ def _build_command(check: Check) -> str | None: if t == "service": return f"systemctl is-active {shlex.quote(tgt)} 2>/dev/null || true" if t == "process": - return f"pgrep -f {shlex.quote(tgt)} >/dev/null && echo PRESENT || echo ABSENT" + return f"sh -c {shlex.quote(_PROCESS_PROBE_SCRIPT)} odin-process-check {shlex.quote(tgt)}" if t in ("log_absent", "log_present"): # target format: "unit=