From 25afd8639f7a1f18aad3a77ca6b0606e89258d40 Mon Sep 17 00:00:00 2001 From: Colton Ehrman Date: Wed, 9 Sep 2026 09:28:10 -0500 Subject: [PATCH] Improve voice and desktop reliability and add companion workflows Add recording, research, usage, local-agent, widget, and benchmark workflows with regression coverage and maintainer documentation. --- .github/workflows/ci.yml | 3 + AGENTS.md | 92 +++- CONTRIBUTING.md | 6 +- README.md | 10 + benchmarks/browser/golden.json | 58 +++ docs/BROWSER_BENCHMARK.md | 210 ++++++++ docs/CONTRIBUTION_REVIEW.md | 133 +++++ docs/COOL_TRICKS.md | 80 +++ docs/RECORDINGS.md | 64 +++ docs/RESEARCH.md | 116 +++++ docs/SELF_HEALING.md | 63 +++ docs/USAGE_AND_COSTS.md | 91 ++++ docs/VOICE_TURN_CONFIRMATION.md | 35 ++ electron.vite.config.ts | 37 +- package.json | 29 +- pnpm-lock.yaml | 56 +- pnpm-workspace.yaml | 3 + scripts/browser-benchmark.ts | 22 + scripts/browser-benchmark/core.ts | 1 + scripts/browser-benchmark/page.ts | 1 + scripts/build-window-control.mjs | 14 + scripts/diagnose.mjs | 40 ++ scripts/doctor.ts | 24 + scripts/latency-report.ts | 23 + scripts/preview-research.tsx | 24 + scripts/probe-confirmed-turns.cjs | 42 ++ scripts/probe-research-start.cjs | 53 ++ scripts/run-tests.mjs | 20 + scripts/smoke-realtime.ts | 35 +- scripts/test-archive-delegation.ts | 170 +++++++ scripts/test-browser-benchmark.ts | 314 ++++++++++++ scripts/test-browser-research-policy.ts | 50 ++ scripts/test-clock.ts | 13 + scripts/test-code-walkthrough.ts | 160 ++++++ scripts/test-codex-history-integration.ts | 54 ++ scripts/test-confirmed-speech.ts | 53 ++ scripts/test-continuous-mic.ts | 35 ++ scripts/test-desktop-bridge.ts | 183 +++++++ scripts/test-desktop-execution.ts | 142 ++++++ scripts/test-diagnostic-report.ts | 287 +++++++++++ scripts/test-diagnostic-runtime.ts | 49 ++ scripts/test-direct-controls.ts | 48 ++ scripts/test-echo-cancellation.ts | 21 + scripts/test-echo-reference-output.ts | 34 ++ scripts/test-email-overview.ts | 46 ++ scripts/test-empty-chat.ts | 40 ++ scripts/test-enhancement-permissions.ts | 111 ++++ scripts/test-git-skill.ts | 99 ++++ scripts/test-incidental-speech.ts | 27 + scripts/test-input-preview.ts | 69 +++ scripts/test-interaction-log.ts | 26 + scripts/test-latency-summary.ts | 63 +++ scripts/test-live-microphone.ts | 32 ++ scripts/test-local-agent-context.ts | 31 ++ scripts/test-local-agents.ts | 77 +++ scripts/test-model-wait.ts | 33 ++ scripts/test-new-session-command.ts | 18 + scripts/test-open-source.ts | 65 +++ scripts/test-open-url.ts | 62 +++ scripts/test-openai-realtime.ts | 85 ++++ scripts/test-permission-deferral.ts | 90 ++++ scripts/test-playback-acknowledgment.ts | 78 +++ scripts/test-playback-captions.ts | 32 ++ scripts/test-playback-speech-evidence.ts | 72 +++ scripts/test-playback-turn-policy.ts | 19 + scripts/test-quit-app.ts | 23 + scripts/test-realtime-feedback.ts | 160 ++++++ scripts/test-realtime-idle.ts | 103 ++++ scripts/test-recording-capture.ts | 34 ++ scripts/test-recording-store.ts | 67 +++ scripts/test-recovery.ts | 48 ++ scripts/test-research-reasoning.ts | 32 ++ scripts/test-research-start.ts | 38 ++ scripts/test-research.ts | 80 +++ scripts/test-response-coordinator.ts | 67 +++ scripts/test-screen-on-wake.ts | 74 +++ scripts/test-search-availability.ts | 38 ++ scripts/test-self-enhancement.ts | 88 ++++ scripts/test-session-control.ts | 67 +++ scripts/test-speech-gate.ts | 58 +++ scripts/test-spoken-turn-guard.ts | 79 +++ scripts/test-spoken-turn-host.ts | 181 +++++++ scripts/test-tricks.ts | 55 ++ scripts/test-usage-desktop.mjs | 85 ++++ scripts/test-usage.ts | 119 +++++ scripts/test-verified-activation.ts | 16 + scripts/test-wake-audio.ts | 29 ++ scripts/test-wake-match.ts | 39 ++ scripts/test-widget-magnets.ts | 57 +++ scripts/test-widgets-desktop.mjs | 205 ++++++++ src/main/agent/browser-research-policy.ts | 46 ++ src/main/agent/chat.ts | 133 ++++- src/main/agent/email-browser-context.ts | 13 + src/main/agent/email-overview.ts | 12 + src/main/agent/enhancement-context.ts | 7 + src/main/agent/latency.ts | 18 + src/main/agent/local-agent-context.ts | 18 + src/main/agent/model-wait.ts | 15 + src/main/agent/permissions.ts | 29 +- src/main/agent/realtime/confirmed-speech.ts | 56 ++ .../agent/realtime/confirmed-turn-config.ts | 15 + src/main/agent/realtime/connection.ts | 27 + .../agent/realtime/desktop-delegations.ts | 31 ++ src/main/agent/realtime/desktop-status.ts | 17 + src/main/agent/realtime/gateway-token.ts | 8 +- src/main/agent/realtime/incidental-speech.ts | 16 + src/main/agent/realtime/notice-buffer.ts | 36 ++ src/main/agent/realtime/openai-codec.ts | 84 +++ .../realtime/playback-speech-evidence.ts | 63 +++ .../agent/realtime/playback-turn-policy.ts | 10 + src/main/agent/realtime/realtime-tools.ts | 18 +- .../agent/realtime/response-coordinator.ts | 72 +++ src/main/agent/realtime/session-host.ts | 478 ++++++++++++++++-- src/main/agent/realtime/spoken-turn-guard.ts | 65 +++ src/main/agent/realtime/turn-grounding.ts | 5 + src/main/agent/research-budget.ts | 6 + src/main/agent/research-reasoning.ts | 30 ++ src/main/agent/research-start.ts | 61 +++ src/main/agent/screen-on-wake.ts | 102 ++++ src/main/agent/system-prompt.ts | 37 +- src/main/benchmarks/core.ts | 118 +++++ src/main/benchmarks/host.ts | 125 +++++ src/main/benchmarks/page.ts | 38 ++ src/main/benchmarks/server.ts | 88 ++++ src/main/config/realtime-models.ts | 10 + src/main/config/schema.ts | 14 +- src/main/config/screen-health.ts | 36 ++ src/main/config/voice-commands.ts | 31 ++ src/main/demos/playground-window.ts | 35 ++ src/main/diagnostics/interaction-log.ts | 44 ++ src/main/diagnostics/latency-summary.ts | 101 ++++ src/main/diagnostics/report.ts | 211 ++++++++ src/main/diagnostics/runtime.ts | 59 +++ src/main/diagnostics/tool-outcome.ts | 19 + src/main/index.ts | 312 ++++++++++-- src/main/ipc/channels.ts | 70 ++- src/main/maintenance/capabilities.ts | 16 + src/main/maintenance/codex-client.ts | 222 ++++++++ src/main/maintenance/desktop-actions.ts | 146 ++++++ src/main/maintenance/desktop-ipc.ts | 167 ++++++ src/main/maintenance/desktop-projects.ts | 33 ++ src/main/maintenance/desktop-tasks.ts | 116 +++++ src/main/maintenance/self-enhancement.ts | 49 ++ src/main/maintenance/session-control.ts | 34 ++ src/main/recordings/host.ts | 199 ++++++++ src/main/recordings/store.ts | 94 ++++ src/main/recordings/types.ts | 28 + src/main/screen-health.ts | 61 +++ src/main/stt/index.ts | 8 +- src/main/stt/openai.ts | 33 +- src/main/tray-icon.ts | 31 ++ src/main/tts/elevenlabs.ts | 40 +- src/main/usage/catalog.json | 313 ++++++++++++ src/main/usage/ledger.ts | 105 ++++ src/main/usage/pricing.ts | 81 +++ src/main/usage/types.ts | 42 ++ src/main/widgets/magnet.ts | 70 +++ src/main/widgets/windows.ts | 288 +++++++++++ src/native/README.md | 13 + src/native/window-control.m | 279 ++++++++++ src/preload/index.ts | 76 ++- src/preload/opendex.d.ts | 3 + src/preload/widget.ts | 28 + src/renderer/index.html | 2 +- src/renderer/src/App.tsx | 83 +-- src/renderer/src/NotchApp.tsx | 48 +- src/renderer/src/components/compact-bar.tsx | 218 +++++++- .../src/components/floating-widget.tsx | 81 +++ .../src/components/input-transcript.tsx | 21 + .../src/components/microphone-feedback.tsx | 71 +++ .../onboarding/onboarding-wizard.tsx | 37 +- .../components/playground/playground-app.tsx | 114 +++++ .../src/components/screen-access-panel.tsx | 70 +++ .../settings/recordings-section.tsx | 89 ++++ .../src/components/settings/sections.tsx | 49 +- .../src/components/settings/settings-view.tsx | 29 +- .../src/components/settings/usage-section.tsx | 92 ++++ .../src/components/spending-meter.tsx | 12 + src/renderer/src/components/status-bar.tsx | 25 +- src/renderer/src/components/task-progress.tsx | 20 + .../src/components/themes/cursor/index.tsx | 62 +-- .../src/components/themes/editorial/index.tsx | 29 +- .../src/components/themes/jarvis/index.tsx | 1 + .../themes/shared/minimal-shell.tsx | 10 +- .../themes/shared/theme-top-bar.tsx | 32 +- .../src/components/themes/siri/index.tsx | 4 +- .../src/components/tool-activity-banner.tsx | 7 +- src/renderer/src/components/ui/button.tsx | 2 +- .../src/components/widget-slot-picker.tsx | 54 ++ src/renderer/src/lib/dex/audio-meter.ts | 3 +- src/renderer/src/lib/dex/cancelled-tools.ts | 8 + src/renderer/src/lib/dex/engines/types.ts | 3 +- src/renderer/src/lib/dex/engines/vosk-wake.ts | 48 +- .../src/lib/dex/engines/wake-audio.ts | 43 ++ .../src/lib/dex/engines/wake-match.ts | 32 ++ src/renderer/src/lib/dex/live-microphone.ts | 16 + .../src/lib/dex/realtime/echo-cancellation.ts | 28 + .../lib/dex/realtime/echo-reference-output.ts | 74 +++ .../src/lib/dex/realtime/input-preview.ts | 57 +++ .../src/lib/dex/realtime/pcm-capture.ts | 140 +++-- .../src/lib/dex/realtime/pcm-player.ts | 6 +- .../src/lib/dex/realtime/playback-captions.ts | 12 + .../src/lib/dex/realtime/realtime-session.ts | 240 ++++++++- .../src/lib/dex/realtime/speech-gate.ts | 43 ++ .../src/lib/dex/realtime/voice-error.ts | 19 + src/renderer/src/lib/dex/research-progress.ts | 27 + .../src/lib/dex/sleep-announcement.ts | 25 + src/renderer/src/lib/dex/sleep-command.ts | 1 + src/renderer/src/lib/dex/state.ts | 25 +- src/renderer/src/lib/dex/use-dex.ts | 354 +++++++++---- src/renderer/src/lib/dex/wake-cue.ts | 30 ++ src/renderer/src/lib/format-tool-call.ts | 2 + .../src/lib/recordings/acquire-capture.ts | 22 + .../src/lib/recordings/preferences.ts | 14 + src/renderer/src/lib/recordings/recorder.ts | 121 +++++ .../src/lib/recordings/use-recording-state.ts | 13 + src/renderer/src/lib/task-progress.ts | 21 + src/renderer/src/lib/use-screen-health.ts | 14 + src/renderer/src/lib/use-usage.ts | 30 ++ src/renderer/src/main.tsx | 11 +- src/renderer/src/styles/globals.css | 2 + src/renderer/src/types/vosk-browser.d.ts | 2 +- src/skills/availability.ts | 13 + src/skills/benchmark/meta.ts | 6 + src/skills/benchmark/skill.ts | 43 ++ src/skills/capabilities/meta.ts | 6 + src/skills/capabilities/skill.ts | 18 + src/skills/clock/skill.ts | 7 +- src/skills/computer/control-targets.ts | 12 + src/skills/computer/execution.ts | 27 + src/skills/computer/meta.ts | 4 + src/skills/computer/screen-capture.ts | 38 +- src/skills/computer/skill.ts | 244 ++++++--- src/skills/computer/view.tsx | 28 + src/skills/computer/window-control.ts | 21 + src/skills/diagnostics/meta.ts | 7 + src/skills/diagnostics/skill.ts | 17 + src/skills/git/meta.ts | 5 + src/skills/git/repository.ts | 101 ++++ src/skills/git/skill.ts | 27 + .../local-agent-actions/archive-fallback.ts | 30 ++ src/skills/local-agent-actions/meta.ts | 7 + src/skills/local-agent-actions/skill.ts | 39 ++ src/skills/local-agents/meta.ts | 7 + src/skills/local-agents/skill.ts | 26 + src/skills/open/code-walkthrough.ts | 137 +++++ src/skills/open/meta.ts | 5 +- src/skills/open/open-source.ts | 59 +++ src/skills/open/open-url.ts | 42 ++ src/skills/open/quit-app.ts | 20 + src/skills/open/skill.ts | 146 ++++-- src/skills/open/verified-activation.ts | 15 + src/skills/open/view.tsx | 4 + src/skills/open/walkthrough-document.ts | 33 ++ src/skills/open/walkthrough-editor.ts | 26 + src/skills/open/walkthrough-window.ts | 66 +++ src/skills/realtime-selection.ts | 11 + src/skills/recording/meta.ts | 6 + src/skills/recording/skill.ts | 16 + src/skills/registry.ts | 56 +- src/skills/research/meta.ts | 7 + src/skills/research/schema.ts | 33 ++ src/skills/research/skill.ts | 27 + src/skills/research/view.tsx | 56 ++ src/skills/self-enhancement/meta.ts | 6 + src/skills/self-enhancement/skill.ts | 35 ++ src/skills/tool-card-layer.tsx | 13 +- src/skills/tool-registry.ts | 4 +- src/skills/tool-view.ts | 2 + src/skills/tricks/catalog.ts | 108 ++++ src/skills/tricks/meta.ts | 8 + src/skills/tricks/skill.ts | 48 ++ src/skills/tricks/view.tsx | 6 + src/skills/types.ts | 18 +- src/skills/web-search/skill.ts | 74 +-- 275 files changed, 15221 insertions(+), 776 deletions(-) create mode 100644 benchmarks/browser/golden.json create mode 100644 docs/BROWSER_BENCHMARK.md create mode 100644 docs/CONTRIBUTION_REVIEW.md create mode 100644 docs/COOL_TRICKS.md create mode 100644 docs/RECORDINGS.md create mode 100644 docs/RESEARCH.md create mode 100644 docs/SELF_HEALING.md create mode 100644 docs/USAGE_AND_COSTS.md create mode 100644 docs/VOICE_TURN_CONFIRMATION.md create mode 100644 scripts/browser-benchmark.ts create mode 100644 scripts/browser-benchmark/core.ts create mode 100644 scripts/browser-benchmark/page.ts create mode 100644 scripts/build-window-control.mjs create mode 100644 scripts/diagnose.mjs create mode 100644 scripts/doctor.ts create mode 100644 scripts/latency-report.ts create mode 100644 scripts/preview-research.tsx create mode 100644 scripts/probe-confirmed-turns.cjs create mode 100644 scripts/probe-research-start.cjs create mode 100644 scripts/run-tests.mjs create mode 100644 scripts/test-archive-delegation.ts create mode 100644 scripts/test-browser-benchmark.ts create mode 100644 scripts/test-browser-research-policy.ts create mode 100644 scripts/test-clock.ts create mode 100644 scripts/test-code-walkthrough.ts create mode 100644 scripts/test-codex-history-integration.ts create mode 100644 scripts/test-confirmed-speech.ts create mode 100644 scripts/test-continuous-mic.ts create mode 100644 scripts/test-desktop-bridge.ts create mode 100644 scripts/test-desktop-execution.ts create mode 100644 scripts/test-diagnostic-report.ts create mode 100644 scripts/test-diagnostic-runtime.ts create mode 100644 scripts/test-direct-controls.ts create mode 100644 scripts/test-echo-cancellation.ts create mode 100644 scripts/test-echo-reference-output.ts create mode 100644 scripts/test-email-overview.ts create mode 100644 scripts/test-empty-chat.ts create mode 100644 scripts/test-enhancement-permissions.ts create mode 100644 scripts/test-git-skill.ts create mode 100644 scripts/test-incidental-speech.ts create mode 100644 scripts/test-input-preview.ts create mode 100644 scripts/test-interaction-log.ts create mode 100644 scripts/test-latency-summary.ts create mode 100644 scripts/test-live-microphone.ts create mode 100644 scripts/test-local-agent-context.ts create mode 100644 scripts/test-local-agents.ts create mode 100644 scripts/test-model-wait.ts create mode 100644 scripts/test-new-session-command.ts create mode 100644 scripts/test-open-source.ts create mode 100644 scripts/test-open-url.ts create mode 100644 scripts/test-openai-realtime.ts create mode 100644 scripts/test-permission-deferral.ts create mode 100644 scripts/test-playback-acknowledgment.ts create mode 100644 scripts/test-playback-captions.ts create mode 100644 scripts/test-playback-speech-evidence.ts create mode 100644 scripts/test-playback-turn-policy.ts create mode 100644 scripts/test-quit-app.ts create mode 100644 scripts/test-realtime-feedback.ts create mode 100644 scripts/test-realtime-idle.ts create mode 100644 scripts/test-recording-capture.ts create mode 100644 scripts/test-recording-store.ts create mode 100644 scripts/test-recovery.ts create mode 100644 scripts/test-research-reasoning.ts create mode 100644 scripts/test-research-start.ts create mode 100644 scripts/test-research.ts create mode 100644 scripts/test-response-coordinator.ts create mode 100644 scripts/test-screen-on-wake.ts create mode 100644 scripts/test-search-availability.ts create mode 100644 scripts/test-self-enhancement.ts create mode 100644 scripts/test-session-control.ts create mode 100644 scripts/test-speech-gate.ts create mode 100644 scripts/test-spoken-turn-guard.ts create mode 100644 scripts/test-spoken-turn-host.ts create mode 100644 scripts/test-tricks.ts create mode 100644 scripts/test-usage-desktop.mjs create mode 100644 scripts/test-usage.ts create mode 100644 scripts/test-verified-activation.ts create mode 100644 scripts/test-wake-audio.ts create mode 100644 scripts/test-wake-match.ts create mode 100644 scripts/test-widget-magnets.ts create mode 100644 scripts/test-widgets-desktop.mjs create mode 100644 src/main/agent/browser-research-policy.ts create mode 100644 src/main/agent/email-browser-context.ts create mode 100644 src/main/agent/email-overview.ts create mode 100644 src/main/agent/enhancement-context.ts create mode 100644 src/main/agent/latency.ts create mode 100644 src/main/agent/local-agent-context.ts create mode 100644 src/main/agent/model-wait.ts create mode 100644 src/main/agent/realtime/confirmed-speech.ts create mode 100644 src/main/agent/realtime/confirmed-turn-config.ts create mode 100644 src/main/agent/realtime/connection.ts create mode 100644 src/main/agent/realtime/desktop-delegations.ts create mode 100644 src/main/agent/realtime/desktop-status.ts create mode 100644 src/main/agent/realtime/incidental-speech.ts create mode 100644 src/main/agent/realtime/notice-buffer.ts create mode 100644 src/main/agent/realtime/openai-codec.ts create mode 100644 src/main/agent/realtime/playback-speech-evidence.ts create mode 100644 src/main/agent/realtime/playback-turn-policy.ts create mode 100644 src/main/agent/realtime/response-coordinator.ts create mode 100644 src/main/agent/realtime/spoken-turn-guard.ts create mode 100644 src/main/agent/realtime/turn-grounding.ts create mode 100644 src/main/agent/research-budget.ts create mode 100644 src/main/agent/research-reasoning.ts create mode 100644 src/main/agent/research-start.ts create mode 100644 src/main/agent/screen-on-wake.ts create mode 100644 src/main/benchmarks/core.ts create mode 100644 src/main/benchmarks/host.ts create mode 100644 src/main/benchmarks/page.ts create mode 100644 src/main/benchmarks/server.ts create mode 100644 src/main/config/screen-health.ts create mode 100644 src/main/config/voice-commands.ts create mode 100644 src/main/demos/playground-window.ts create mode 100644 src/main/diagnostics/interaction-log.ts create mode 100644 src/main/diagnostics/latency-summary.ts create mode 100644 src/main/diagnostics/report.ts create mode 100644 src/main/diagnostics/runtime.ts create mode 100644 src/main/diagnostics/tool-outcome.ts create mode 100644 src/main/maintenance/capabilities.ts create mode 100644 src/main/maintenance/codex-client.ts create mode 100644 src/main/maintenance/desktop-actions.ts create mode 100644 src/main/maintenance/desktop-ipc.ts create mode 100644 src/main/maintenance/desktop-projects.ts create mode 100644 src/main/maintenance/desktop-tasks.ts create mode 100644 src/main/maintenance/self-enhancement.ts create mode 100644 src/main/maintenance/session-control.ts create mode 100644 src/main/recordings/host.ts create mode 100644 src/main/recordings/store.ts create mode 100644 src/main/recordings/types.ts create mode 100644 src/main/screen-health.ts create mode 100644 src/main/tray-icon.ts create mode 100644 src/main/usage/catalog.json create mode 100644 src/main/usage/ledger.ts create mode 100644 src/main/usage/pricing.ts create mode 100644 src/main/usage/types.ts create mode 100644 src/main/widgets/magnet.ts create mode 100644 src/main/widgets/windows.ts create mode 100644 src/native/README.md create mode 100644 src/native/window-control.m create mode 100644 src/preload/widget.ts create mode 100644 src/renderer/src/components/floating-widget.tsx create mode 100644 src/renderer/src/components/input-transcript.tsx create mode 100644 src/renderer/src/components/microphone-feedback.tsx create mode 100644 src/renderer/src/components/playground/playground-app.tsx create mode 100644 src/renderer/src/components/screen-access-panel.tsx create mode 100644 src/renderer/src/components/settings/recordings-section.tsx create mode 100644 src/renderer/src/components/settings/usage-section.tsx create mode 100644 src/renderer/src/components/spending-meter.tsx create mode 100644 src/renderer/src/components/task-progress.tsx create mode 100644 src/renderer/src/components/widget-slot-picker.tsx create mode 100644 src/renderer/src/lib/dex/cancelled-tools.ts create mode 100644 src/renderer/src/lib/dex/engines/wake-audio.ts create mode 100644 src/renderer/src/lib/dex/engines/wake-match.ts create mode 100644 src/renderer/src/lib/dex/live-microphone.ts create mode 100644 src/renderer/src/lib/dex/realtime/echo-cancellation.ts create mode 100644 src/renderer/src/lib/dex/realtime/echo-reference-output.ts create mode 100644 src/renderer/src/lib/dex/realtime/input-preview.ts create mode 100644 src/renderer/src/lib/dex/realtime/playback-captions.ts create mode 100644 src/renderer/src/lib/dex/realtime/speech-gate.ts create mode 100644 src/renderer/src/lib/dex/realtime/voice-error.ts create mode 100644 src/renderer/src/lib/dex/research-progress.ts create mode 100644 src/renderer/src/lib/dex/sleep-announcement.ts create mode 100644 src/renderer/src/lib/dex/sleep-command.ts create mode 100644 src/renderer/src/lib/dex/wake-cue.ts create mode 100644 src/renderer/src/lib/recordings/acquire-capture.ts create mode 100644 src/renderer/src/lib/recordings/preferences.ts create mode 100644 src/renderer/src/lib/recordings/recorder.ts create mode 100644 src/renderer/src/lib/recordings/use-recording-state.ts create mode 100644 src/renderer/src/lib/task-progress.ts create mode 100644 src/renderer/src/lib/use-screen-health.ts create mode 100644 src/renderer/src/lib/use-usage.ts create mode 100644 src/skills/availability.ts create mode 100644 src/skills/benchmark/meta.ts create mode 100644 src/skills/benchmark/skill.ts create mode 100644 src/skills/capabilities/meta.ts create mode 100644 src/skills/capabilities/skill.ts create mode 100644 src/skills/computer/control-targets.ts create mode 100644 src/skills/computer/execution.ts create mode 100644 src/skills/computer/window-control.ts create mode 100644 src/skills/diagnostics/meta.ts create mode 100644 src/skills/diagnostics/skill.ts create mode 100644 src/skills/git/meta.ts create mode 100644 src/skills/git/repository.ts create mode 100644 src/skills/git/skill.ts create mode 100644 src/skills/local-agent-actions/archive-fallback.ts create mode 100644 src/skills/local-agent-actions/meta.ts create mode 100644 src/skills/local-agent-actions/skill.ts create mode 100644 src/skills/local-agents/meta.ts create mode 100644 src/skills/local-agents/skill.ts create mode 100644 src/skills/open/code-walkthrough.ts create mode 100644 src/skills/open/open-source.ts create mode 100644 src/skills/open/open-url.ts create mode 100644 src/skills/open/quit-app.ts create mode 100644 src/skills/open/verified-activation.ts create mode 100644 src/skills/open/walkthrough-document.ts create mode 100644 src/skills/open/walkthrough-editor.ts create mode 100644 src/skills/open/walkthrough-window.ts create mode 100644 src/skills/realtime-selection.ts create mode 100644 src/skills/recording/meta.ts create mode 100644 src/skills/recording/skill.ts create mode 100644 src/skills/research/meta.ts create mode 100644 src/skills/research/schema.ts create mode 100644 src/skills/research/skill.ts create mode 100644 src/skills/research/view.tsx create mode 100644 src/skills/self-enhancement/meta.ts create mode 100644 src/skills/self-enhancement/skill.ts create mode 100644 src/skills/tricks/catalog.ts create mode 100644 src/skills/tricks/meta.ts create mode 100644 src/skills/tricks/skill.ts create mode 100644 src/skills/tricks/view.tsx diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3035b08..9eb8a0d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -35,6 +35,9 @@ jobs: - name: Typecheck run: pnpm typecheck + - name: Regression tests + run: pnpm test + # Full electron-vite build (main + preload + renderer). Catches bundling # and import errors that `tsc --noEmit` alone misses. No secrets required. - name: Build diff --git a/AGENTS.md b/AGENTS.md index 3abba97..101d672 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,5 +1,13 @@ # OpenDex — agent notes +## Self-diagnostics and local agent milestones + +For diagnostics, self-healing, or local Codex integration work, read and update +`docs/SELF_HEALING.md`. It records acceptance gates, verified connection limits, +decisions, and next actions. Never mark a milestone complete from mocks alone +when its acceptance gate requires a live desktop round trip. Keep raw runtime +logs and conversation content out of that progress record. + OpenDex is an **Electron** desktop app (electron-vite + React + Tailwind v4). It is a voice-first agentic harness, generalized from a Next.js demo. ## Process model (important) @@ -60,7 +68,7 @@ OpenDex is an **Electron** desktop app (electron-vite + React + Tailwind v4). It - **Session grant:** `makePermissionRequester(sender)` holds a per-requester `sessionAllow` set, so an "Allow once" approval covers the rest of *that command's* multi-step tool loop (a fresh requester is built per `chatStart`, so it never silently persists across commands). Needed so computer-use doesn't re-prompt on every click. ## Computer-use (Phase 6) — see & control the desktop -- **Model-agnostic, gated.** `src/skills/computer/` is the `computer` skill (`sensitive`, `optIn`; `skill.ts` + `screen-capture.ts` + `view.tsx`). Tools: `captureScreen` (optional `region` to **zoom** + `displayId` to pick a monitor), `click`, `moveMouse`, `drag`, `typeText`, `pressKeys`, `scroll` (optional `x,y` to aim the target pane), `wait`. Acting tools return a **fresh screenshot to the model** via the AI SDK's `toModelOutput` content path (`{type:"content", value:[{type:"text"},{type:"media", data, mediaType}]}`) — so the screenshot→act→screenshot loop works through the AI Gateway with **any vision model**, not just Anthropic's beta computer-use tool. `SkillTool.toModelOutput` is threaded through `buildToolSet` → `tool()`. +- **Model-agnostic, gated.** `src/skills/computer/` is the `computer` skill (`sensitive`, `optIn`; `skill.ts` + `screen-capture.ts` + `view.tsx`). Tools: `captureScreen` (no arguments, full screenshot), `zoomScreen` (required `region`, needs a prior screenshot), `captureDisplay` (known `displayId` to pick a monitor), `click`, `moveMouse`, `drag`, `typeText`, `pressKeys`, `scroll` (optional `x,y` to aim the target pane), `wait`. Acting tools return a **fresh screenshot to the model** via the AI SDK's `toModelOutput` content path (`{type:"content", value:[{type:"text"},{type:"media", data, mediaType}]}`) — so the screenshot→act→screenshot loop works through the AI Gateway with **any vision model**, not just Anthropic's beta computer-use tool. `SkillTool.toModelOutput` is threaded through `buildToolSet` → `tool()`. - **Adaptive screenshot cadence** (`finishAction`): not every action snapshots. `click`/`drag`/`scroll`/`wait` default to returning a screenshot (they change the view); `typeText`/`pressKeys` default to **none** so the model can chain related keystrokes without a round-trip per key. Every acting tool takes an optional `screenshot` boolean to override, and the system prompt instructs the model to batch and only look when it needs to. Pairs with the `prepareStep` pruning (keep last 2 images) to keep the loop fast. - **Settled + diffed frames.** When an action does snapshot, it goes through `captureStable()` (re-captures until two consecutive frames stop differing, capped ~1s) so the model never acts on a half-loaded/spinner frame, then frame-diffs against the last frame the model saw (32×32 grayscale `signature` + `framesDiffer`): if nothing changed it returns a short **"no visible change"** text note and **omits the image** (cheaper, and a misclick signal). Net per-step image payload trends down. - **Human-like mechanics.** Long `typeText` is **pasted** via the Electron `clipboard` (save → write → ⌘/Ctrl+V → restore) instead of per-keystroke; short text still types. `drag` uses nut.js press/animated-move/release. Cursor moves **animate** (`mouse.move`+`straightTo`) so the session is watchable; the `computer.animateCursor` config flag (Settings → Skills) flips to instant `setPosition` for max speed. @@ -88,3 +96,85 @@ OpenDex is an **Electron** desktop app (electron-vite + React + Tailwind v4). It ## Status Phases 1–4 done (4a cloud/web + 4b free offline); **5a done** (skills + permission gate + Open built-in); **6 done** (computer-use: screen capture + mouse/keyboard, gated & opt-in); **7 done** (always-on visibility: overlay HUD + notch mode + summon hotkey + tray, hide-not-close window model). Roadmap: 5b MCP + more built-ins → signed releases + auto-update. + +## Cool tricks + +`src/skills/tricks/` provides `listTricks`, `chooseTrick`, and `showPlayground`. +The chooser returns a recipe, not completion; the normal agent loop performs it +through existing tools and permission gates. Eligibility uses main-only +`SkillExecutionContext` from `buildToolSet`, filtering disabled/unready/never-granted +skills and macOS desktop permissions. `catalog.ts` is the extension point; see +`docs/COOL_TRICKS.md`. The local particle playground runs in its own `#playground` +window, owned by `src/main/demos/playground-window.ts`. Choosing a trick does not +start recording. Selection history is in memory for the current app run. + +## Manual interaction recordings + +Settings → Recordings (also in the tray) records a selected display plus optional +microphone/system audio. Capture is manual, off until Start; the tray shows REC +and offers Stop. The Settings renderer owns the recorder and its window hides on +close while recording. Main owns bounded file writes, library/export/trash, and +the range-capable playback protocol. Code: `src/main/recordings/` and +`src/renderer/src/lib/recordings/`; see `docs/RECORDINGS.md`. Files live only in +userData/recordings, outside the repo. Do not upload or commit recording files. +This is separate from diagnostic logs, which still contain no audio or images. + +## Local interaction diagnosis + +When the user asks about their last Dex interaction, run `pnpm diagnose` first (or `node scripts/diagnose.mjs`). Use `--last 3` to compare sessions. The local recorder is initialized at app startup and writes `diagnostics/interactions.jsonl` under Electron userData, normally `~/Library/Application Support/opendex`. It retains three approximately 5 MB files. Override the read directory with `OPENDEX_DIAGNOSTICS_DIR` when testing. Reports include transcripts, response IDs/status, tools, playback/interruption events and model/capture timings. Audio and images are not stored. Assistant transcripts describe generated content, not proof every word was heard; compare playback events. No transcript can be reconstructed for interactions before recording was installed. Do not ask the user to repeat their interaction until checking this report. Logs contain conversation content; do not upload or commit them. + +### Recording quick controls + +`controlRecording` (the recording skill, direct in both voice modes) and the notch +record/stop button share `recordings/host.ts` coordination. The Settings renderer +can be created hidden and announces readiness before main sends a start request; +results are correlated and time-bounded. Capture still belongs exclusively to +Settings. Saved screen/audio preferences live in renderer localStorage and are +shared by Settings and quick starts. Stop waits briefly for the final file flush +before returning to the model. Recording must only start on an explicit user request. + +### English transcription and screen demos + +Realtime input transcription explicitly sends `language: "en"`; do not add a +transcription `prompt` without checking model support (the configured gateway +model rejects that field). Display transcripts are asynchronous guidance, not +proof of exactly what the voice model heard. Prefer user corrections over logs. +Cloud STT also specifies English. + +Screen detective now uses `computer.describeScreen` as a direct, permission-gated +read-only tool: one fresh capture plus a bounded vision description, returning +text to realtime. It reuses the wake observer with a task-oriented demo purpose, +not the older wake image or a multi-step desktop agent. Ignore Dex overlays and +never infer blocked controls merely from a progress indicator. + +### Realtime barge-in audio forwarding + +The active session forwards every echo-cancelled microphone frame exactly once +to server turn detection. Silero/SpeechGate now supplies diagnostics only, never +zeroes or delays mic frames: low-confidence double-talk was previously discarded +and barge-in requests disappeared before the model heard them. Wake detection +remains the gate for opening a session. Echo-reference output and browser audio +processing remain enabled. No raw audio is added to diagnostics. + +For realtime models with input transcription, `SpokenTurnGuard` holds spoken +response audio, captions, and tool execution until a nonempty transcript arrives +for that input item. Empty transcription or a five-second post-speech timeout +ends the session quietly, discarding invented server context and returning to +wake listening. Typed requests and tool continuations remain explicit. This is +an output guard, not a microphone gate; transcription can still misrecognize +noise, and models without transcription retain their existing behavior. + +### Guided browser research + +`src/skills/research/` adds `updateResearch`: a complete, validated snapshot of +the plan, source status, sourced findings, and gaps. It is a display tool, not a +search service or independent verifier. The model must actually read sources +through existing browser/computer tools. Main and notch show the latest active +research record; notch height is contributed by `ToolView.notchHeight`. +Realtime delegated workers send substantive milestones through the existing +session IPC, keyed to the active tool's response epoch. `ResponseCoordinator` +serializes narration and rejects stale progress; progress-only responses cannot +execute tools. `ResearchMilestones` throttles speech without periodic updates. +`delegatedReport` selects the final report instead of concatenated narration. +Research gets 96 steps after using the progress tool; ordinary tasks retain 40. +See `docs/RESEARCH.md` for the workflow, validation, and limitations. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 475c793..a8a1c71 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -17,6 +17,7 @@ state machine), see `AGENTS.md`. pnpm install pnpm dev # run the app with HMR pnpm typecheck # tsc --noEmit — run this before opening a PR +pnpm test # local regressions; no provider credentials required pnpm smoke:chat # exercise the agent loop without launching Electron ``` @@ -198,7 +199,10 @@ and `OverlayTranscript` already do this for you. ## Conventions -- **Run `pnpm typecheck` before a PR.** There's no separate lint/test gate yet. +- **Run `pnpm typecheck`, `pnpm test`, and `pnpm build` before a PR.** + `pnpm test` initializes Electron once before parallel test workers; it does not + launch the app. Desktop UI checks are separate; see + [the review guide](docs/CONTRIBUTION_REVIEW.md#validation). - Match the surrounding code's comment density and naming. Comments explain *why*, not *what*. - Keep `meta.ts` dependency-free, and never import a skill's `skill.ts` from the diff --git a/README.md b/README.md index df42e52..793cb24 100644 --- a/README.md +++ b/README.md @@ -48,6 +48,8 @@ OpenDex is an agentic harness built around voice. It's fully customizable: chang - 🧠 **Bring any model** — Apple Intelligence on-device (macOS, free), your own OpenAI, Anthropic, or xAI key, or Vercel AI Gateway (one key, many models). A hosted OpenDex plan is on the way. - **Can run offline** — Vosk wake word + local Whisper + system TTS are a local-first option. No accounts, no uploads — only the LLM call leaves your machine, and you can skip that too on Apple Silicon. - 🔌 **Pluggable voice I/O** — Choose between push-to-talk, Vosk, or Web Speech for wake; local Whisper/Vosk, OpenAI, or Web Speech for transcription; ElevenLabs or the OS voice for output or switch to a fully-integrated Realtime stack using OpenAI Realtime or xAI Voice for a more natural conversation + + For direct OpenAI speech-to-speech, select **Settings → Voice mode → Realtime voice → OpenAI direct** and save your OpenAI API key. This uses GPT Realtime 2 without Vercel AI Gateway for voice. Gateway remains available for OpenAI and xAI voice models. Screen control uses the separately configured language model. - 🎨 **Build your own themes** — Jarvis HUD, Talking Dot, or Typing Cursor. Each one is a full interface, not just a skin, and they react to your mic. - 🛠️ **Build your own Skills** — the agent can open apps, search the web, and more. Risky actions pause for Allow once / Always / Deny; your choice sticks per skill. - **Computer-use (off by default)** — with your OK, it can screenshot the desktop and drive mouse and keyboard. Any vision model works; every action goes through the same permission gate. @@ -144,3 +146,11 @@ Electron · electron-vite · React 19 · Tailwind CSS 4 · Vercel AI SDK v6 · E ## License [MIT](LICENSE) — contributions welcome. + +## Additional workflows + +See the [contribution review guide](docs/CONTRIBUTION_REVIEW.md) for voice and +desktop changes, setup, new skills, testing, and known limits. Feature guides cover +[recordings](docs/RECORDINGS.md), [research](docs/RESEARCH.md), +[usage estimates](docs/USAGE_AND_COSTS.md), [demos](docs/COOL_TRICKS.md), and +[browser benchmarks](docs/BROWSER_BENCHMARK.md). diff --git a/benchmarks/browser/golden.json b/benchmarks/browser/golden.json new file mode 100644 index 0000000..40e79d1 --- /dev/null +++ b/benchmarks/browser/golden.json @@ -0,0 +1,58 @@ +{ + "version": 1, + "id": "dex-browser-golden", + "scenarios": [ + { + "id": "catalog-search", + "task": "Search the catalog for lantern, then open the Amber Lantern (not the Blue Lantern).", + "budgetMs": 120000, + "steps": [ + { "title": "Catalog search", "controls": [ + { "id": "query", "label": "Search catalog", "kind": "input" }, + { "id": "search", "label": "Search", "kind": "button" }, + { "id": "offers", "label": "Special offers", "kind": "button" } + ], "success": { "target": "search", "values": { "query": "lantern" } } }, + { "title": "Search results", "controls": [ + { "id": "blue", "label": "Blue Lantern — $30", "kind": "button" }, + { "id": "amber", "label": "Amber Lantern — $25", "kind": "button" }, + { "id": "back", "label": "Back to catalog", "kind": "button" } + ], "success": { "target": "amber", "values": {} } } + ] + }, + { + "id": "delivery-form", + "task": "Complete the fictional delivery form: recipient Alex Example, city Austin, delivery Express. Save the delivery. This fixture sends nothing and makes no purchase.", + "budgetMs": 180000, + "steps": [ + { "title": "Delivery details", "controls": [ + { "id": "recipient", "label": "Recipient", "kind": "input" }, + { "id": "city", "label": "City", "kind": "input" }, + { "id": "delivery", "label": "Delivery", "kind": "select", "options": ["Choose delivery", "Standard", "Express"] }, + { "id": "save", "label": "Save delivery", "kind": "button" }, + { "id": "reset", "label": "Cancel delivery", "kind": "button" } + ], "success": { "target": "save", "values": { "recipient": "Alex Example", "city": "Austin", "delivery": "Express" } } } + ] + }, + { + "id": "settings-navigation", + "task": "Navigate to Settings, then Notifications. Set frequency to Weekly and save preferences.", + "budgetMs": 150000, + "steps": [ + { "title": "Workspace", "controls": [ + { "id": "projects", "label": "Projects", "kind": "button" }, + { "id": "settings", "label": "Settings", "kind": "button" }, + { "id": "help", "label": "Help", "kind": "button" } + ], "success": { "target": "settings", "values": {} } }, + { "title": "Settings", "controls": [ + { "id": "profile", "label": "Profile", "kind": "button" }, + { "id": "notifications", "label": "Notifications", "kind": "button" }, + { "id": "billing", "label": "Billing", "kind": "button" } + ], "success": { "target": "notifications", "values": {} } }, + { "title": "Notifications", "controls": [ + { "id": "frequency", "label": "Frequency", "kind": "select", "options": ["Daily", "Weekly", "Never"] }, + { "id": "save", "label": "Save preferences", "kind": "button" } + ], "success": { "target": "save", "values": { "frequency": "Weekly" } } } + ] + } + ] +} diff --git a/docs/BROWSER_BENCHMARK.md b/docs/BROWSER_BENCHMARK.md new file mode 100644 index 0000000..e785fb0 --- /dev/null +++ b/docs/BROWSER_BENCHMARK.md @@ -0,0 +1,210 @@ +# Repeatable browser benchmark + +## Ask Dex to benchmark itself + +After loading this build, say **“Dex, benchmark yourself.”** The built-in +`benchmarkDex` tool starts the local fixture automatically. Dex must perform all +three tasks with its normal computer controls, then read the measured result and +summarize it. Realtime uses its existing desktop worker for those screen actions; +pipeline mode uses the same tools directly. No terminal command is needed. + +The first fully attempted suite (successes or timeouts, with no abandoned tasks) +becomes the saved baseline. Say **“Benchmark yourself again”** for a comparison, +or **“What are my benchmark results?”** to read the current run. Results persist +under Electron userData in `browser-benchmarks`, with a separate directory per run. +Baseline selection persists across app restarts; current-run status is in memory. +Duplicate starts reuse the active run. A hard five-minute limit starts when the built-in fixture opens, includes navigation +and waiting, and cancels its bound desktop worker as well as finalizing the fixture; a bound desktop worker ending or being cancelled now finalizes its +fixture immediately. Stopping a blocked run records unfinished tasks without replacing the +baseline. Dex must respect disabled Computer/Open skills and OS access requirements. + +Benchmark comparison wording does not authorize research-source shortcuts. The +research policy permits `openUrl` only for the exact active main-owned benchmark +entry URL, still through its normal permission wrapper. Other localhost addresses, +fixture endpoints, modified URLs and expired run URLs receive no exception. During +work, spoken status questions preserve the worker; explicit Stop still cancels it. +Progress is read from fixture counts rather than inferred from completed clicks. + +Metadata captures the app version, model, voice mode, display sizes/scales and +grants. Browser version/zoom and cold/warm state are not automatically observed; +keep those consistent. This measures the browser workflow, not automatic tuning. +After building, quit and relaunch Dex through the normal app entry point. Verify +that the request starts real browser actions, all three outcomes appear, and a +second request reports baseline deltas. Also verify denial and Stop. Build and +fixture tests do not establish live voice acceptance. On macOS, capture and +Accessibility grants belong to the launching app identity; development and +packaged launches may have different grants. + +## Developer CLI + +The developer CLI serves a private, loopback-only fixture site. Ask Dex to operate +it with the existing computer-control workflow. It does not start another agent, +change configuration, bypass permission gates, or connect to real shopping/accounts. +No additional dependencies, credentials or running-app update are required for the +harness itself. Dex still needs a configured vision-capable model, enabled Computer +and Open skills, their normal grants, and OS Screen Recording/Accessibility access. +A denied or unavailable capability is a setup blocker, not a reason to alter policy. + +## Run a baseline and comparison + +From this checkout: + +```sh +pnpm benchmark:browser --out /tmp/dex-baseline --environment 'revision=; model=; browser=; viewport=1280x900; scale=2; OS=; mode=pipeline; grants=Ask; state=warm' +``` + +Use real environment values. The CLI prints a unique local URL and a prompt. Give +that prompt to Dex in a fresh session. Dex opens the page, clicks **Start scenario**, +performs the task using screenshots and ordinary computer tools, and repeats for +the remaining scenarios until **Benchmark complete**. Do not help with clicks or +typing during measurement. If a permission prompt appears, answer normally and +record the grant policy consistently across runs. Use the same browser dimensions, +zoom, display scale, model, voice mode and cold/warm setup on every comparison. +Each step starts from clean fixture fields; reloading retains the measured step +and timer. Fixture controls deliberately do not navigate to external sites. + +Read `/tmp/dex-baseline/report.md` and `report.json`. Press Ctrl+C to close the +server. Ctrl+C during a scenario records `aborted`; remaining tasks are `not-run`. +An active scenario times out at its budget and the next Start page becomes ready. +Waiting to press Start is unmeasured; there is no automatic desktop takeover. + +After changing one variable, run: + +```sh +pnpm benchmark:browser --out /tmp/dex-candidate --baseline /tmp/dex-baseline/report.json --environment '' +``` + +Repeat the printed prompt in a fresh Dex session. The resulting Markdown is both +the current run report and the comparison report: per-scenario status transitions, +time, misclick and score deltas. Negative time/misclick deltas and positive score +deltas indicate improvement. Time deltas are withheld unless both attempts succeeded. +Run several trials per revision; one noisy model run is not evidence of a trend. +The CLI refuses to overwrite an existing `report.json`, so use a new output folder. +While running, reports are partial checkpoints; only completed or Ctrl+C-finalized +reports should be used as baselines. Scenario/hash/order mismatches are rejected. + +## Golden definitions and scoring + +`benchmarks/browser/golden.json` supplies three scenarios: search and choose a +specific result; fill and submit a form; navigate settings and save a selection. +Pass `--suite path/to/suite.json` to use a custom set of three to five scenarios. +The strict Zod schema in `src/main/benchmarks/core.ts` is the format contract. +Each scenario has a unique ID, visible task, time budget, and ordered fixture steps. +Each step defines labelled button/input/select controls, one exact success button +and exact required field values. All criteria must be met in order; a wrong action +leaves the step unchanged. Add distractor buttons to measure target selection. +This is a bounded fixture workflow, not arbitrary HTML or a real-site test runner. +Definitions are validated before serving; task text is escaped, never executed. + +The server checks clicks and form state against the golden criteria. It does not +trust a model's completion message. Timing uses the server's monotonic clock from +Start to final valid action, including subsequent reasoning, tool and permission +waits. It excludes URL opening, initial model latency and waits between scenarios. +Timeouts are capped at the scenario budget. A misclick is a document click outside +a control or on a non-goal button in that step. Input/select focus is allowed. +A correct button with wrong field values counts as an invalid submission instead. +Browser chrome, other windows, scroll accuracy and mouse movement are not measured. +Humans can complete the fixture too; attribution to Dex is an operator protocol, +not an anti-cheating guarantee. Client events require trusted browser interaction; +source/network inspection or scripted DOM actions invalidate a Dex benchmark. + +For success, score is rounded to the nearest integer: + +`100 × (0.70 + 0.20 × (1 − elapsedMs / budgetMs) + 0.10 / (1 + misclicks + invalidSubmissions))` + +Failure, timeout, aborted and not-run score zero. The report includes mean score +and all raw counts; no score hides a failed task. Suite content and fixture/scoring version +are fingerprinted, preventing comparisons across changed goals or budgets. Bump +the version and schema when changing scoring semantics. Environment labels remain +operator supplied and are displayed for review, not silently assumed equivalent. + +Given the same validated suite, environment and measured event timeline, JSON and +Markdown output are deterministic. Actual agent actions and elapsed times are +stochastic and are expected to differ. JSON retains the entire suite, ordered click +trace, step/revision, coordinates, submitted fixture values, outcome and elapsed +time for failure reproduction. It captures no screenshots, audio or conversations. +Use only synthetic data in custom fixtures; local output may contain typed values. +Keep run outputs outside the repository; no upload or recording is performed. + +## Verification and activation + +```sh +pnpm test:browser-benchmark +pnpm typecheck +pnpm build +``` + +Tests validate golden criteria, timing, exact event replay, misclick accounting, +timeouts/cancellation, comparison compatibility and escaping. They do not prove +Dex desktop or voice acceptance. To verify live, run baseline and candidate as +above, observe Dex completing every task without help, deliberately exercise one +wrong target in a separate labelled manual trial, and check failure/timeout output. +Confirm skill denial stops Dex instead of bypassing access. The CLI needs permission +to bind a local loopback port; if blocked, allow that operation through the existing +development environment approval path. It has no public network listener. + +The CLI and the built-in skill share the main-process fixture server and scoring +implementation. CLI use requires no app restart; the new built-in skill requires +loading this build. This implementation does not restart Dex automatically. + +## Control grounding and run limits + +On macOS, Computer screenshots include bounded, read-only Accessibility observations +of the foreground window’s interactive controls. Centers are converted into the +latest screenshot coordinate space, including Retina scale and zoom crops. Dex +still uses its permission-gated mouse/keyboard controls and checks window freshness; +this does not execute page source or expose benchmark answers. Applications without +usable Accessibility trees retain the screenshot workflow. Field values and secure +text fields are excluded. Traversal is bounded and may omit some controls. + +Clicking an input label correctly counts as field focus. Fixture version 2 prevents +comparing scores to the older label-misclick behavior. The built-in time limit also +persists a stop reason and never promotes an abandoned run to baseline. Existing +per-scenario limits remain unchanged. The standalone CLI does not own Dex’s worker; +use the built-in request for the hard worker cancellation guarantee. + +Automatic benchmark voice updates announce only verified scenario outcomes, rather +than repeating periodic current-step snapshots that can become stale during speech. +An explicit status question reads fresh main-owned measurements when its response +can begin. Snapshot phrasing identifies the latest check; completed milestones are +past verified outcomes. The normal Stop behavior remains available. + +The development build was live-tested with two uninterrupted Safari runs: all three +scenarios passed twice with zero misclicks. Scores were 96.33 and 97.00. The real +realtime model and desktop worker were used via typed Dex requests; this does not +establish acoustic voice acceptance or a long-term performance trend. See the +self-healing record for validation, artifact identifiers and remaining live gates. + +## Recorded outcomes and consistency + +`report.json` now includes `summary.status` (`running`, `succeeded`, `failed`, +`incomplete`, or `error`), aggregate counts/metrics, and a fixture snapshot with +revision, scenario index, step, started state and terminal completion signal. +Each finalized row retains timing, misclicks, invalid submissions, completed steps, +score and transition. Without a baseline the transition describes execution; +with a baseline it describes the baseline-to-current outcome. Row status `success` +means completed successfully; `not-run` is reserved for unstarted scenarios. +Mean score uses the whole suite denominator; elapsed time sums scenario durations +and excludes navigation and waits between scenarios. Zero errors are valid metrics. + +The skill retains its lifecycle `status` (`running`/`finished`/`error`) and exposes +`outcome`, `metrics`, and `consistency` separately. A full successful run has +`outcome: succeeded`. The reports compare accepted fixture step signals, timing, +error counts and scoring with the rows. Inconsistencies retain the original evidence +and show explicit scenario-specific errors; they never repair records by guessing +success or replacing them with not-run. Inconsistent baselines are refused. + +Writes replace each report atomically and verify its contents before acknowledging +success. Completion callbacks use only verified persisted snapshots. JSON and +Markdown are separate files, not a single crash-atomic transaction; an interrupted +write may leave different revisions, and JSON is the authoritative evidence. +A filesystem failure surfaces as a recording error in the fixture and host status; +it cannot promote a baseline. Existing historical reports are not rewritten. + +After building and relaunching Dex, use a fresh session and request a benchmark. Let Dex complete all tasks +through its normal permission gates, then ask for benchmark status. Verify all +three rows have `success`, finite elapsed time, nonzero scores, error counts and +transitions, `summary.status: succeeded`, and `consistency.ok: true` in the run's +report.json. Confirm report.md and spoken summary agree, including after worker +completion. Repeat under the same browser conditions to verify baseline transitions. +Synthetic regression tests do not establish live desktop or acoustic acceptance. diff --git a/docs/CONTRIBUTION_REVIEW.md b/docs/CONTRIBUTION_REVIEW.md new file mode 100644 index 0000000..dfe0926 --- /dev/null +++ b/docs/CONTRIBUTION_REVIEW.md @@ -0,0 +1,133 @@ +# Contribution review guide + +This contribution combines voice and desktop reliability fixes with new recording, +research, usage, demo, and local-agent workflows. It is based on upstream v1.1.14 +(`3e89834`). The feature areas below share the main IPC host, preload contract, +voice hook, and tool registry; review those integration points after the modules. + +## Scope and suggested review order + +| Area | Resulting behavior | Start here | Regression coverage | +| --- | --- | --- | --- | +| Voice lifecycle and wake | Release/reacquire the mic; retain wake audio; add a wake cue, follow-up listening, sleep/new-session controls, and live input previews. | `lib/dex/use-dex.ts`, `engines/wake-audio.ts`, `engines/vosk-wake.ts` in the renderer | `test-live-microphone`, `test-wake-*`, `test-realtime-idle`, `test-new-session-command`, `test-input-preview` | +| Realtime response ownership | Coordinate responses and tool continuations; wait for confirmed spoken input; preserve work through incidental speech and status questions; cancel superseded desktop work. | `src/main/agent/realtime/session-host.ts`, `response-coordinator.ts`, `confirmed-speech.ts`, `playback-speech-evidence.ts` | `test-spoken-turn-*`, `test-response-coordinator`, `test-confirmed-speech`, `test-playback-*`, `test-desktop-execution` | +| Realtime connection and recovery | Add direct OpenAI voice; show actionable provider failures; buffer early errors until the preload subscribes. | `realtime/connection.ts`, `openai-codec.ts`, `notice-buffer.ts`; renderer `realtime-session.ts`, `voice-error.ts` | `test-openai-realtime`, `test-realtime-feedback` | +| Screen and app controls | Recover from missing screen access; opt-in wake observation; fresh screen descriptions; reference-based zoom; direct macOS window, volume, activation, and normal quit controls. | `src/skills/computer/`, `src/skills/open/`, `src/native/`; main and renderer screen-health modules | `test-screen-on-wake`, `test-direct-controls`, `test-open-url`, `test-quit-app`, `test-verified-activation` | +| Permissions and tools | Share concurrent skill prompts within one command; preserve cancellation and standing denial; confirm Git mutations individually; advertise only usable search capabilities. | `src/main/agent/permissions.ts`, `src/skills/registry.ts`, `availability.ts`, `git/`, `web-search/` | `test-permission-deferral`, `test-enhancement-permissions`, `test-git-skill`, `test-search-availability` | +| Research and task narration | Show plans, sources, findings, and gaps; coordinate one browser worker; require visible source-link navigation; budget research separately; surface stalled/empty results and concise email overviews. | `src/skills/research/`, `src/main/agent/research-*`, `browser-research-policy.ts`, `email-overview.ts`, `model-wait.ts` | `test-research*`, `test-browser-research-policy`, `test-email-overview`, `test-empty-chat`, `test-model-wait` | +| Recordings | Explicit screen/mic/system-audio recording, local library, playback, export, Trash, and quick controls. | `src/main/recordings/`, renderer `lib/recordings/`, `src/skills/recording/` | `test-recording-store`, `test-recording-capture` | +| Usage | Persist per-request usage and estimated/reported costs, including incomplete and unpriced requests; show launch/day totals and Settings history. | `src/main/usage/`, renderer `usage-section.tsx`, `spending-meter.tsx` | `test-usage`; separate `test-usage-desktop.mjs` | +| Diagnostics and local agents | Inspect local interactions and latency; discover/read Codex tasks; permission-gated task dispatch/update/revise/archive and source enhancement. | `src/main/diagnostics/`, `src/main/maintenance/`, skills `diagnostics/`, `local-agents/`, `local-agent-actions/`, `self-enhancement/`, `capabilities/` | `test-diagnostic-*`, `test-interaction-log`, `test-latency-summary`, `test-local-*`, `test-desktop-bridge`, `test-session-control`, `test-self-enhancement` | +| Demos and source walkthroughs | Choose eligible tool-backed demos, show a local particle playground, open the app's source in an editor, and navigate a code walkthrough. | `src/skills/tricks/`, `src/main/demos/`, `src/skills/open/code-walkthrough.ts` and `walkthrough-*` | `test-tricks`, `test-open-source`, `test-code-walkthrough` | +| Compact UI and widgets | Improve transcript wrapping, playback captions, mic/progress/error feedback, button targets and tray visibility; detach widgets into explicit exclusive desktop slots. | renderer `compact-bar.tsx`, `NotchApp.tsx`, themes; `src/main/widgets/`, `src/preload/widget.ts` | `test-widget-magnets`, `test-playback-captions`; separate `test-widgets-desktop.mjs` | +| Browser benchmark | Run three local fixture tasks, persist verified results/baselines, surface inconsistent or failed writes, and cancel bound workers on timeout. | `src/main/benchmarks/`, `src/skills/benchmark/`, `benchmarks/browser/golden.json` | `test-browser-benchmark`, `test-browser-research-policy` | +| Build and small fixes | Bundle speech detector assets, compile the macOS addon, add a restricted widget preload, fix pnpm workspace configuration, and use the local timezone for unspecified clock requests. | `electron.vite.config.ts`, `scripts/build-window-control.mjs`, `pnpm-workspace.yaml`, `src/skills/clock/` | `test-clock`, typecheck, production build | + +Test names above are files under `scripts/` with the `.ts` extension unless noted. +Renderer paths abbreviated in the table are under `src/renderer/src/`. Tests mix +pure logic, source assertions, and mocked hosts; passing them is not equivalent +to a live model, microphone, browser, or desktop integration trial. + +## Setup and compatibility + +Install with the declared pnpm version (10.8.1), then run `pnpm install +--frozen-lockfile`. Existing configuration merges the new `computer.screenOnWake` +field as `false`; pipeline and gateway defaults are retained. Direct OpenAI voice +is an explicit Settings choice using the existing OpenAI key. The configured +language model still handles delegated desktop work. + +Diagnostics, local-agent discovery/actions, and self-enhancement skills are opt-in. +Other new skills use the existing enablement and permission conventions. Turning +off the Diagnostics skill does **not** disable local transcript storage. Screen +observation on wake is off by default and requires the existing computer access. +Recording begins only through an explicit control or command. + +On macOS, building now requires Xcode command-line tools for the Node-API window +control addon. Other platforms skip native compilation. New runtime dependency: +`@ricky0123/vad-web`; new build dependency: `node-api-headers`. Speech detector and +ONNX runtime assets are bundled for local use. The addon is unpacked from ASAR. +See [native controls](../src/native/README.md) before packaging for another Mac +architecture. No release/version bump is included. + +## Data and review-sensitive behavior + +| Data or integration | Behavior and limitation | +| --- | --- | +| Diagnostic history | **Automatically stores conversation transcripts locally**, with best-effort secret redaction and approximately 15 MB rotation. No opt-out/delete UI yet. Permissioned diagnosis can pass history to the chosen model. See [diagnostics](SELF_HEALING.md). | +| Video recordings | Explicit start; local screen/audio files; manual export and deletion; capture duration/size caps. See [recordings](RECORDINGS.md). | +| Usage journal | Local counters and cost metadata, without prompts/media; append-only and replayed in memory. No pruning, budget enforcement, or invoice reconciliation. See [usage](USAGE_AND_COSTS.md). | +| Local-agent bridge | Reads selected local task history and can dispatch work after permission. macOS desktop mutations are restricted to one tested version/build; private IPC compatibility is a maintenance risk. | +| Research policy | Source pages must be reached through visible links; several URL/clipboard/search-API shortcuts are blocked during research. This is a deliberate workflow restriction requiring product review. | +| Voice interruption | Confirmed transcription avoids false interruption but adds latency. English transcription is explicit. Mic forwarding stays continuous; speech classification is diagnostic-only. | +| Model work and costs | Research can use 96 steps instead of the ordinary 40. Screen descriptions and local-agent tasks may incur separate model usage; the ledger cannot observe external-agent billing. | + +## Validation + +Run from the repository root: + +```sh +pnpm install --frozen-lockfile +pnpm typecheck +pnpm test +pnpm build +``` + +`pnpm test` resolves Electron before starting parallel test workers, avoiding +competing binary extraction on a fresh install. It includes every +`scripts/test-*.ts` file. CI runs the same test entry point, typecheck, and build. +No paid provider request is needed for this suite. Individual `test:*` scripts +remain available for focused iteration. + +Optional isolated desktop checks after building: + +```sh +node scripts/test-usage-desktop.mjs +node scripts/test-widgets-desktop.mjs +``` + +These launch real Electron windows with temporary profiles and synthetic data. +They do not prove real provider accounting or acoustic behavior. Live provider +smoke tests (`pnpm smoke:chat`, `pnpm smoke:realtime`) require configured keys and +may incur charges. Do not attach private diagnostic logs or recordings to a PR. + +Submission checks on macOS arm64, Node 25.6.1, and pnpm 10.8.1: + +- Frozen-lockfile install, typecheck, and production/native build passed. +- All 307 regression tests passed with no skipped tests. +- Both isolated Electron harnesses (usage and widgets) passed. +- No paid-provider smoke test or live acoustic test was run for this submission. +- Linux CI uses Node 22; that environment has not been reproduced locally. + +Build warnings include +mixed static/dynamic imports, large renderer chunks, and deprecated macOS app +lookup. Signing/notarization, cross-architecture packaging, Windows/Linux runtime, +provider billing agreement, and final combined-branch acoustic acceptance remain +unverified. Treat this as a broad contribution for review, not a release claim. + +## Manual acceptance checklist + +1. Start with both fresh and existing profiles. Confirm provider selection, + screen-access recovery, and opt-in defaults; never copy a personal profile. +2. Test wake, follow-up, sleep, Stop, new session, real interruption, and a status + question during a tool task. Confirm one reply and no abandoned worker. +3. Exercise a provider startup failure and an in-session failure; verify visible + recovery text in both main and notch. Restore a usable provider and verify a + successful audible response. Repeated failure must not replay work. +4. Exercise Allow once, Always, Deny, Never, parallel calls, and cancellation. + Confirm a denied tool cannot run and a new command has fresh session scope. +5. Record a short clip explicitly, stop, play, export, and Trash it. Verify the + indicator, window-close behavior, and incomplete-save handling. +6. Inspect usage with reported and unknown costs. Review transcript-storage + defaults before enabling this build for general distribution. +7. Research through visible browser links, interrupt it, and confirm source + cards and final report agree with visited evidence. Complete a benchmark and + repeat under the same conditions; compare persisted rows with its summary. +8. Verify widget slot conflicts, detach/close, resizing, displays, and app quit. + Test native controls with an app that can reject a resize or quit request. +9. On the supported Codex desktop version, verify discovery, exact task targeting, + create/follow/result, revision, and archive rejection. Tests with mocked IPC + alone cannot satisfy this gate. + +Further feature details: [confirmed voice turns](VOICE_TURN_CONFIRMATION.md), +[research](RESEARCH.md), [demos](COOL_TRICKS.md), and +[browser benchmark](BROWSER_BENCHMARK.md). diff --git a/docs/COOL_TRICKS.md b/docs/COOL_TRICKS.md new file mode 100644 index 0000000..1085fec --- /dev/null +++ b/docs/COOL_TRICKS.md @@ -0,0 +1,80 @@ +# Cool tricks + +Say “Dex, show me something cool” to select and perform one built-in demo. +“Show me another trick” chooses another eligible recipe. “What tricks do you +know?” lists available demos without running them. You can also request a title +from that list, or say “show the constellation” to open that playground scene. + +The skill is enabled by default and can be disabled in Settings → Skills & tools. +It works through the normal text/pipeline and realtime tool loops. A demo never +automatically starts recording; use Settings → Recordings first to capture it. + +## Starter library + +| Trick | Demonstrates | Needs | +| --- | --- | --- | +| Gravity playground | Local interactive visuals, orbit/constellation, pointer and keyboard input | Cool tricks | +| Two cities, one moment | Live weather and time comparison | Weather, clock, network | +| A small jump through time | Live date-line comparison | Clock | +| A tiny rabbit hole | Opening a search directly in the browser | Open apps & URLs | +| Window boomerang | Minimize and restore the same Safari window | macOS, Open, computer control, Accessibility | +| Screen detective | A fresh screen observation with a useful next step | Computer control, screen access, vision-capable desktop agent | + +Selection uses enabled/configured skills and standing permissions. Desktop demos +also check macOS system permissions. Network services can still fail at runtime; +the agent is instructed to stop and explain rather than fake success. Normal +sensitive-tool prompts still apply. Recipes stop on errors, denial, or a new user +request. No step may bypass a disabled skill or start an indefinite demo loop. + +The first surprise favors the visual playground. After that, selection rotates +randomly through the eligible library without repeats until the cycle is used up. +Selection history lasts for the current app run and includes explicitly requested +tricks. Choosing a recipe counts as a selection, even if a later action fails. + +## Adding a trick + +Add a `Trick` entry in `src/skills/tricks/catalog.ts`: + +1. Give it a stable ID, short title, description, and optional spoken inspiration in `intro`. +2. Declare every required skill and any platform/screen requirements. +3. Provide tool steps with concrete inputs or clear instructions for the normal + agent loop. Use existing direct tools where possible; delegate visual work. +4. Define what evidence counts as success and how to stop on failure. +5. Add eligibility or selection tests where the new dependency changes behavior. + +For a capability Dex does not yet have, build it as a normal self-contained skill +first, with its own metadata, permission requirements, implementation, and tests. +Then compose it into a demo recipe. Recipes should demonstrate working capability, +not promise functionality that is still on the roadmap. + +`chooseTrick` returns `selected_not_performed` deliberately. The model must execute +the returned steps through the normal tool set. It does not directly call sensitive +tools from inside the chooser, so the existing permission and interruption paths +remain in use. Main-only `SkillExecutionContext` supplies the configured capability +snapshot without sending config or secrets in a tool result. + +The playground is a reusable local Electron window (`#playground?scene=…`). It is +a built-in particle illustration, not a physical simulation or newly generated app. +It respects reduced-motion preferences on startup and has keyboard controls. Its +window is independent of the notch and the main voice session. + +## Checks + +`node_modules/.bin/tsx --test scripts/test-tricks.ts` + +Checks availability, permission/platform exclusions, no-repeat cycles, unavailable +explicit requests, and the distinction between selection and execution. + +## Spoken style + +Perform the demo without announcing selection or narrating each step. Intros are +optional inspiration, not scripts. Visual demos usually need at most one short +comment or interaction hint, with no closing capability recap. Information demos +should deliver the interesting finding. Success criteria and verification metadata +guide honest behavior internally; they are not spoken disclaimers. Explain real +failures or uncertainty only when they affect the requested result. + +Screen detective uses the permission-gated `describeScreen` direct tool, which +captures once and runs one bounded vision request. It does not delegate a general +desktop loop or use the older wake snapshot. It reports relevant observations +rather than listing generic interface furniture, and never clicks or types. diff --git a/docs/RECORDINGS.md b/docs/RECORDINGS.md new file mode 100644 index 0000000..94c29bd --- /dev/null +++ b/docs/RECORDINGS.md @@ -0,0 +1,64 @@ +# Interaction recordings + +Open Settings → Recordings (also available from the tray menu). Choose a display, +leave Microphone and System audio checked to include both sides of the conversation, +then press Start recording before waking Dex. Stop in Settings or the tray menu. +The tray displays REC while capture is active. Closing Settings hides its window +during recording so it can keep capturing while Dex works in other apps. + +Saved recordings appear below the controls. Select one for playback, export a copy +through the save dialog, show the original file, or move it to the system Trash. +Export does not upload or publish anything. There is no automatic recording. + +Capture includes the selected screen and its notifications. System audio includes +other apps. Use headphones to minimize microphone pickup of speaker playback. +Recordings are stored under Electron userData/recordings, normally +`~/Library/Application Support/opendex/recordings`, with private file permissions. +Each clip has a JSON sidecar. Do not commit or upload these files as diagnostics. +Recordings stop at thirty minutes or two gigabytes; existing clips remain until +the user removes them. There is no automatic deletion of older recordings. + +The encoder prefers MP4 with H.264/AAC and falls back to another supported MP4 or +WebM format. Export keeps the recorded format. Interrupted clips are preserved +when possible, but a crash can leave an incomplete, unplayable video. Ordinary +quit requests give the renderer up to eight seconds to flush the final chunk. + +## Implementation + +- `src/main/recordings/host.ts` owns IPC, display grants, the library, export and + the streaming playback protocol. Only the Settings renderer can begin or write + a recording. A display grant is scoped to that recording and consumed once. +- `store.ts` serializes bounded writes and persists metadata. Renderer chunks + arrive once a second; recordings are never accumulated as a single RAM blob. +- The Settings renderer owns `lib/recordings/recorder.ts`, separate from the voice + session. It mixes system audio and a dedicated microphone stream into one track, + without routing that mix to speakers. Cleanup releases only its own resources. +- Pending capture requests can be cancelled or time out; late streams are stopped. + No voice model, diagnostic transcript, API key, or screenshot tool is involved. +- Playback uses a restricted `opendex-recording://video/` URL with byte ranges. + +macOS requires screen/audio capture permission. Packaged builds include +NSAudioCaptureUsageDescription; launching from a development host may also depend +on that host's permission configuration. See the official +[Electron capture notes](https://www.electronjs.org/docs/latest/api/desktop-capturer). +The local macOS capture path has been exercised; Windows and Linux still require +live platform validation. System audio defaults off on Linux. + +## Checks + +`node_modules/.bin/tsx --test scripts/test-recording-store.ts scripts/test-recording-capture.ts` +checks ordered writes, interrupted recovery, cancellation, timeout, late-stream +cleanup, file permissions, ID validation, and trash routing. + +## Quick controls + +The notch has an always-visible record circle that becomes a stop square during +capture. Say “Dex, start recording” or “Dex, stop recording” for the same controls +in pipeline or realtime voice mode. The Interaction recording skill can be disabled +for voice; the manual notch stop remains available. No demo starts capture on its own. + +Quick start reuses the last screen, microphone, and system-audio choices in Settings. +Initially it uses the first available display, microphone on, and system audio on +(except Linux). If a saved display is disconnected, it uses the first available +display. The settings renderer can start hidden and owns the capture throughout. +Start returns actual capture readiness; stop flushes and saves asynchronously. diff --git a/docs/RESEARCH.md b/docs/RESEARCH.md new file mode 100644 index 0000000..72f9180 --- /dev/null +++ b/docs/RESEARCH.md @@ -0,0 +1,116 @@ +# Guided browser research + +Ask Dex to investigate a question, compare options, or research a topic in a +particular browser. For example: “Research the tradeoffs between these options +in Chrome. Show me your plan, compare original sources, and explain where the +evidence disagrees.” Opening a search page alone remains a simple action. + +The Research companion skill is enabled by default. It publishes a tailored +plan before research, then updates the current activity, source record, +source-linked findings, and unresolved questions. It appears in the existing +tool-result area and a scrollable notch view. Source links can be opened from +the record. There are no additional search credentials for the progress view; +actual browsing still uses the existing computer tools and permissions. + +The model is instructed to break broad questions into subquestions, vary queries +and search engines as useful, open original sources, collect specific data with +dates/units/methods, and compare independent evidence. Depth follows the question +and evidence gaps, rather than a fixed source count. Search-engine snippets are +leads. Source status distinguishes found, read, and unavailable pages. A finding +must reference a source included as read in the same snapshot; a completed +record requires findings and a finished plan. These are agent-reported claims, +not an independent verifier of browser reading or factual accuracy. + +The model sends complete snapshots through `updateResearch`, with up to five +plan steps, twenty-four sources, twelve findings, and four open questions. +Snapshots remain in the current interaction; this is not a saved research +library or document exporter. The latest active research record takes precedence +over individual search-result cards. The skill can be disabled in Skills & tools. + +Realtime browser research uses the existing `run_task` worker. The parent +presents the plan once and hands off the question, requested browser, constraints, +and plan. The worker updates the record as it investigates. Substantive findings, +changes of direction, and blockers can request a short spoken update, at most +once every fifteen seconds and only when local playback is quiet. There is no +periodic “still working” narration. Progress requests are tied to the current +tool's response epoch; interruptions discard queued milestones. Progress-only +responses cannot execute tools. The final handoff uses the last assistant report +rather than concatenating earlier plans and progress into the answer. + +The pipeline retains forty tool/generation steps for routine tasks. Calling +`updateResearch` enables ninety-six steps, with a closing reminder near the +limit. This is bounded research: reaching the limit without a report is surfaced +as incomplete. Microphone forwarding, permission gates, and provider selection +are unchanged. + +Validation: targeted tests cover source references, completion state, safe link +schemes, milestone throttling and interruption, response coordination, report +selection, and the larger research budget. The mocked session host also checks +that a progress-only response cannot launch another task. `preview-research.tsx` +renders sample data for visual QA of main/notch layouts; it does not perform +research. Real-world source quality and full browser investigations still need +live evaluation with the user's chosen model and browser permissions. + +## GPT-5 research startup latency + +The direct OpenAI `gpt-5` research worker now uses minimal reasoning while +planning and gathering evidence, and medium reasoning after a research record +enters comparing, synthesizing, complete, or blocked. Other models/providers and +ordinary requests retain their existing settings. The research manual requires +a stage update before comparison or synthesis. A progress-only step temporarily +removes updateResearch from the next step's active tools, preventing consecutive +plan-card loops while retaining real tool actions and final text. + +The main loop records provider-stream-ready, reasoning-start/end, and +action-generation-start timing metadata; it does not log reasoning content. +Tool errors now settle their visible invocation instead of remaining running. + +The local `scripts/probe-research-start.cjs` runs with the configured model and +tool definitions. It never executes browser/computer tools. `--effort=minimal` +compares the first action with the default; `--flow` uses the real chat loop, +executes only the side-effect-free research-record tool, and stops when another +tool is selected. It prints metadata and may print the generated final answer. + +## Plan and first URL in one model call + +`withResearchStart` composes the already permission-wrapped updateResearch and +openUrl tools into a desktop-worker `startResearch` action. The model supplies +a compact plan plus its first HTTP(S) URL together. Execution publishes a clean +planning record, then calls the existing URL wrapper immediately, with no model +round trip in between. Nested tool events preserve the research card and visible +URL action. Cancellation is checked before each operation; URL denial remains a +denial. The action is absent if either underlying tool is unavailable. Existing +tab requests and supplied plans use their appropriate browser tools directly. + +Startup constructs empty evidence/source lists itself; the model cannot invent +read evidence in this initial record. The combined result participates in the +research step budget and reasoning-stage selection. startResearch is removed +from subsequent steps so a worker cannot repeatedly restart the plan. + +### Single browsing owner and search-based startup + +Realtime research now delegates planning and browsing to the worker; it must +not open a competing page or duplicate the plan before run_task. New worker +startup takes a query plus a supported search engine and optional browser; +the app encodes the query into a search URL. It no longer accepts a generated +article path in startResearch. User-supplied URLs and existing tabs remain +ordinary browser tasks. Explicit 404/503/access errors require an explained +fallback to an accessible result, not repeated waiting on the error page. + +### Visible links only + +The research policy requires reaching source pages by clicking links +visibly presented in the browser. Source addresses must not be invented, +copied, pasted, typed, or opened directly, including addresses supplied by a +search API. Search queries may be entered to obtain visible results. Visited +page URLs may still be recorded for citations. + +The desktop worker's browser-research policy wraps existing tool executors, +retaining their permission gates. During research it refuses direct openUrl +calls except narrowly validated query-only Google/Bing/DuckDuckGo entry points, +URL-shaped typeText input, copy/paste keyboard chords, and webSearch API calls. +The policy activates for recognized research requests and when updateResearch +runs, including inside startResearch. Research and delegation instructions +require screenshot-grounded clicks and error recovery through visible links. +The coordinate-click tool itself does not prove a target is a visible link; +that grounding remains an agent instruction, not a DOM-level guarantee. diff --git a/docs/SELF_HEALING.md b/docs/SELF_HEALING.md new file mode 100644 index 0000000..cfdb793 --- /dev/null +++ b/docs/SELF_HEALING.md @@ -0,0 +1,63 @@ +# Diagnostics and local agent integration + +This guide describes the contribution's current behavior and acceptance gates. +Development-session journals and private interaction histories are not included. +See [the contribution review guide](CONTRIBUTION_REVIEW.md) for scope and checks. + +## Inspecting an interaction + +Run `pnpm diagnose` for the latest interaction, or `pnpm diagnose -- --last 3` +to compare recent sessions. `pnpm dex:doctor` inspects local agent availability; +`pnpm diagnose:latency` reads aggregate latency measurements. Missing credentials, +permissions, provider credits, and unsupported adapters are setup blockers, not +proof that application code needs modification. Generated assistant text is not +proof that playback completed; compare playback and interruption events. + +Interaction history is initialized at application startup, independently of the +opt-in Diagnostics skill. It contains user/assistant transcripts, response/tool +identifiers, timing, and playback events. It does not store audio or screenshots. +It rotates across three approximately 5 MB files in `userData/diagnostics`. +Pattern-based credential redaction is best effort: ordinary personal information +in spoken text can remain. There is currently no recording opt-out or deletion +UI for this transcript history. Quit Dex before manually removing the diagnostic +files. This default requires maintainer review before release; it is separate +from anonymous analytics and manual video recording. + +Do not upload or commit runtime history. The Diagnostics skill can return history +to the selected model after its normal permission gate; local storage alone does +not mean a diagnostic request remains entirely on-device. Aggregate latency files +exclude conversation content, but do not independently separate every permission +wait or overlapping operation. They cannot establish a speed improvement alone. + +## Local agents and source changes + +The read-only local-agent inventory, task actions, and self-enhancement skills are +opt-in and permission-gated. The desktop adapter currently supports macOS and +checks app version `26.901.51231`, build `8109`, before using the local socket. +It validates socket ownership and rejects unknown versions. This is a version- +specific integration, not a stable public API or general compatibility promise. +The CLI/history adapter and desktop task adapter expose different capabilities; +missing live desktop access must not be presented as successful task control. + +Self-enhancement prepares reviewable source changes through an available local +agent. It does not automatically install, restart, deploy, or activate those +changes. Task follow-up, revise, and scrap actions must resolve an actual target; +archive fallback uses the existing permission-gated desktop worker. Repository +Git operations have their own permission and per-action confirmation rules. + +## Acceptance gates + +- Run relevant regression tests, `pnpm typecheck`, and `pnpm build`. +- A desktop integration milestone requires a real create/follow/result round + trip in the supported app; mocked transport tests are insufficient. +- A voice milestone requires a real microphone, audible output, and interruption + trial with a usable provider. Provider rejection is not a successful voice test. +- Loading a source change and observing the changed behavior are separate gates. +- Record only a concise behavior summary, validation, limits, and next action. + Keep raw runtime logs and conversation content out of this document. + +Prior local development verified that an early provider failure can reach the +notch recovery UI. Successful acoustic acceptance of the final combined branch, +full browser benchmark completion, provider billing reconciliation, signed +packages, and Windows/Linux behavior remain unverified for this submission. +No live milestone is marked complete by the automated submission checks. diff --git a/docs/USAGE_AND_COSTS.md b/docs/USAGE_AND_COSTS.md new file mode 100644 index 0000000..8c6b9d2 --- /dev/null +++ b/docs/USAGE_AND_COSTS.md @@ -0,0 +1,91 @@ +# Usage and costs + +Dex keeps a local accounting journal at `userData/usage/ledger.jsonl`, separate +from rotating interaction diagnostics. Tracking starts with the first request +made by a build containing this feature; old diagnostic logs and account bills +are not imported. + +## What the user sees + +The main window and notch display **Since launch** and **Today**. Clicking either +opens Settings → Usage & costs. That page shows launch, local-calendar day, +local-calendar month and all tracked totals; provider/model and activity +breakdowns; and paginated, expandable request history. Connections are grouped +under one provider heading with activity rows and a known-cost subtotal. Input +transcription without billing data says **Cost not reported** and is explicitly +excluded from that subtotal. Voice reconnects do not +reset the launch total. Relaunching Dex does, while historical totals persist. + +- `≈` means at least one included cost is an estimate. +- `+` means additional requests are pending or have no dollar cost available. +- An entirely unpriced total says **Cost unknown**, not `$0.00`. +- Small nonzero costs display `<$0.01`; request details retain six decimal places. +- Storage/read failures remain visible so totals are not presented as complete. + +## Coverage + +| Request path | Usage and pricing | +| --- | --- | +| Pipeline chat and delegated task steps | AI SDK per-step tokens, including cache reads/writes. Gateway-reported `providerMetadata.gateway.cost` when present, otherwise a model list-rate estimate. | +| Screen observation / screen detective | Same accounting around the separate vision request. | +| Realtime voice | Every response is accounted before speech suppression, using original provider response usage. Text and audio have separate rates; cache modality details are required to price cached audio. | +| Realtime input transcription | Separate unpriced entry because the current gateway integration does not identify the billed transcription model/cost. | +| Cloud transcription | WAV chunk parsing for duration; published approximate OpenAI per-minute rates for recognized models. | +| ElevenLabs | Character count and response `character-cost` credits; dollars unavailable without plan/allowance reconciliation. | +| Tavily basic search | One request / one credit on success; dollars unavailable without plan/allowance reconciliation. | +| Apple model | No API charge. | +| Local wake/STT and system voice | No paid API requests; not recorded as individual zero-cost events. | +| External agents / integrations running their own work | Their downstream usage is not exposed by this ledger. Subscription/account usage remains separate. | + +The history groups pipeline steps by one invocation and realtime responses by +one voice connection. It does not yet stitch separately requested transcription, +TTS and delegated invocations into a single end-to-end user-command bill. + +Interrupted, failed or crashed requests without final usage are **unpriced**, +not assumed free. SDK-internal retries are included in the logical request; +unreported usage from failed attempts cannot be reconstructed. Provider invoices +can differ because of subscriptions, free credits, routing, long contexts, +service tiers, discounts, retries, taxes, or other apps using the same account. +Budget enforcement, account reconciliation and historical imports are not part +of this first version. + +## Implementation + +- `src/main/usage/ledger.ts`: append-only pending/final journal records, replay, + deduplication, crash recovery, rollups and bounded history pages. Main process + owns all writes. The renderer cannot insert charges or send prices. +- `src/main/usage/pricing.ts`: allowlisted numeric counters, token/audio pricing, + gateway-reported cost parsing and WAV duration parsing. +- `src/main/usage/catalog.json`: standard pricing snapshot from the public + [Gateway models API](https://ai-gateway.vercel.sh/v1/models), retrieved September + 8, 2026. Each priced record saves the actual rates and snapshot date used. + Custom/unlisted models remain unpriced. Updating this snapshot affects future + records only. +- GPT Realtime 2 cached audio uses the separately verified + [OpenAI model rate](https://developers.openai.com/api/docs/models/gpt-realtime-2). + Transcription estimates use [OpenAI pricing](https://developers.openai.com/api/docs/pricing). +- [ElevenLabs response metadata](https://elevenlabs.io/docs/api-reference/introduction/) + and [Tavily search credits](https://docs.tavily.com/documentation/api-reference/endpoint/search) + supply usage quantities, not invented subscription allocations. +- `usage:summary`, `usage:history`, `usage:changed`: typed preload bridge, shared + across windows. Changes are coalesced for 250ms; idle views refresh every 30s + for local date boundaries. + +No prompts, transcripts, audio, screenshots, API keys or provider response bodies +are retained. Journal files are created with mode 0600 in a 0700 directory. +Totals currently replay the complete journal in memory; history responses are +limited to 100 requests. A future high-volume version should add indexed storage. + +## Verification + +- `pnpm exec tsx --test scripts/test-usage.ts`: pricing, cache/audio accounting, + nonzero/unknown distinctions, persistence, duplicate completion, crash/torn-tail + recovery, local date boundaries and concurrent group interruption. +- `pnpm build` then `node scripts/test-usage-desktop.mjs`: real Electron windows, + built preload and renderer, isolated ledger; verifies empty → pending → priced + updates, unpriced warnings, notch navigation and meter containment. Captures + Settings/notch screenshots to a temporary directory. Uses synthetic billing + fixtures, no credentials, network requests, mic or changes to user data. +- A live provider accounting check was attempted on September 8, 2026. The + separate test process could not decrypt the app's OS-protected saved keys, so + no live provider request was made. Invoice agreement remains unverified. diff --git a/docs/VOICE_TURN_CONFIRMATION.md b/docs/VOICE_TURN_CONFIRMATION.md new file mode 100644 index 0000000..1337d34 --- /dev/null +++ b/docs/VOICE_TURN_CONFIRMATION.md @@ -0,0 +1,35 @@ +# Confirmed spoken turns + +OpenAI realtime sessions use semantic VAD with `create_response: false` and +`interrupt_response: false`. `confirmedTurnOptions` supplies the full provider +`audio` object because the gateway's OpenAI adapter merges provider options +shallowly. PCM rate, output voice, and English input transcription are preserved. + +Tentative speech-started/stopped events are logged but do not flush playback, +cancel generation, advance the tool epoch, or close the session. A matching, +nonempty final transcript confirms the next user turn. Main then interrupts +playback/generation and asks ResponseCoordinator for one response after the +previous generation finishes. Empty or timed-out transcriptions are discarded; +late or superseded transcripts cannot authorize a newer turn. Microphone frames +continue flowing exactly once; local speech detection remains diagnostic-only. + +This conservative policy adds transcription latency to voice interruptions. +Stop and the global interrupt shortcut still act immediately. Other providers +retain their existing server turn detection behavior. + +Validation on 2026-09-08: 29 targeted tests passed, TypeScript checking passed, +and Electron production build passed. Host tests reproduce an empty speech +event during a valid reply, preserve a delegated research tool through noise, +and verify that a real confirmed interruption waits for cancellation completion. +The read-only Electron probe (`scripts/probe-confirmed-turns.cjs`) connected to +the configured provider and observed both turn-detection flags acknowledged as +false. It sends no microphone audio or user text. A live microphone interaction +has not yet verified the acoustic behavior of this revision. + +Official semantics: https://developers.openai.com/api/docs/guides/realtime-vad + +Isolated non-command vocalizations (for example Hmm, um, uh, erm) are now ignored +while playback, generation, or a current delegated task is active. This check +runs before response cancellation and does not change the tool epoch. Real +short commands, yes/no, and longer corrections remain eligible. Diagnostics +record incidental-speech-ignored without claiming what acoustically caused it. diff --git a/electron.vite.config.ts b/electron.vite.config.ts index 7b813a4..6882064 100644 --- a/electron.vite.config.ts +++ b/electron.vite.config.ts @@ -1,8 +1,21 @@ +import { createRequire } from "node:module"; +import { dirname } from "node:path"; +import { readFileSync } from "node:fs"; import { resolve } from "node:path"; import { defineConfig, externalizeDepsPlugin } from "electron-vite"; import react from "@vitejs/plugin-react"; import tailwindcss from "@tailwindcss/vite"; +const require = createRequire(import.meta.url); +const vadDir = dirname(require.resolve("@ricky0123/vad-web")); +const ortDir = dirname(createRequire(require.resolve("@ricky0123/vad-web")).resolve("onnxruntime-web/wasm")); +const vadAssets = new Map([ + ["silero_vad_v5.onnx", resolve(vadDir, "silero_vad_v5.onnx")], + ["vad.worklet.bundle.min.js", resolve(vadDir, "vad.worklet.bundle.min.js")], + ["ort-wasm-simd-threaded.mjs", resolve(ortDir, "ort-wasm-simd-threaded.mjs")], + ["ort-wasm-simd-threaded.wasm", resolve(ortDir, "ort-wasm-simd-threaded.wasm")], +]); + // electron-vite auto-detects entry points from the conventional locations: // src/main/index.ts · src/preload/index.ts · src/renderer/index.html export default defineConfig({ @@ -11,6 +24,14 @@ export default defineConfig({ }, preload: { plugins: [externalizeDepsPlugin()], + build: { + rollupOptions: { + input: { + index: resolve(__dirname, "src/preload/index.ts"), + widget: resolve(__dirname, "src/preload/widget.ts"), + }, + }, + }, }, renderer: { resolve: { @@ -31,6 +52,20 @@ export default defineConfig({ port: 5173, strictPort: true, }, - plugins: [react(), tailwindcss()], + plugins: [react(), tailwindcss(), { + name: "local-speech-detector-assets", + generateBundle() { + for (const [name, path] of vadAssets) this.emitFile({ type: "asset", fileName: `vad/${name}`, source: readFileSync(path) }); + }, + configureServer(server) { + server.middlewares.use((req, res, next) => { + const name = req.url?.split("?")[0]?.replace(/^\/vad\//, ""); + const path = name && vadAssets.get(name); + if (!path || !req.url?.startsWith("/vad/")) return next(); + res.setHeader("Content-Type", name.endsWith(".wasm") ? "application/wasm" : name.endsWith(".onnx") ? "application/octet-stream" : "text/javascript"); + res.end(readFileSync(path)); + }); + }, + }], }, }); diff --git a/package.json b/package.json index 0116119..268b770 100644 --- a/package.json +++ b/package.json @@ -14,19 +14,30 @@ }, "main": "./out/main/index.js", "scripts": { - "dev": "electron-vite dev", - "build": "electron-vite build", + "benchmark:browser": "tsx scripts/browser-benchmark.ts", + "test:browser-benchmark": "tsx --test scripts/test-browser-benchmark.ts", + "dev": "node scripts/build-window-control.mjs && electron-vite dev", + "build": "electron-vite build && node scripts/build-window-control.mjs", "start": "electron-vite preview", "typecheck": "tsc --noEmit", "smoke:chat": "tsx scripts/smoke-chat.ts", "smoke:realtime": "tsx scripts/smoke-realtime.ts", - "dist": "electron-vite build && electron-builder", - "release": "electron-vite build && electron-builder --publish always", + "dist": "pnpm build && electron-builder", + "release": "pnpm build && electron-builder --publish always", "cut:patch": "pnpm version patch -m \"Release v%s\" && git push --follow-tags", "cut:minor": "pnpm version minor -m \"Release v%s\" && git push --follow-tags", "cut:major": "pnpm version major -m \"Release v%s\" && git push --follow-tags", "notes:preview": "tsx scripts/generate-release-notes.ts HEAD \"$(git tag --sort=-v:refname | head -1)\"", - "review:pr": "tsx scripts/review-pr.ts origin/main HEAD" + "review:pr": "tsx scripts/review-pr.ts origin/main HEAD", + "diagnose": "node scripts/diagnose.mjs", + "dex:doctor": "tsx scripts/doctor.ts", + "test:voice-turns": "tsx --test scripts/test-desktop-execution.ts scripts/test-archive-delegation.ts scripts/test-new-session-command.ts scripts/test-incidental-speech.ts scripts/test-playback-acknowledgment.ts scripts/test-confirmed-speech.ts scripts/test-playback-speech-evidence.ts scripts/test-playback-turn-policy.ts scripts/test-response-coordinator.ts scripts/test-spoken-turn-host.ts scripts/test-realtime-feedback.ts", + "test:bridge-history": "tsx --test scripts/test-codex-history-integration.ts", + "test:self-enhancement": "tsx --test scripts/test-self-enhancement.ts scripts/test-enhancement-permissions.ts", + "test:maintenance": "tsx --test scripts/test-diagnostic-report.ts scripts/test-local-agents.ts scripts/test-diagnostic-runtime.ts scripts/test-local-agent-context.ts scripts/test-desktop-bridge.ts", + "diagnose:latency": "tsx scripts/latency-report.ts", + "test:latency": "tsx --test scripts/test-latency-summary.ts", + "test": "node scripts/run-tests.mjs" }, "dependencies": { "@ai-sdk/anthropic": "^3.0.86", @@ -38,11 +49,13 @@ "@meridius-labs/apple-on-device-ai": "^1.6.2", "@nut-tree-fork/nut-js": "^4.2.6", "@picovoice/web-voice-processor": "^4.0.10", + "@ricky0123/vad-web": "^0.0.30", "ai": "^6.0.193", "dotenv": "^17.4.2", "electron-log": "^5.4.3", "electron-updater": "^6.6.2", "vosk-browser": "^0.0.8", + "ws": "^8.21.3", "zod": "^4.4.3" }, "devDependencies": { @@ -54,6 +67,7 @@ "@types/node": "^20", "@types/react": "^19", "@types/react-dom": "^19", + "@types/ws": "^8.18.1", "@vitejs/plugin-react": "^5.2.0", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", @@ -62,6 +76,7 @@ "electron-vite": "^5.0.0", "lucide-react": "^1.21.0", "motion": "^12.42.2", + "node-api-headers": "^1.9.0", "react": "19.2.4", "react-dom": "19.2.4", "tailwind-merge": "^3.6.0", @@ -90,7 +105,8 @@ "package.json" ], "asarUnpack": [ - "node_modules/@nut-tree-fork/**/*" + "node_modules/@nut-tree-fork/**/*", + "out/native/*.node" ], "mac": { "target": [ @@ -118,6 +134,7 @@ "notarize": true, "extendInfo": { "NSMicrophoneUsageDescription": "OpenDex listens to your voice to run agentic commands.", + "NSAudioCaptureUsageDescription": "OpenDex records system audio, including its spoken replies, when you start an interaction recording.", "NSCameraUsageDescription": "OpenDex may capture the screen to see and control your desktop." } }, diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 44a6a8f..fe1fbf4 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -35,6 +35,9 @@ importers: '@picovoice/web-voice-processor': specifier: ^4.0.10 version: 4.0.10 + '@ricky0123/vad-web': + specifier: ^0.0.30 + version: 0.0.30 ai: specifier: ^6.0.193 version: 6.0.193(zod@4.4.3) @@ -50,6 +53,9 @@ importers: vosk-browser: specifier: ^0.0.8 version: 0.0.8 + ws: + specifier: ^8.21.3 + version: 8.21.3 zod: specifier: ^4.4.3 version: 4.4.3 @@ -78,6 +84,9 @@ importers: '@types/react-dom': specifier: ^19 version: 19.2.3(@types/react@19.2.15) + '@types/ws': + specifier: ^8.18.1 + version: 8.18.1 '@vitejs/plugin-react': specifier: ^5.2.0 version: 5.2.0(vite@7.3.5(@types/node@20.19.41)(jiti@2.7.0)(lightningcss@1.32.0)(tsx@4.22.4)) @@ -102,6 +111,9 @@ importers: motion: specifier: ^12.42.2 version: 12.42.2(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + node-api-headers: + specifier: ^1.9.0 + version: 1.9.0 react: specifier: 19.2.4 version: 19.2.4 @@ -1534,6 +1546,9 @@ packages: '@radix-ui/rect@1.1.2': resolution: {integrity: sha512-xnXE7wG13PI+cxieVssYXlQJuYVRhH9NBoxt3KNwzghDIA69GMm7d4wXRouHIYjE+KvS6U/MsMO73NdS2MH9ZA==} + '@ricky0123/vad-web@0.0.30': + resolution: {integrity: sha512-cJyYrh4YeeUBJcbR9Bic/bFDyB9qBkAepvpuWM3vLxnAi7bC3VHzf51UeNdT+OtY4D7MLAgV8iJMc4z41ZnaWg==} + '@rolldown/pluginutils@1.0.0-rc.3': resolution: {integrity: sha512-eybk3TjzzzV97Dlj5c+XrBFW57eTNhzod66y9HrBlzJ6NsCrWCp/2kaPS3K9wJmurBC0Tdw4yPjXKZqlznim3Q==} @@ -1828,6 +1843,9 @@ packages: '@types/verror@1.10.11': resolution: {integrity: sha512-RlDm9K7+o5stv0Co8i8ZRGxDbrTxhJtgjqjFyVh/tXQyl/rYtTKlnTvZ88oSTeYREWurwx20Js4kTuKCsFkUtg==} + '@types/ws@8.18.1': + resolution: {integrity: sha512-ThVF6DCVhA8kUGy+aazFQ4kXQ7E1Ty7A3ypFOe0IcJV8O/M511G99AW24irKrW56Wt44yG9+ij8FaqoBGkuBXg==} + '@types/yauzl@2.10.3': resolution: {integrity: sha512-oJoftv0LSuaDZE3Le4DbKX+KS9G36NzOeSap90UIK0yMA/NhKJhqlSGtNDORNRaIbQfzjXDrQa0ytJ6mNRGz/Q==} @@ -2868,6 +2886,9 @@ packages: node-addon-api@5.0.0: resolution: {integrity: sha512-CvkDw2OEnme7ybCykJpVcKH+uAOLV2qLqiyla128dN9TkEWfrYmxG6C2boDe5KcNQqZF3orkqzGgOMvZ/JNekA==} + node-api-headers@1.9.0: + resolution: {integrity: sha512-2oNILP4jXwRB4ywnYKjVk1YyJ96n2D4EOVJO6S3oYZ5PtbJrw3Yt9TpAuX3nBLMuzn74rnfGQrv13pS9vC+YiA==} + node-api-version@0.2.1: resolution: {integrity: sha512-2xP/IGGMmmSQpI1+O/k72jF/ykvZ89JeuKX3TLJAYPDVLUalrshrLHkeVcCCZqG/eEa635cr8IBYzgnDvM2O8Q==} @@ -2921,6 +2942,9 @@ packages: onnxruntime-common@1.24.3: resolution: {integrity: sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==} + onnxruntime-common@1.29.0: + resolution: {integrity: sha512-/F63/e2VJoaVXGGNu6S5QH7jivBThGO95OzAVXXQ8hTta/b1QxI8udHa6cI3+3mAb5WWIIaMMwfZw01oivjJ1g==} + onnxruntime-node@1.24.3: resolution: {integrity: sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==} os: [win32, darwin, linux] @@ -2928,6 +2952,9 @@ packages: onnxruntime-web@1.26.0-dev.20260416-b7804b056c: resolution: {integrity: sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw==} + onnxruntime-web@1.29.0: + resolution: {integrity: sha512-LuQlpX6MFLJZu756erwUeb1mNfoJGbs1kzDwJGNlf5RvfYMdqhcY3vNpDPK40CUV2HoWTkIj+uS0o36GFHjeYw==} + p-cancelable@2.1.1: resolution: {integrity: sha512-BZOr3nRQHOntUjTrH8+Lh54smKHoHyur8We1V8DSMVrl5A2malOOwuJRnKRDjSnkoeBh4at6BwEnb5I7Jl31wg==} engines: {node: '>=8'} @@ -3497,8 +3524,8 @@ packages: wrappy@1.0.2: resolution: {integrity: sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==} - ws@8.21.0: - resolution: {integrity: sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g==} + ws@8.21.3: + resolution: {integrity: sha512-201TZ/kPWxoPr/OKWjquZR1SWKXcvxdH+e1xrx89b3YbmzLMFCLfnaG1HFIgWzJOEWZ7MvpK++odZufgYR50Rw==} engines: {node: '>=10.0.0'} peerDependencies: bufferutil: ^4.0.1 @@ -3881,7 +3908,7 @@ snapshots: dependencies: command-exists: 1.2.9 node-fetch: 2.7.0 - ws: 8.21.0 + ws: 8.21.3 transitivePeerDependencies: - bufferutil - encoding @@ -4843,6 +4870,10 @@ snapshots: '@radix-ui/rect@1.1.2': {} + '@ricky0123/vad-web@0.0.30': + dependencies: + onnxruntime-web: 1.29.0 + '@rolldown/pluginutils@1.0.0-rc.3': {} '@rollup/rollup-android-arm-eabi@4.61.0': @@ -5077,6 +5108,10 @@ snapshots: '@types/verror@1.10.11': optional: true + '@types/ws@8.18.1': + dependencies: + '@types/node': 20.19.41 + '@types/yauzl@2.10.3': dependencies: '@types/node': 20.19.41 @@ -6248,6 +6283,8 @@ snapshots: node-addon-api@5.0.0: optional: true + node-api-headers@1.9.0: {} + node-api-version@0.2.1: dependencies: semver: 7.8.1 @@ -6295,6 +6332,8 @@ snapshots: onnxruntime-common@1.24.3: {} + onnxruntime-common@1.29.0: {} + onnxruntime-node@1.24.3: dependencies: adm-zip: 0.5.17 @@ -6310,6 +6349,15 @@ snapshots: platform: 1.3.6 protobufjs: 7.6.2 + onnxruntime-web@1.29.0: + dependencies: + flatbuffers: 25.9.23 + guid-typescript: 1.0.9 + long: 5.3.2 + onnxruntime-common: 1.29.0 + platform: 1.3.6 + protobufjs: 7.6.2 + p-cancelable@2.1.1: {} p-finally@1.0.0: {} @@ -6857,7 +6905,7 @@ snapshots: wrappy@1.0.2: {} - ws@8.21.0: {} + ws@8.21.3: {} xhr@2.6.0: dependencies: diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index 581a9d5..cc26c4c 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -1,3 +1,6 @@ +packages: + - "." + ignoredBuiltDependencies: - sharp - unrs-resolver diff --git a/scripts/browser-benchmark.ts b/scripts/browser-benchmark.ts new file mode 100644 index 0000000..2ebb2ff --- /dev/null +++ b/scripts/browser-benchmark.ts @@ -0,0 +1,22 @@ +import { readFile } from 'node:fs/promises'; +import { resolve } from 'node:path'; +import { suiteSchema, type Report } from './browser-benchmark/core'; +import { startBenchmarkServer } from '../src/main/benchmarks/server'; + +async function main() { + const args = process.argv.slice(2); + function option(name: string, fallback?: string) { const i = args.indexOf(name); if (i < 0) return fallback; if (!args[i + 1] || args[i + 1].startsWith('--')) throw Error(`Missing value for ${name}`); return args[i + 1]; } + for (let i = 0; i < args.length; i += 2) if (!['--suite', '--out', '--baseline', '--environment', '--port'].includes(args[i])) throw Error(`Unknown option ${args[i]}`); + const suite = suiteSchema.parse(JSON.parse(await readFile(resolve(option('--suite', 'benchmarks/browser/golden.json')!), 'utf8'))); + const environment = option('--environment'); + if (!environment?.trim() || environment.length > 2000) throw Error('--environment is required: record revision, model, browser/version, viewport, display scale, OS, voice mode, grants and cold/warm state'); + const out = resolve(option('--out', `/tmp/dex-browser-${Date.now()}`)!); + const baselinePath = option('--baseline'); + const baseline: Report | undefined = baselinePath ? JSON.parse(await readFile(resolve(baselinePath), 'utf8')) : undefined; + const server = await startBenchmarkServer({ suite, environment, out, baseline, port: Number(option('--port', '0')), onComplete: async () => { console.log(`Benchmark complete. Reports: ${out}`); } }); + // The standalone CLI stays alive until the operator stops it. + const keepAlive = setInterval(() => {}, 1000); + console.log(`Ask Dex: Open ${server.url} and complete all ${suite.scenarios.length} browser benchmark scenarios using only screenshots and computer controls. Click Start scenario for each task, follow its instructions, and stop at Benchmark complete. Keep normal permission gates.\nReports: ${out}\nCtrl+C finalizes unfinished scenarios.`); + process.once('SIGINT', () => { clearInterval(keepAlive); void server.close().then(() => console.log(`Finalized reports: ${out}`)).catch(error => { console.error(error); process.exitCode = 1; }); }); +} +void main().catch(error => { console.error(error instanceof Error ? error.message : error); process.exitCode = 1; }); diff --git a/scripts/browser-benchmark/core.ts b/scripts/browser-benchmark/core.ts new file mode 100644 index 0000000..945dca1 --- /dev/null +++ b/scripts/browser-benchmark/core.ts @@ -0,0 +1 @@ +export * from '../../src/main/benchmarks/core'; diff --git a/scripts/browser-benchmark/page.ts b/scripts/browser-benchmark/page.ts new file mode 100644 index 0000000..f9428a8 --- /dev/null +++ b/scripts/browser-benchmark/page.ts @@ -0,0 +1 @@ +export * from '../../src/main/benchmarks/page'; diff --git a/scripts/build-window-control.mjs b/scripts/build-window-control.mjs new file mode 100644 index 0000000..c8a6286 --- /dev/null +++ b/scripts/build-window-control.mjs @@ -0,0 +1,14 @@ +import { createRequire } from 'node:module'; +import { mkdirSync } from 'node:fs'; +import { dirname, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { execFileSync } from 'node:child_process'; +if (process.platform === 'darwin') { + const require = createRequire(import.meta.url); + const root = resolve(dirname(fileURLToPath(import.meta.url)), '..'); + const headers = require('node-api-headers').include_dir; + const output = resolve(root, 'out/native'); + mkdirSync(output, { recursive: true }); + execFileSync('xcrun', ['clang', '-shared', '-fPIC', '-fobjc-arc', '-undefined', 'dynamic_lookup', '-DNAPI_VERSION=8', '-I', headers, + '-framework', 'Cocoa', '-framework', 'ApplicationServices', resolve(root, 'src/native/window-control.m'), '-o', resolve(output, 'window-control.node')], { stdio: 'inherit' }); +} diff --git a/scripts/diagnose.mjs b/scripts/diagnose.mjs new file mode 100644 index 0000000..6514f6c --- /dev/null +++ b/scripts/diagnose.mjs @@ -0,0 +1,40 @@ +import { readFileSync } from 'node:fs'; +import { homedir } from 'node:os'; +import { join } from 'node:path'; +const dir = process.env.OPENDEX_DIAGNOSTICS_DIR ?? join(homedir(), 'Library/Application Support/opendex/diagnostics'); +const lastIndex = process.argv.indexOf('--last'); +const count = Math.max(1, Math.min(20, Number(process.argv[lastIndex + 1]) || 1)); +const events = []; +for (const suffix of ['.2', '.1', '']) { + try { + for (const line of readFileSync(join(dir, 'interactions.jsonl' + suffix), 'utf8').split('\n')) { + try { const e = JSON.parse(line); if (Number.isFinite(e.at)) events.push(e); } catch {} + } + } catch {} +} +const starts = events.filter(e => e.event === 'session-start'); +const sessions = starts.slice(-count).map(start => { + const next = starts.find(e => e.at > start.at); + const own = events.filter(e => e.sessionId === start.sessionId); + const end = own.find(e => e.event === 'session-end'); + const until = end?.at ?? next?.at ?? Date.now(); + const related = events.filter(e => e.sessionId === start.sessionId || + (!e.sessionId && e.at >= start.at && e.at <= until && ['timing','desktop-request','desktop-result'].includes(e.event))); + const calls = own.filter(e => e.event === 'tool-call'); + return { + sessionId: start.sessionId, started: new Date(start.at).toISOString(), model: start.model, + wakeScreen: start.wakeScreen, ended: Boolean(end || own.find(e => e.event === 'client-close')), + transcript: own.filter(e => ['user-transcript','assistant-transcript'].includes(e.event)) + .map(e => ({ seconds: +((e.at-start.at)/1000).toFixed(3), role: e.event === 'user-transcript' ? 'user':'assistant', text: e.text, status: e.status, source: e.source, responseId: e.responseId })), + tools: calls.map(e => { + const done = own.find(r => r.event === 'tool-result' && r.callId === e.callId); + return { tool: e.tool, callId: e.callId, parentCallId: e.parentCallId, targetId: done?.targetId, ms: done ? done.at-e.at : null, status: done ? done.failed ? 'failed' : done.outcome ?? 'returned' : 'pending-or-interrupted' }; + }), + errors: own.filter(e => e.event === 'response-error'), + cancellations: own.filter(e => e.event === 'response-done' && e.status !== 'completed'), + playbackInterruptions: own.filter(e => e.event === 'playback-interrupted'), + timeline: related.map(({at,...e}) => ({ seconds: +((at-start.at)/1000).toFixed(3), ...e })), + }; +}); +console.log(JSON.stringify({ directory: dir, note: 'Generated transcripts are not proof of audible playback. Timings without a sessionId are correlated by time, not guaranteed causality.', sessions, + ...(sessions.length ? {} : { message: 'No recorded voice sessions yet.', recentEvents: events.slice(-20) }) }, null, 2)); diff --git a/scripts/doctor.ts b/scripts/doctor.ts new file mode 100644 index 0000000..0e3e588 --- /dev/null +++ b/scripts/doctor.ts @@ -0,0 +1,24 @@ +import { homedir } from "node:os"; +import { join } from "node:path"; +import { readDiagnosticReport } from "../src/main/diagnostics/report"; +import { listDesktopTasks, bridgeFailure } from "../src/main/maintenance/desktop-tasks"; + +async function main() { + const args = process.argv.slice(2); + const value = (name: string) => { const i = args.indexOf(name); return i < 0 ? undefined : args[i + 1]; }; + if (args.includes("--help")) { + console.log("Dex Doctor: --last 3 [--codex] [--cwd /project/path]. Reads aggregate history; --codex inspects the local desktop bridge. Never starts or messages agents. Override the history directory with OPENDEX_DIAGNOSTICS_DIR."); + return; + } + const base = process.platform === "darwin" ? join(homedir(), "Library", "Application Support") + : process.platform === "win32" ? process.env.APPDATA ?? join(homedir(), "AppData", "Roaming") + : process.env.XDG_CONFIG_HOME ?? join(homedir(), ".config"); + const directory = process.env.OPENDEX_DIAGNOSTICS_DIR ?? join(base, "opendex", "diagnostics"); + const history = await readDiagnosticReport(directory, Number(value("--last") ?? 3)); + const localAgents = args.includes("--codex") ? await listDesktopTasks({ cwd: value("--cwd") ?? process.cwd() }).catch(bridgeFailure) : undefined; + console.log(JSON.stringify({ history, ...(localAgents ? { localAgents } : {}), note: "CLI reports retained history only; live runtime inspection is available inside Dex." }, null, 2)); +} +void main().catch(() => { + console.error("Dex Doctor could not complete inspection. Check local file access and connector availability."); + process.exitCode = 1; +}); diff --git a/scripts/latency-report.ts b/scripts/latency-report.ts new file mode 100644 index 0000000..bf38b81 --- /dev/null +++ b/scripts/latency-report.ts @@ -0,0 +1,23 @@ +import { homedir, tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { mkdtemp } from 'node:fs/promises'; +import { createLatencyStore, createSpanObserver, latencyReport, readLatencySamples } from '../src/main/diagnostics/latency-summary'; + +async function main() { + const sample = process.argv.includes('--sample'); + const directory = sample ? await mkdtemp(join(tmpdir(), 'opendex-latency-')) : process.env.OPENDEX_DIAGNOSTICS_DIR ?? join( + process.platform === 'darwin' ? join(homedir(), 'Library/Application Support') : process.platform === 'win32' ? process.env.APPDATA ?? join(homedir(), 'AppData/Roaming') : process.env.XDG_CONFIG_HOME ?? join(homedir(), '.config'), + 'opendex', 'diagnostics'); + if (sample) { + const store = createLatencyStore(directory); + const observe = createSpanObserver('desktop-agent', store.add); + observe('start', 0); + observe('first-text', 1200); + observe('tool-call', 1250, { call: 'fixture' }); + observe('tool-result', 1400, { call: 'fixture' }); + await store.flush(); + } + console.log(JSON.stringify({ source: sample ? 'synthetic sample (not live acceptance)' : 'local timing summary', + ...latencyReport(readLatencySamples(directory)) }, null, 2)); +} +void main(); diff --git a/scripts/preview-research.tsx b/scripts/preview-research.tsx new file mode 100644 index 0000000..ccd6541 --- /dev/null +++ b/scripts/preview-research.tsx @@ -0,0 +1,24 @@ +// Static visual QA fixture, not a live research session. Writes only to /tmp. +import { renderToStaticMarkup } from "react-dom/server"; +import React from "react"; +import { copyFileSync, mkdirSync, readdirSync, writeFileSync } from "node:fs"; +import { ResearchCard } from "../src/skills/research/view"; +import type { ResearchUpdate } from "../src/skills/research/schema"; +const data: ResearchUpdate = { + topic: "Where do the energy estimates disagree?", stage: "comparing", milestone: "finding", + update: "The reports cover different years. I’m checking their definitions before comparing the totals.", + plan: [{ title: "Find primary datasets and their methodology", status: "done" }, { title: "Compare dates, definitions, and assumptions", status: "active" }, { title: "Explain the agreement and remaining uncertainty", status: "pending" }], + sources: [ + { title: "National energy statistics", url: "https://example.org/statistics", status: "read", note: "Sample source: includes a defined reporting period and methodology." }, + { title: "Independent analysis of demand", url: "https://example.org/analysis", status: "found", note: "Sample source: still needs reading before its claims can be used." }, + { title: "Historical data archive", url: "https://example.org/archive", status: "unavailable", note: "Sample source: access failed; no findings attributed to it." }, + ], + findings: [{ text: "The reporting period must match before the totals can be compared.", sourceUrls: ["https://example.org/statistics"] }], + openQuestions: ["Do both estimates include the same categories of energy use?"], +}; +const dir = "/tmp/opendex-research-preview"; +mkdirSync(dir, { recursive: true }); +const css = readdirSync("out/renderer/assets").find(file => file.endsWith(".css"))!; +copyFileSync(`out/renderer/assets/${css}`, `${dir}/style.css`); +writeFileSync(`${dir}/index.html`, `Dex research layout preview

Layout preview · sample data, not research findings

Main view

${renderToStaticMarkup()}

Notch view

${renderToStaticMarkup()}
`); +console.log(`${dir}/index.html`); diff --git a/scripts/probe-confirmed-turns.cjs b/scripts/probe-confirmed-turns.cjs new file mode 100644 index 0000000..7957fb1 --- /dev/null +++ b/scripts/probe-confirmed-turns.cjs @@ -0,0 +1,42 @@ +// Read-only configuration handshake. No audio, screenshots, or user text sent. +const { app } = require('electron'); +const { join } = require('node:path'); +const { homedir } = require('node:os'); +require('tsx/cjs'); +app.setName('opendex'); +app.setPath('userData', join(homedir(), 'Library/Application Support/opendex')); +app.whenReady().then(async () => { + let ws; + const finish = code => { try { ws?.close(); } catch {} app.exit(code); }; + setTimeout(() => { console.error('Configuration probe timed out.'); finish(1); }, 20000); + try { + const { initConfig, getConfig } = require('../src/main/config/store.ts'); + const { confirmedTurnOptions } = require('../src/main/agent/realtime/confirmed-turn-config.ts'); + const { gateway } = require('@ai-sdk/gateway'); + initConfig(); + const cfg = getConfig(); + const model = gateway.experimental_realtime(cfg.realtime.model); + const token = await gateway.experimental_realtime.getToken({ model: cfg.realtime.model }); + const connection = model.getWebSocketConfig(token); + ws = new WebSocket(connection.url, connection.protocols); + ws.addEventListener('open', async () => ws.send(JSON.stringify(await model.serializeClientEvent({ type: 'session-update', config: { + inputAudioFormat: { type: 'audio/pcm', rate: 24000 }, outputAudioFormat: { type: 'audio/pcm', rate: 24000 }, + turnDetection: { type: 'semantic-vad' }, providerOptions: confirmedTurnOptions(cfg.realtime.voice), + } })))); + ws.addEventListener('message', event => { + const raw = JSON.parse(String(event.data)); + const health = model.getHealthCheckResponse?.(raw); + if (health) { ws.send(JSON.stringify(health)); return; } + const parsed = model.parseServerEvent(raw); + for (const e of Array.isArray(parsed) ? parsed : [parsed]) { + if (e.type === 'error') { console.error('Provider rejected the configuration.', { code: e.code }); finish(1); } + if (e.type === 'session-updated') { + const session = e.raw?.session ?? raw.session ?? raw.raw?.session; + const td = session?.audio?.input?.turn_detection ?? session?.turn_detection; + console.log('Provider turn detection:', td ?? 'No acknowledged settings returned'); + finish(td?.create_response === false && td?.interrupt_response === false ? 0 : 1); + } + } + }); + } catch { console.error('Configuration probe failed. No credentials logged.'); finish(1); } +}); diff --git a/scripts/probe-research-start.cjs b/scripts/probe-research-start.cjs new file mode 100644 index 0000000..cf483a3 --- /dev/null +++ b/scripts/probe-research-start.cjs @@ -0,0 +1,53 @@ +// Local read-only latency probe. Tool definitions are supplied without execute. +const { app } = require('electron'); +const { join } = require('node:path'); +const { homedir } = require('node:os'); +require('tsx/cjs'); +app.setName('opendex'); +app.setPath('userData', join(homedir(), 'Library/Application Support/opendex')); +app.whenReady().then(async () => { + try { + const { initConfig, getConfig } = require('../src/main/config/store.ts'); + initConfig(); + const cfg = getConfig(); + const { resolveModel } = require('../src/main/agent/llm/resolve-model.ts'); + const { buildSystemPrompt } = require('../src/main/agent/system-prompt.ts'); + const { buildToolSet, skillSystemPrompts } = require('../src/skills/registry.ts'); + const { streamText } = require('ai'); + const definitions = buildToolSet({ config: cfg, requestPermission: async () => false }); + const toolDefs = Object.fromEntries(Object.entries(definitions).map(([name, t]) => [name, {description:t.description,inputSchema:t.inputSchema}])); + const system = buildSystemPrompt({config:cfg,skillPrompts:skillSystemPrompts(cfg)}); + const model = await resolveModel(cfg); + let modelRequests = 0; + if (typeof model !== 'string') { + const doStream = model.doStream.bind(model); + model.doStream = options => { modelRequests++; return doStream(options); }; + } + const effort = process.argv.find(x => x.startsWith('--effort='))?.split('=')[1]; + const start = Date.now(); const abort = new AbortController(); + const timer = setTimeout(() => abort.abort(), 70000); + console.log(JSON.stringify({provider:cfg.llm.provider,model:cfg.llm.model,effort:effort??'default',systemChars:system.length,tools:Object.keys(toolDefs)})); + if (process.argv.includes('--flow')) { + const { streamChat } = require('../src/main/agent/chat.ts'); + const originalLog = console.log; + console.log = (...args) => { if (String(args[0]).startsWith('[opendex chat]')) return; originalLog(...args); }; + const safeTools = { ...toolDefs, updateResearch: definitions.updateResearch, openUrl: { ...toolDefs.openUrl, execute: async () => ({ error: 'Probe: no browser action executed.' }) } }; + const output = await streamChat({model,system,tools:safeTools,signal:abort.signal, + messages:[{role:'user',content:'Research whether a heat pump makes sense for a home in Chicago. Read credible original sources, compare cold-weather performance and running costs, check incentives, and present cited findings. Update the research record as you work.'}], + onDelta:() => {}, onToolResult:r=>originalLog(JSON.stringify({event:'tool-result',tool:r.toolName,stage:r.output?.stage,error:r.output?.error})), onToolCall: call => { + originalLog(JSON.stringify({event:'tool-call',tool:call.toolName,stage:call.input?.stage,modelRequests,ms:Date.now()-start})); + if(!['updateResearch','startResearch'].includes(call.toolName)) abort.abort(); + }}); + clearTimeout(timer); originalLog(JSON.stringify({done:true,ms:Date.now()-start,final:output.filter(m=>m.role==='assistant').at(-1)?.content})); app.exit(0); return; + } + const result = streamText({model,system,tools:toolDefs,maxRetries:0,abortSignal:abort.signal, + ...(effort ? {providerOptions:{openai:{reasoningEffort:effort}}}:{}), + prompt:'Research whether a heat pump makes sense for a home in Chicago. Read credible original sources, compare cold-weather performance and running costs, check incentives, and present cited findings. Update the research record as you work.'}); + const seen = new Set(); + for await (const part of result.fullStream) { + if (!seen.has(part.type)) { seen.add(part.type); console.log(JSON.stringify({event:part.type,ms:Date.now()-start,...(part.type==='tool-call'?{tool:part.toolName}:{}),...(part.type==='error'?{errorName:part.error?.name,statusCode:part.error?.statusCode}: {})})); } + if (part.type==='text-delta' || part.type==='tool-call') { abort.abort(); break; } + } + clearTimeout(timer); console.log(JSON.stringify({done:true,ms:Date.now()-start})); app.exit(0); + } catch(e) { console.error(JSON.stringify({errorName:e?.name,statusCode:e?.statusCode})); app.exit(1); } +}); diff --git a/scripts/run-tests.mjs b/scripts/run-tests.mjs new file mode 100644 index 0000000..ff97f32 --- /dev/null +++ b/scripts/run-tests.mjs @@ -0,0 +1,20 @@ +// Resolve Electron once before test workers import it concurrently. On a fresh +// install Electron can download its binary on first require; concurrent extraction +// races on the same directory. This does not launch the desktop application. +import { createRequire } from "node:module"; +import { readdirSync } from "node:fs"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; + +const require = createRequire(import.meta.url); +require("electron"); +const root = fileURLToPath(new URL("../", import.meta.url)); +const files = readdirSync(new URL("./", import.meta.url)) + .filter(name => /^test-.*\.ts$/.test(name)) + .sort() + .map(name => `scripts/${name}`); +const result = spawnSync(process.execPath, [require.resolve("tsx/cli"), "--test", ...files], { + cwd: root, stdio: "inherit", +}); +if (result.error) throw result.error; +process.exitCode = result.status ?? 1; diff --git a/scripts/smoke-realtime.ts b/scripts/smoke-realtime.ts index e5baaba..f59bb87 100644 --- a/scripts/smoke-realtime.ts +++ b/scripts/smoke-realtime.ts @@ -1,14 +1,15 @@ -// Standalone smoke test for the gateway realtime codec — mints a token, opens +// Standalone smoke test for either realtime provider — connects, opens // the WebSocket from Node (no Electron, no audio), runs a text round-trip and a // tool-call round-trip, and prints the streamed transcript. -// Usage: `pnpm smoke:realtime [model-id]` (default: openai/gpt-realtime-2) +// Usage: `pnpm smoke:realtime [model-id] [--openai]` (default: gateway, openai/gpt-realtime-2) import { config as loadEnv } from "dotenv"; -import { gateway } from "@ai-sdk/gateway"; +import { connectRealtime } from "../src/main/agent/realtime/connection"; import { z } from "zod"; -loadEnv(); +loadEnv({ quiet: true }); -const modelId = process.argv[2] ?? "openai/gpt-realtime-2"; +const provider = process.argv.includes("--openai") ? "openai" : "gateway"; +const modelId = process.argv.slice(2).find(arg => !arg.startsWith("--")) ?? "openai/gpt-realtime-2"; const TIMEOUT_MS = 60_000; // Mirrors how realtime-tools.ts will flatten a skill tool for the session. @@ -22,18 +23,8 @@ const weatherTool = { }; async function main() { - if (!process.env.AI_GATEWAY_API_KEY) { - console.error("[smoke] FAIL: AI_GATEWAY_API_KEY not set"); - process.exit(1); - } - - console.log(`[smoke] model=${modelId}`); - const { token, url } = await gateway.experimental_realtime.getToken({ model: modelId }); - console.log(`[smoke] token minted, url=${url.split("?")[0]}`); - - const model = gateway.experimental_realtime(modelId); - const config = model.getWebSocketConfig({ token, url }); - const ws = new WebSocket(config.url, config.protocols); + console.log(`[smoke] provider=${provider} model=${modelId}`); + const { codec: model, ws } = await connectRealtime(provider, modelId); const send = async (event: Parameters[0]) => ws.send(JSON.stringify(await model.serializeClientEvent(event))); @@ -69,7 +60,7 @@ async function main() { await send({ type: "response-create" }); }); - ws.addEventListener("message", async (msg) => { + ws.addEventListener("message", async (msg: { data: unknown }) => { const raw = JSON.parse(String(msg.data)); const keepalive = model.getHealthCheckResponse?.(raw); if (keepalive) { @@ -132,8 +123,10 @@ async function main() { } }); - ws.addEventListener("close", (e) => { - console.log(`[smoke] ws closed code=${e.code} reason=${e.reason}`); + ws.addEventListener("close", (e: { code: number }) => { + console.log(`[smoke] ws closed code=${e.code}`); + clearTimeout(timeout); + process.exit(1); }); ws.addEventListener("error", () => { console.error("[smoke] FAIL: ws error"); @@ -142,6 +135,6 @@ async function main() { } main().catch((err) => { - console.error("[smoke] error", err); + console.error("[smoke] error", err instanceof Error ? err.message : "Connection failed"); process.exit(1); }); diff --git a/scripts/test-archive-delegation.ts b/scripts/test-archive-delegation.ts new file mode 100644 index 0000000..3540c60 --- /dev/null +++ b/scripts/test-archive-delegation.ts @@ -0,0 +1,170 @@ +import { test, mock } from 'node:test'; +import assert from 'node:assert/strict'; +import { gateway } from '@ai-sdk/gateway'; +import { startRealtimeSession, endRealtimeSession, sendRealtimeClientMessage, delegatedDesktopJob } from '../src/main/agent/realtime/session-host'; +import { DesktopDelegations } from '../src/main/agent/realtime/desktop-delegations'; +import { realtimeArchiveFallback } from '../src/skills/local-agent-actions/archive-fallback'; +const target = 'b96dc42d-cd8c-47ea-836a-9302a378a397'; +const rejection = { state: 'rejected', failureKind: 'active-writer', taskId: target, + fallback: { state: 'ready_not_performed', taskId: target, task: 'Archive the exact selected task through the permitted desktop workflow.' } }; + +test('only a matching confirmed archive rejection may dispatch computer work', () => { + assert.ok(realtimeArchiveFallback('archiveLocalAgentTask', { taskId: target }, rejection)); + for (const output of [{ ...rejection, state: 'unconfirmed' }, { ...rejection, fallback: { state: 'unavailable' } }, { ...rejection, taskId: 'wrong' }]) + assert.equal(realtimeArchiveFallback('archiveLocalAgentTask', { taskId: target }, output), undefined); + assert.equal(realtimeArchiveFallback('webSearch', { taskId: target }, rejection), undefined); +}); + +test('desktop queue serializes workers, ignores duplicate results, and drops cancelled work', () => { + const dispatched: string[] = []; + const queue = new DesktopDelegations(job => dispatched.push(job.callId)); + const a = { callId: 'a', name: 'run_task', task: 'a' }, b = { callId: 'b', name: 'run_task', task: 'b' }; + queue.enqueue(a); queue.enqueue(a); queue.enqueue(b); + assert.deepEqual(dispatched, ['a']); + assert.equal(queue.beginResult('b'), undefined); + assert.equal(queue.beginResult('a'), a); + assert.equal(queue.beginResult('a'), undefined); + queue.clear(); queue.finish(a); + assert.deepEqual(dispatched, ['a']); + assert.equal(queue.beginResult('a'), undefined); +}); + +test('host dispatches archive fallback without another model decision and verifies before answering', async () => { + const previousSocket = globalThis.WebSocket, previousKey = process.env.AI_GATEWAY_API_KEY; + let socket!: FakeSocket; + class FakeSocket extends EventTarget { + static OPEN = 1; readyState = 1; sent: any[] = []; + constructor() { super(); socket = this; setImmediate(() => this.dispatchEvent(new Event('open'))); } + send(data: string) { this.sent.push(JSON.parse(data)); } + close() { this.readyState = 3; this.dispatchEvent(new Event('close')); } + async receive(value: object) { this.dispatchEvent(new MessageEvent('message', { data: JSON.stringify(value) })); await tick(); } + } + const tick = () => new Promise(resolve => setImmediate(resolve)); + globalThis.WebSocket = FakeSocket as unknown as typeof WebSocket; + process.env.AI_GATEWAY_API_KEY = 'test-only'; + const token = mock.method(gateway.experimental_realtime, 'getToken', async () => ({ token: 'test-only', url: 'wss://example.invalid' })); + const notices: any[] = []; let verified = 0; + const sessionId = 'archive-delegation-test'; + const outputs = () => socket.sent.filter(e => e.type === 'conversation-item-create' && e.item.type === 'function-call-output'); + try { + await startRealtimeSession({ sessionId, model: 'openai/gpt-realtime-2', voice: '', instructions: 'Test', + toolDefs: [{ name: 'run_task', description: '', parameters: {} }], + tools: { archiveLocalAgentTask: { execute: async () => rejection }, verifyLocalAgentTaskArchive: { execute: async () => { verified++; return { state: 'archived', taskId: target }; } } } as never, + transcribesInput: true, notify: n => notices.push(n) }); + sendRealtimeClientMessage(sessionId, { type: 'user-text', text: 'Archive the selected tasks' }); await tick(); + await socket.receive({ type: 'response-created', responseId: 'r' }); + for (const callId of ['a', 'b']) await socket.receive({ type: 'function-call-arguments-done', responseId: 'r', callId, name: 'archiveLocalAgentTask', arguments: JSON.stringify({ taskId: target }) }); + await socket.receive({ type: 'response-done', responseId: 'r', status: 'completed' }); + assert.deepEqual(notices.filter(n => n.type === 'run-task').map(n => n.toolCallId), ['a']); + assert.equal(outputs().length, 0, 'rejection must not return to the model before fallback'); + sendRealtimeClientMessage(sessionId, { type: 'tool-result', toolCallId: 'a', name: 'run_task', output: { result: 'Clicked archive' } }); await tick(); + assert.equal(verified, 1); + assert.equal(outputs()[0].item.name, 'archiveLocalAgentTask'); + assert.equal(JSON.parse(outputs()[0].item.output).state, 'archived'); + assert.deepEqual(notices.filter(n => n.type === 'run-task').map(n => n.toolCallId), ['a', 'b']); + sendRealtimeClientMessage(sessionId, { type: 'tool-result', toolCallId: 'a', name: 'run_task', output: { result: 'duplicate' } }); await tick(); + assert.equal(verified, 1); + const worker = delegatedDesktopJob(sessionId, 'b')!; + sendRealtimeClientMessage(sessionId, { type: 'cancel-response' }); + assert.equal(worker.controller?.signal.aborted, true, 'main aborts the real worker signal'); + sendRealtimeClientMessage(sessionId, { type: 'tool-result', toolCallId: 'b', name: 'run_task', output: { result: 'late' } }); await tick(); + assert.equal(outputs().length, 2, 'only the verified result and explicit cancellation receipt are sent'); + assert.equal(JSON.parse(outputs()[1].item.output).state, 'cancelled'); + } finally { + endRealtimeSession(sessionId); token.mock.restore(); globalThis.WebSocket = previousSocket; + if (previousKey === undefined) delete process.env.AI_GATEWAY_API_KEY; else process.env.AI_GATEWAY_API_KEY = previousKey; + } +}); + +for (const interrupt of ['speech', 'status-then-stop', 'status-then-complete', 'playback-noise-then-stop', 'typed', 'cancel', 'close', 'verification'] as const) test(`${interrupt}: desktop work follows the requested lifecycle`, async () => { + const previousSocket = globalThis.WebSocket, previousKey = process.env.AI_GATEWAY_API_KEY; + let socket!: FakeSocket; + const tick = () => new Promise(resolve => setImmediate(resolve)); + class FakeSocket extends EventTarget { + static OPEN = 1; readyState = 1; sent: any[] = []; + constructor() { super(); socket = this; setImmediate(() => this.dispatchEvent(new Event('open'))); } + send(data: string) { this.sent.push(JSON.parse(data)); } + close() { this.readyState = 3; this.dispatchEvent(new Event('close')); } + async receive(value: object) { this.dispatchEvent(new MessageEvent('message', { data: JSON.stringify(value) })); await tick(); } + } + globalThis.WebSocket = FakeSocket as unknown as typeof WebSocket; + process.env.AI_GATEWAY_API_KEY = 'test-only'; + const token = mock.method(gateway.experimental_realtime, 'getToken', async () => ({ token: 'test-only', url: 'wss://example.invalid' })); + const notices: any[] = [], sessionId = `abort-worker-${interrupt}`; + try { + await startRealtimeSession({ sessionId, model: 'openai/gpt-realtime-2', voice: '', instructions: 'Test', toolDefs: [{ name: 'run_task', description: '', parameters: {} }], tools: { archiveLocalAgentTask: { execute: async () => rejection }, verifyLocalAgentTaskArchive: { execute: async () => ({ state: 'unarchived', taskId: target }) } } as never, transcribesInput: true, notify: n => notices.push(n) }); + sendRealtimeClientMessage(sessionId, { type: 'user-text', text: 'Do the desktop work' }); await tick(); + await socket.receive({ type: 'response-created', responseId: 'r' }); + for (const callId of ['active', 'queued']) await socket.receive({ type: 'function-call-arguments-done', responseId: 'r', callId, name: interrupt === 'verification' ? 'archiveLocalAgentTask' : 'run_task', arguments: interrupt === 'verification' ? JSON.stringify({ taskId: target }) : '{"task":"test"}' }); + await socket.receive({ type: 'response-done', responseId: 'r', status: 'completed' }); + const worker = delegatedDesktopJob(sessionId, 'active')!; + assert.ok(worker); + assert.equal(delegatedDesktopJob(sessionId, 'queued'), undefined, 'queued handoffs cannot start early'); + if (interrupt === 'verification') { + sendRealtimeClientMessage(sessionId, { type: 'tool-result', toolCallId: 'active', name: 'run_task', output: { result: 'Claimed done' } }); + await tick(); + assert.equal(delegatedDesktopJob(sessionId, 'active'), undefined); + assert.deepEqual(notices.filter(n => n.type === 'run-task').map(n => n.toolCallId), ['active']); + assert.equal(notices.find(n => n.type === 'tool-result' && n.result.toolCallId === 'active').result.output.state, 'unarchived'); + assert.equal(notices.find(n => n.type === 'tool-result' && n.result.toolCallId === 'queued').result.output.state, 'cancelled'); + return; + } + if (interrupt === 'status-then-stop' || interrupt === 'status-then-complete') { + const before = notices.length; + for (const [index, transcript] of ['What are you doing?', "Dex, what's going on?", 'Dex was wrong.', "No, I'm asking what's going on."].entries()) { + await socket.receive({ type: 'speech-started', itemId: `status-${index}` }); + await socket.receive({ type: 'speech-stopped', itemId: `status-${index}` }); + await socket.receive({ type: 'input-transcription-completed', itemId: `status-${index}`, transcript }); + } + assert.equal(worker.controller!.signal.aborted, false); + assert.equal(delegatedDesktopJob(sessionId, 'active'), worker); + assert.equal(notices.slice(before).some(n => n.type === 'speech-started'), false); + assert.ok(socket.sent.some(e => e.type === 'response-create' && e.options?.instructions?.includes('has not cancelled'))); + if (interrupt === 'status-then-complete') { + await socket.receive({ type: 'response-created', responseId: 'status-response' }); + await socket.receive({ type: 'response-done', responseId: 'status-response', status: 'completed' }); + sendRealtimeClientMessage(sessionId, { type: 'tool-result', toolCallId: 'active', name: 'run_task', output: { result: 'Verified fixture result' } }); + await tick(); + assert.equal(worker.controller!.signal.aborted, false); + assert.ok(notices.some(n => n.type === 'tool-result' && n.result.output.result === 'Verified fixture result')); + assert.ok(delegatedDesktopJob(sessionId, 'queued')); + return; + } + sendRealtimeClientMessage(sessionId, { type: 'cancel-response' }); + } else if (interrupt === 'playback-noise-then-stop') { + sendRealtimeClientMessage(sessionId, { type: 'diagnostic', event: 'playback-start' }); + for (let i = 0; i < 9; i++) sendRealtimeClientMessage(sessionId, { type: 'audio', chunk: new ArrayBuffer(1536), speechProbability: 0.99 }); + await socket.receive({ type: 'speech-started', itemId: 'noise', raw: { audio_start_ms: 0 } }); + await socket.receive({ type: 'speech-stopped', itemId: 'noise', raw: { audio_end_ms: 288 } }); + await socket.receive({ type: 'input-transcription-completed', itemId: 'noise', transcript: 'left.' }); + assert.equal(worker.controller!.signal.aborted, false, 'playback noise must not cancel the active worker'); + assert.equal(notices.some(n => n.type === 'user-transcript' && n.text === 'left.'), false); + await socket.receive({ type: 'speech-started', itemId: 'stop', raw: { audio_start_ms: 0 } }); + await socket.receive({ type: 'speech-stopped', itemId: 'stop', raw: { audio_end_ms: 288 } }); + await socket.receive({ type: 'input-transcription-completed', itemId: 'stop', transcript: 'Stop.' }); + } else if (interrupt === 'speech') { + await socket.receive({ type: 'speech-started', itemId: 'stop' }); + assert.equal(worker.controller!.signal.aborted, false, 'unconfirmed detection is not authorization to interrupt'); + await socket.receive({ type: 'speech-stopped', itemId: 'stop' }); + await socket.receive({ type: 'input-transcription-completed', itemId: 'stop', transcript: 'Dex stop.' }); + } else if (interrupt === 'typed') sendRealtimeClientMessage(sessionId, { type: 'user-text', text: 'Stop' }); + else if (interrupt === 'cancel') sendRealtimeClientMessage(sessionId, { type: 'cancel-response' }); + else endRealtimeSession(sessionId); + await tick(); + assert.equal(worker.controller!.signal.aborted, true); + let clicks = 0; + await assert.rejects((async () => { worker.controller!.signal.throwIfAborted(); clicks++; })()); + assert.equal(clicks, 0); + sendRealtimeClientMessage(sessionId, { type: 'tool-result', toolCallId: 'active', name: 'run_task', output: { result: 'late' } }); await tick(); + assert.equal(delegatedDesktopJob(sessionId, 'active'), undefined); + assert.deepEqual(notices.filter(n => n.type === 'run-task').map(n => n.toolCallId), ['active']); + assert.equal(notices.filter(n => n.type === 'tool-result' && n.result.output.state === 'cancelled').length, 2); + const receipt = notices.find(n => n.type === 'tool-result' && n.result.output.state === 'cancelled').result.output; + assert.match(receipt.notice, /session controller/); + if (interrupt === 'speech') assert.equal(receipt.reason, 'accepted spoken input interrupted the active task'); + if (interrupt === 'cancel' || interrupt === 'status-then-stop') assert.equal(receipt.reason, 'explicit cancellation requested'); + } finally { + endRealtimeSession(sessionId); token.mock.restore(); globalThis.WebSocket = previousSocket; + if (previousKey === undefined) delete process.env.AI_GATEWAY_API_KEY; else process.env.AI_GATEWAY_API_KEY = previousKey; + } +}); diff --git a/scripts/test-browser-benchmark.ts b/scripts/test-browser-benchmark.ts new file mode 100644 index 0000000..a981a26 --- /dev/null +++ b/scripts/test-browser-benchmark.ts @@ -0,0 +1,314 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import golden from '../benchmarks/browser/golden.json'; +import { Benchmark, suiteSchema, fingerprint, report, markdown, consistencyErrors } from './browser-benchmark/core'; +import { page } from './browser-benchmark/page'; +import { spawn } from 'node:child_process'; +import { mkdtemp, readFile, rm, mkdir } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { once } from 'node:events'; +import { BenchmarkHost } from '../src/main/benchmarks/host'; +import { benchmarkBlocker } from '../src/skills/benchmark/skill'; +import { DEFAULT_CONFIG } from '../src/main/config/schema'; +import { cancelPendingTools } from '../src/renderer/src/lib/dex/cancelled-tools'; +import { taskProgress } from '../src/renderer/src/lib/task-progress'; +import { diagnosticToolOutcome, toolFailureCode } from '../src/main/diagnostics/tool-outcome'; + +const suite = suiteSchema.parse(golden); +function complete(run: Benchmark) { + while (run.current) { + run.start(); + while (run.started !== null) { + const { target, values } = run.current.steps[run.step].success; + run.click({ target, values, revision: run.revision, x: 100, y: 200 }); + } + } +} +test('golden criteria drive completion independently of claims; metrics and exact replay are deterministic', () => { + let now = 0; + const run = new Benchmark(suite, () => now); + run.start(); + const click = (target: string, values = {}) => run.click({ revision: run.revision, target, values, x: 10, y: 20 }); + click('offers'); click(''); click('query'); click('search', { query: 'wrong' }); + assert.equal(run.step, 0); + now = 1000; click('search', { query: 'lantern', unrelated: 'do not retain' }); + assert.equal(run.step, 1); + now = 2000; click('amber'); + assert.deepEqual(run.rows[0], { id: 'catalog-search', status: 'success', elapsedMs: 2000, misclicks: 2, invalidSubmissions: 1, score: 92, completedSteps: 2 }); + complete(run); + assert.equal(run.rows.length, 3); + assert.ok(run.rows.every(r => r.status === 'success')); + assert.ok(!JSON.stringify(run.events).includes('do not retain')); + const replay = new Benchmark(suite, () => now); + let scenario = ''; + for (const e of run.events) { + if (scenario !== e.scenario) { now = 0; replay.start(); scenario = e.scenario; } + now = e.elapsedMs; + replay.click(e); + } + assert.equal(JSON.stringify(report(replay, 'fixture')), JSON.stringify(report(run, 'fixture'))); + assert.equal(markdown(report(run, 'fixture')), markdown(report(replay, 'fixture'))); +}); +test('timeouts, cancellation, duplicate starts and stale clicks cannot produce success', () => { + let now = 0; + const run = new Benchmark(suite, () => now); + run.start(); now = 500; run.start(); assert.equal(run.started, 0); + run.click({ revision: 0, target: 'search', values: { query: 'lantern' }, x: 0, y: 0 }); + assert.equal(run.events.length, 0); + now = suite.scenarios[0].budgetMs; run.tick(); + assert.equal(run.rows[0].status, 'timeout'); assert.equal(run.rows[0].score, 0); + assert.equal(run.rows[0].elapsedMs, now); + run.start(); run.stop(); + assert.deepEqual(run.rows.map(r => r.status), ['timeout', 'aborted', 'not-run']); + assert.equal(run.rows[2].elapsedMs, null); +}); +test('comparison shows regression and recovery; changed definitions and incomplete baselines are rejected', () => { + const fast = new Benchmark(suite, () => 0); complete(fast); + const slow = new Benchmark(suite, () => 0); slow.start(); slow.finish('timeout'); complete(slow); + const a = report(fast, 'fast'), b = report(slow, 'slow'); + assert.match(markdown(b, a), /success → timeout/); + assert.match(markdown(a, b), /timeout → success/); + assert.match(markdown(a, a), /\| 0 \| 0 \| 0 \| success → success/); + const changed = structuredClone(suite); changed.scenarios[0].budgetMs++; + assert.notEqual(fingerprint(changed), fingerprint(suite)); + assert.throws(() => markdown(a, { ...b, suiteHash: fingerprint(changed) }), /Incompatible/); + assert.throws(() => markdown(a, { ...b, rows: [] }), /Incomplete/); +}); +test('suite validation rejects broken or ambiguous criteria and fixture escapes task text', () => { + const broken = structuredClone(golden); broken.scenarios[0].steps[0].success.target = 'missing'; + assert.equal(suiteSchema.safeParse(broken).success, false); + const duplicate = structuredClone(golden); duplicate.scenarios[1].id = duplicate.scenarios[0].id; + assert.equal(suiteSchema.safeParse(duplicate).success, false); + const custom = structuredClone(suite); custom.scenarios[0].task = ''; + const run = new Benchmark(custom); + assert.match(page(run, 'nonce'), /<script>/); + assert.ok(!page(run, 'nonce').includes('"success":')); +}); + +test('real CLI serves isolated fixtures, writes baseline/comparison, and finalizes cancellation', { timeout: 20000 }, async () => { + const dir = await mkdtemp(join(tmpdir(), 'dex-benchmark-test-')); + async function launch(name: string, baseline?: string) { + const out = join(dir, name); + const child = spawn(process.execPath, ['--import', 'tsx', 'scripts/browser-benchmark.ts', '--out', out, '--environment', 'synthetic protocol test; not a Dex run', ...(baseline ? ['--baseline', baseline] : [])], { stdio: ['ignore', 'pipe', 'pipe'] }); + let stderr = ''; + child.stderr.on('data', b => { stderr += b; }); + const url = await new Promise((ok, fail) => { + let stdout = ''; + child.on('error', fail); + child.once('exit', code => fail(Error(`CLI exited ${code}: ${stderr}`))); + child.stdout.on('data', b => { stdout += b; const match = stdout.match(/http:\/\/127\.0\.0\.1:\d+\/[a-f0-9]+/); if (match) ok(match[0]); }); + }); + const origin = new URL(url).origin; + const post = async (path: string, body: unknown) => { + const r = await fetch(url + path, { method: 'POST', headers: { Origin: origin, 'Content-Type': 'application/json' }, body: JSON.stringify(body) }); + assert.equal(r.status, 200); return r.json() as Promise<{ revision: number }>; + }; + const stop = async () => { const exited = once(child, 'exit'); child.kill('SIGINT'); await exited; assert.equal(child.exitCode, 0, stderr); }; + return { out, url, origin, post, stop }; + } + try { + const first = await launch('baseline'); + try { + const html = await fetch(first.url); + assert.match(await html.text(), /Start scenario/); + assert.match(html.headers.get('content-security-policy')!, /frame-ancestors 'none'/); + assert.equal((await fetch(first.url + '/start', { method: 'POST', headers: { Origin: 'https://example.com', 'Content-Type': 'application/json' }, body: '{}' })).status, 403); + assert.equal((await fetch(first.origin + '/guess')).status, 403); + for (const s of suite.scenarios) { + let state = await first.post('/start', {}); + for (const p of s.steps) state = await first.post('/click', { revision: state.revision, ...p.success, x: 25, y: 50 }); + } + assert.match(await (await fetch(first.url)).text(), /Benchmark complete/); + } finally { await first.stop(); } + const baseline = JSON.parse(await readFile(join(first.out, 'report.json'), 'utf8')); + assert.deepEqual(baseline.rows.map((r: { status: string }) => r.status), ['success', 'success', 'success']); + const second = await launch('candidate', join(first.out, 'report.json')); + try { await second.post('/start', {}); } finally { await second.stop(); } + const comparison = await readFile(join(second.out, 'report.md'), 'utf8'); + assert.match(comparison, /success → aborted/); + assert.match(comparison, /success → not-run/); + } finally { await rm(dir, { recursive: true, force: true }); } +}); + +test('in-app host is singleton per run, persists its baseline, compares later runs and finalizes stop', async () => { + const dir = await mkdtemp(join(tmpdir(), 'dex-benchmark-host-')); + const host = new BenchmarkHost(dir); + const post = async (url: string, path: string, body: unknown) => { + const response = await fetch(url + path, { method: 'POST', headers: { Origin: new URL(url).origin, 'Content-Type': 'application/json' }, body: JSON.stringify(body) }); + assert.equal(response.status, 200); return response.json() as Promise<{ revision: number }>; + }; + try { + const results = await Promise.all([host.start('fixture'), host.start('fixture')]) as Array<{ url: string }>; + assert.equal(results[0].url, results[1].url); + const url = results[0].url; + const worker = host.bindWorker(new AbortController().signal)!; + assert.equal(worker.milestone(), undefined); + for (const s of suite.scenarios) { + let state = await post(url, '/start', {}); + for (const p of s.steps) state = await post(url, '/click', { revision: state.revision, ...p.success, x: 0, y: 0 }); + assert.equal(worker.milestone(), `Verified benchmark result: ${s.id} passed.`); + assert.doesNotMatch(worker.milestone()!, /current|still running|steps completed/); + } + const beforeClose = host.status(); + assert.equal(beforeClose.outcome, 'succeeded'); + const savedBeforeClose = JSON.parse(await readFile(beforeClose.reportPath!.replace('report.md', 'report.json'), 'utf8')); + assert.deepEqual(savedBeforeClose.rows, beforeClose.rows); + assert.deepEqual(savedBeforeClose.summary, beforeClose.metrics); + assert.equal(savedBeforeClose.consistency.ok, true); + assert.equal(savedBeforeClose.fixture.complete, true); + assert.equal(savedBeforeClose.summary.succeeded, 3); + assert.ok(savedBeforeClose.rows.every((r: {elapsedMs: number; score: number; transition: string}) => Number.isFinite(r.elapsedMs) && r.score > 0 && r.transition === 'running → success')); + await worker.finish(); + await host.stop(); + assert.equal(host.status().outcome, 'succeeded'); + assert.equal(host.status().status, 'finished'); + assert.ok(!host.status().summary?.includes('Baseline environment')); + const baseline = JSON.parse(await readFile(join(dir, 'baseline.json'), 'utf8')); + assert.equal(baseline.rows.length, 3); + const resumed = new BenchmarkHost(dir); + try { + const next = await resumed.start('candidate') as { url: string }; + await post(next.url, '/start', {}); + const stopped = await resumed.stop(); + assert.match(stopped.summary!, /success → aborted/); + assert.equal(await readFile(join(dir, 'baseline.json'), 'utf8'), JSON.stringify(baseline, null, 2) + '\n'); + } finally { await resumed.stop(); } + } finally { await host.stop(); await rm(dir, { recursive: true, force: true }); } +}); + +test('benchmark startup reports disabled or never-granted prerequisites without enabling them', () => { + const context = { config: structuredClone(DEFAULT_CONFIG), platform: process.platform, availableSkillIds: ['open'] }; + const before = JSON.stringify(context); + assert.match(benchmarkBlocker(context)!, /computer/); + assert.equal(JSON.stringify(context), before); + assert.equal(benchmarkBlocker({ ...context, availableSkillIds: ['open', 'computer'] }), undefined); + assert.match(benchmarkBlocker()!, /unavailable/); +}); + +test('worker cancellation finalizes the fixture and leaves truthful progress without a baseline', async () => { + const dir=await mkdtemp(join(tmpdir(),'dex-benchmark-cancel-')); + const host=new BenchmarkHost(dir); + try { + await host.start('fixture'); + const controller=new AbortController(); const worker=host.bindWorker(controller.signal)!; + assert.equal(worker.milestone(),undefined); + assert.match(worker.progress(),/0 of 3/); assert.match(worker.progress(),/No fixture clicks/); + controller.abort(); await worker.finish(); + assert.equal(host.status().status,'finished'); + assert.deepEqual(host.status().rows?.map(r=>r.status),['not-run','not-run','not-run']); + await assert.rejects(readFile(join(dir,'baseline.json')),{code:'ENOENT'}); + } finally {await host.stop();await rm(dir,{recursive:true,force:true});} +}); +test('cancelled worker cards stop showing an active click and preserve completed history',()=>{ + const tools=[{id:'task',name:'run_task',input:{},result:null,status:'running' as const},{id:'click',name:'click',input:{},result:null,status:'running' as const},{id:'done',name:'captureScreen',input:{},result:{ok:true},status:'done' as const}]; + assert.equal(taskProgress('speaking',tools)?.label,'Clicking a control'); + const cancelled=cancelPendingTools(tools,new Set(['task','click'])); + assert.equal(taskProgress('speaking',cancelled),null); + assert.equal(cancelled[2],tools[2]); + assert.equal(cancelled[1].status,'error'); +}); +test('diagnostics preserve safe failure/cancellation categories without raw output',()=>{ + assert.equal(toolFailureCode(new Error('The foreground window changed or moved since the screenshot.')), 'stale-desktop-frame'); + assert.deepEqual(diagnosticToolOutcome({error:'private page text',code:'browser-research-navigation'}),{failed:true,reason:'browser-research-navigation'}); + assert.deepEqual(diagnosticToolOutcome({error:'private',reason:'private'}),{failed:true}); + assert.equal(diagnosticToolOutcome({state:'cancelled',reason:'accepted spoken input interrupted the active task'}).reason,'accepted spoken input interrupted the active task'); +}); + +test('hard run deadline aborts the actual worker, persists reason, and cannot establish a partial baseline', async () => { + const dir=await mkdtemp(join(tmpdir(),'dex-benchmark-deadline-')); + const host=new BenchmarkHost(dir,80); + try { + await host.start('deadline test'); + const controller=new AbortController(); + const worker=host.bindWorker(controller.signal,()=>controller.abort())!; + await new Promise(resolve=>setTimeout(resolve,140)); + assert.equal(controller.signal.aborted,true); + await worker.finish(); + assert.match(host.status().stopReason!,/maximum time/i); + const saved=JSON.parse(await readFile(host.status().reportPath!.replace('report.md','report.json'),'utf8')); + assert.match(saved.stopReason,/time limit/); + assert.equal(saved.rows.every((r: {status: string})=>r.status==='not-run'),true); + await assert.rejects(readFile(join(dir,'baseline.json')),{code:'ENOENT'}); + } finally {await host.stop();await rm(dir,{recursive:true,force:true});} +}); + +import { controlTargets } from '../src/skills/computer/control-targets'; +test('native control centers use the latest zoom/display scale and reject off-image or malformed targets',()=>{ + const shot={width:200,height:100,offsetX:-100,offsetY:200,scaleX:0.5,scaleY:0.5}; + const target={role:'AXButton',label:'Search',x:-80,y:210,width:20,height:10}; + assert.match(controlTargets([target],shot),/Search.*\(60, 30\)/); + assert.equal(controlTargets([{...target,x:500},{...target,width:NaN}],shot),''); + assert.match(page(new Benchmark(suite),'test'),/isTrusted/); + const run=new Benchmark(suite);run.start(); + assert.match(page(run,'test'),/