From a915e401d555be0a00b51687cfe2e9ad0b632756 Mon Sep 17 00:00:00 2001 From: Stahldavid Date: Thu, 24 Sep 2026 17:01:18 +0300 Subject: [PATCH] fix: improve search evidence and context selection in 1.17.3 --- .claude-plugin/marketplace.json | 2 +- .cursor-plugin/marketplace.json | 2 +- README.md | 4 + docs/avaliacao-qualidade-busca.md | 62 ++++ docs/benchmarks/search-quality-2026-09-24.csv | 44 +++ .../benchmarks/search-quality-2026-09-24.json | 25 ++ docs/plano-qualidade-busca-contexto.md | 109 ++++++ docs/search-quality.md | 42 +++ package-lock.json | 12 +- packages/cli/CHANGELOG.md | 9 + packages/cli/package.json | 4 +- packages/cli/src/usage.ts | 2 +- packages/core/CHANGELOG.md | 6 + packages/core/package.json | 2 +- packages/core/src/tool/agent-output.ts | 3 +- .../core/src/tool/helper-expansion.test.ts | 80 +++++ packages/core/src/tool/helper-expansion.ts | 111 ++++++ packages/core/src/tool/search-quality.test.ts | 92 +++++ packages/core/src/tool/search-quality.ts | 112 ++++++ packages/core/src/tool/search-schema.ts | 2 +- packages/core/src/tool/sensegrep-pipeline.ts | 41 ++- packages/core/src/tool/sensegrep.ts | 20 +- packages/core/src/tool/sensegrep.txt | 14 +- packages/mcp/CHANGELOG.md | 9 + packages/mcp/package.json | 4 +- packages/mcp/src/http-server.ts | 2 +- packages/mcp/src/server.ts | 2 +- packages/vscode/package.json | 2 +- .../.cursor-plugin/plugin.json | 2 +- plugin/sensegrep-cursor/.mcp.json | 2 +- .../.claude-plugin/marketplace.json | 2 +- .../.claude-plugin/plugin.json | 2 +- plugin/sensegrep-plugin/.mcp.json | 2 +- plugins/sensegrep/.codex-plugin/plugin.json | 2 +- plugins/sensegrep/.mcp.json | 2 +- scripts/evaluate-search-quality.mjs | 72 ++++ .../search-quality-context-curaai.json | 200 +++++++++++ scripts/fixtures/search-quality-curaai.json | 334 ++++++++++++++++++ server.json | 4 +- 39 files changed, 1402 insertions(+), 41 deletions(-) create mode 100644 docs/avaliacao-qualidade-busca.md create mode 100644 docs/benchmarks/search-quality-2026-09-24.csv create mode 100644 docs/benchmarks/search-quality-2026-09-24.json create mode 100644 docs/plano-qualidade-busca-contexto.md create mode 100644 docs/search-quality.md create mode 100644 packages/core/src/tool/helper-expansion.test.ts create mode 100644 packages/core/src/tool/helper-expansion.ts create mode 100644 packages/core/src/tool/search-quality.test.ts create mode 100644 packages/core/src/tool/search-quality.ts create mode 100644 scripts/evaluate-search-quality.mjs create mode 100644 scripts/fixtures/search-quality-context-curaai.json create mode 100644 scripts/fixtures/search-quality-curaai.json diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index b2f3e33..8d676cb 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": "./plugin/sensegrep-plugin", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Adds sensegrep MCP tools + smart usage instructions to Claude Code.", - "version": "1.17.2", + "version": "1.17.3", "author": { "name": "sensegrep" }, diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json index 3bcada4..115b8c2 100644 --- a/.cursor-plugin/marketplace.json +++ b/.cursor-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": "plugin/sensegrep-cursor", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns.", - "version": "1.17.2" + "version": "1.17.3" } ] } diff --git a/README.md b/README.md index 25b8a88..4788584 100644 --- a/README.md +++ b/README.md @@ -412,6 +412,10 @@ for Ollama, HTTP requests use `httpBatchSize` (default 16), including splits ins each index batch. Request estimates exclude retries; dry runs do not subtract vectors that will be reused. Other providers retain approximate request estimates. +### Search evidence and context + +Search preserves diverse file anchors while admitting complementary symbols and bounded local helper expansion. `context` defaults to **12,000 estimated output tokens**; use 4,000 for a compact pack or 8,000 for broader investigation. Explicit file/token limits remain respected. Weak-evidence warnings are advisory, not proof that code is absent. See [search quality and budgets](docs/search-quality.md) for behavior, limitations, and a reproducible comparison script. + ### Index compatibility Each index records the embedding provider, model, dimension, distance metric, and a non-secret endpoint/configuration fingerprint. If you change provider, model, base URL, dimension, local server pooling behavior, or task-prefix strategy, rebuild the index with `sensegrep index --root . --full --no-watch`. Same dimension does **not** make embeddings interchangeable; two 768-dimensional models still produce different vector spaces. diff --git a/docs/avaliacao-qualidade-busca.md b/docs/avaliacao-qualidade-busca.md new file mode 100644 index 0000000..d1fcf06 --- /dev/null +++ b/docs/avaliacao-qualidade-busca.md @@ -0,0 +1,62 @@ +# Avaliação das correções de busca e contexto + +Data: 24/09/2026. Baseline: CLI global 1.17.2. Candidato: build local com as alterações de qualidade, ainda sem publicação. Nenhuma integração Jev foi adicionada. + +## Ambiente e método + +O índice original do CuraAI estava desatualizado. A comparação final usa uma cópia isolada dos arquivos atuais, incluindo código ainda não rastreado pelo Git, e um índice incremental atualizado apenas nessa cópia: **940 arquivos, 6.513 chunks**, verificação estrita sem alterações, ausências ou remoções pendentes. O índice do projeto original foi preservado. + +Modelo: Ollama `qwen3-embedding:0.6b`, 1.024 dimensões. Ambas as CLIs consultaram a mesma snapshot: `chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429`. + +Conjunto principal: 28 perguntas positivas (22 históricas e seis controles/perguntas adicionais), quatro negativas e nove contextos. Onze perguntas positivas têm alvos explícitos por símbolo; as demais preservam os alvos históricos por arquivo. Contextos cobrem saldo mínimo, autenticação e voz com 1.200, 4.000 e 8.000 tokens. + +A execução alternou a ordem das CLIs. As 28 buscas positivas do candidato confirmaram cache de embedding aquecido nos diagnósticos. Tempos abaixo são de processo completo, não apenas da busca vetorial. São medições de uma passagem, não intervalos estatísticos de confiança. Os resultados medem as alterações combinadas; não isolam causalmente o ganho de cada heurística. + +## Resultado principal + +| Métrica | 1.17.2 | Candidato | +|---|---:|---:| +| Arquivo esperado no top 5 | 26/28 | 26/28 | +| Arquivo esperado no top 10 | 27/28 | 28/28 | +| Símbolo esperado no top 5 | 8/11 | 10/11 | +| Símbolo esperado no top 10 | 8/11 | 11/11 | +| Símbolo central presente nos nove contextos | 6/9 | 9/9 | +| Consultas negativas com aviso de evidência fraca | 0/4 | 3/4 | +| Avisos indevidos nas 28 perguntas positivas | 0 | 0 | +| Latência mediana | 2.536 ms | 2.828 ms | +| Latência p95 | 3.415 ms | 3.615 ms | + +O custo mediano adicional foi de 292 ms (aproximadamente 11,5%). A expansão não faz novas chamadas de embedding. + +## Casos que motivaram a mudança + +- **Saldo mínimo em inglês:** `shouldIgnoreMinimumPayout` saiu de ausente no top 10 para primeiro lugar. Também passou a ser o primeiro trecho nos três orçamentos de contexto. +- **Saldo mínimo em português:** a função saiu de ausente no top 10 para quarto lugar. A preferência por implementação preserva a relevância do arquivo, evitando que telas com mais palavras correspondentes desloquem o módulo de regras. +- **Constante como alvo:** `MAX_SMALL_BALANCE_AGE_MS` permaneceu em primeiro na pergunta explícita sobre seu valor. +- **Criptografia ampla:** `encryptPayoutDestination` saiu de ausente no top 10 para nono lugar. A recuperação melhorou; esse caso ainda não alcança o top 5. Não há alegação de recuperação perfeita. +- **Contexto de criptografia:** nos testes adicionais de 4.000 e 8.000 tokens, ambos os helpers (`encryptPayoutDestination` e `decryptPayoutDestination`) aparecem com código. O helper de criptografia, antes omitido em ambos os orçamentos, passou a ser o quinto trecho. +- **Autenticação e voz:** os helpers centrais continuaram nos contextos pequenos; os nove contextos do conjunto principal passaram a colocar o alvo em primeiro lugar. +- **Kubernetes/Kafka/EKS:** passou a receber aviso de evidência fraca. A consulta sobre Terraform/Neptune continuou `not-assessed`: termos genéricos compartilhados ainda podem impedir o aviso conservador. `not-assessed` não afirma suficiência. + +## Decisão sobre tokens + +O padrão de `context`/`audit` já era **12.000 tokens** e foi mantido. Os 1.200 tokens eram um teste de estresse. Para uso com orçamento explícito, 4.000 é uma opção compacta e 8.000 acomoda investigação entre arquivos. A seleção foi corrigida também sob 1.200, sem depender apenas de aumentar o teto. + +## Validação automatizada e reprodução + +- Build e typecheck de todos os workspaces: aprovados. +- Vitest: **266 testes aprovados, em 46 arquivos**. +- Consistência de versões: aprovada; a versão publicada não foi alterada. +- Testes novos cobrem diversidade, constantes, implementação, limites de tokens, projeção JSON, português, imports com alias, escopo de arquivos, ambiguidade, cancelamento, timeout de banco e fontes desatualizadas. + +Dados principais: [CSV por caso](benchmarks/search-quality-2026-09-24.csv) e [totais JSON](benchmarks/search-quality-2026-09-24.json). + +A saída com conteúdo de código foi validada separadamente: **11/11 contextos contêm o símbolo esperado**, contra 6/11 no baseline, e todos respeitam os limites de tokens e bytes. O CSV reúne 28 buscas positivas, quatro negativas e esses 11 contextos (43 casos distintos). Os totais JSON resumem apenas as buscas. Os ajustes finais da seleção de contexto não alteram o caminho de busca sem orçamento usado na medição de latência principal. + +Casos e executor: `scripts/fixtures/search-quality-curaai.json`, `scripts/fixtures/search-quality-context-curaai.json` e `scripts/evaluate-search-quality.mjs`. O segundo manifesto inclui conteúdo de código no JSON e dois contextos adicionais de criptografia. Veja comandos e limitações em [search-quality.md](search-quality.md). + +O indicador de evidência permanece heurístico, sem calibração probabilística. A avaliação usa um repositório e um modelo de embeddings; não comprova generalização para todos os idiomas, modelos ou bases de código. A expansão estrutural é limitada a um salto em chamadas locais JS/TS, e não substitui um grafo completo ou análise de tipos. + +## Controle sem cache + +Três perguntas foram repetidas com cache de embedding desativado, confirmando uma requisição ao Ollama por busca. Tempos baseline/candidato: criptografia 5.736/3.145 ms; saldo mínimo 3.804/3.966 ms; Clerk 3.352/3.384 ms. Essa amostra pequena e variável não sustenta uma alegação de aceleração; a comparação principal com cache aquecido é mais adequada para observar o custo da seleção e expansão. diff --git a/docs/benchmarks/search-quality-2026-09-24.csv b/docs/benchmarks/search-quality-2026-09-24.csv new file mode 100644 index 0000000..60d13a4 --- /dev/null +++ b/docs/benchmarks/search-quality-2026-09-24.csv @@ -0,0 +1,44 @@ +"case","expectedFile","expectedSymbol","contextTokens","projection","baselineFileRank","candidateFileRank","baselineSymbolRank","candidateSymbolRank","baselineMs","candidateMs","baselineSufficiency","candidateSufficiency","snapshot" +"auth-en","apps/web/lib/auth/syncClerkRoleSession.ts",,,"diagnostic","4","4",,,"2536","2660","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"auth-pt","apps/web/lib/auth/syncClerkRoleSession.ts",,,"diagnostic","2","2",,,"2918","3396","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"payment","convex/actions/livekit.ts",,,"diagnostic","2","2",,,"2320","2785","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"consent","convex/lib/patientConsent.ts",,,"diagnostic","4","4",,,"3416","3615","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"email","convex/lib/resend.ts",,,"diagnostic","1","2",,,"2277","2197","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"calendar","convex/mutations/googleCalendarWatch.ts",,,"diagnostic","2","2",,,"2516","2411","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"bridge","convex/lib/agentBridgePrompt.ts",,,"diagnostic","4","4",,,"2509","2907","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"webhook","convex/http.ts",,,"diagnostic","1","1",,,"2389","2324","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"sms-format","convex/lib/gtiSms.ts",,,"diagnostic","1","1",,,"2217","2080","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"sms-timeout","convex/lib/gtiSms.ts",,,"diagnostic","1","1",,,"2156","2170","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"voice-race","convex/mutations/agentBridge.ts",,,"diagnostic","6","6",,,"3391","3352","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"signed-pdf","convex/lib/prescriptionDelivery.ts",,,"diagnostic","2","2",,,"2299","2631","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"calendar-token","convex/mutations/doctors.ts",,,"diagnostic","3","3",,,"2787","2804","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"consent-pt","convex/lib/patientConsent.ts",,,"diagnostic","4","4",,,"2795","3177","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"privacy","apps/web/lib/analytics/posthog.ts",,,"diagnostic","1","1",,,"2141","2093","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"refund","convex/actions/asaas.ts",,,"diagnostic","1","1",,,"2326","2828","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"pix-crypto","convex/lib/payoutDestinationCrypto.ts","encryptPayoutDestination",,"diagnostic","0","9","0","9","2582","2853","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"payout-hold","convex/lib/payoutPolicy.ts","getCommissionHoldMs",,"diagnostic","2","2","2","2","3129","3475","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"credit-reuse","convex/lib/consultationCreditRedemption.ts","applyAvailableCreditForPatientConsultation",,"diagnostic","1","2","1","2","3207","3098","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"seed-auth","convex/mutations/medicalStore.ts",,,"diagnostic","2","2",,,"2391","2263","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"voice-draft","convex/lib/agentBridgePrompt.ts","formatDoctorRequestPayload",,"diagnostic","1","4","1","4","3342","3275","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"payout-minimum","convex/lib/payoutPolicy.ts","shouldIgnoreMinimumPayout",,"diagnostic","1","1","0","1","3324","3558","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"constant-control","convex/lib/payoutPolicy.ts","MAX_SMALL_BALANCE_AGE_MS",,"diagnostic","1","1","1","1","3179","3053","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"minimum-pt","convex/lib/payoutPolicy.ts","shouldIgnoreMinimumPayout",,"diagnostic","4","4","0","4","3415","3653","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"voice-concurrency","convex/mutations/agentBridge.ts","markContextConsumed",,"diagnostic","1","1","1","1","3175","3030","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"heldout-cancel","convex/mutations/agentBridge.ts","markContextConsumed",,"diagnostic","3","2","3","2","2531","2479","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"heldout-crypto","convex/lib/payoutDestinationCrypto.ts","decryptPayoutDestination",,"diagnostic","3","3","3","3","2055","2271","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"heldout-clerk","apps/web/lib/auth/syncClerkRoleSession.ts","syncClerkRoleSession",,"diagnostic","1","1","1","1","3044","3172","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"negative-0",,,,"diagnostic",,,,,"2002","2215","not-assessed","weak-evidence","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"negative-1",,,,"diagnostic",,,,,"1938","1869","not-assessed","weak-evidence","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"negative-2",,,,"diagnostic",,,,,"1949","2124","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"negative-3",,,,"diagnostic",,,,,"1654","2058","not-assessed","weak-evidence","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-payout-minimum-1200","convex/lib/payoutPolicy.ts","shouldIgnoreMinimumPayout","1200","content+diagnostic","2","1","0","1","3220","3700","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-payout-minimum-4000","convex/lib/payoutPolicy.ts","shouldIgnoreMinimumPayout","4000","content+diagnostic","3","1","0","1","3688","3750","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-payout-minimum-8000","convex/lib/payoutPolicy.ts","shouldIgnoreMinimumPayout","8000","content+diagnostic","3","1","0","1","3618","3735","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-auth-en-1200","apps/web/lib/auth/syncClerkRoleSession.ts","syncClerkRoleSession","1200","content+diagnostic","1","1","1","1","2216","2592","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-auth-en-4000","apps/web/lib/auth/syncClerkRoleSession.ts","syncClerkRoleSession","4000","content+diagnostic","2","1","2","1","2484","2647","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-auth-en-8000","apps/web/lib/auth/syncClerkRoleSession.ts","syncClerkRoleSession","8000","content+diagnostic","2","1","2","1","2487","2657","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-voice-draft-1200","convex/lib/agentBridgePrompt.ts","formatDoctorRequestPayload","1200","content+diagnostic","1","1","1","1","3048","3325","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-voice-draft-4000","convex/lib/agentBridgePrompt.ts","formatDoctorRequestPayload","4000","content+diagnostic","2","1","2","1","3257","3444","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-voice-draft-8000","convex/lib/agentBridgePrompt.ts","formatDoctorRequestPayload","8000","content+diagnostic","3","1","3","1","3695","3816","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-crypto-4000","convex/lib/payoutDestinationCrypto.ts","encryptPayoutDestination","4000","content+diagnostic","0","2","0","5","2678","2914","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" +"context-crypto-8000","convex/lib/payoutDestinationCrypto.ts","encryptPayoutDestination","8000","content+diagnostic","0","2","0","5","2720","2881","not-assessed","not-assessed","chunks_1790256700685_d9b6a21f8dfd4e0a907b5ffcba9c11fb:1790256707429" diff --git a/docs/benchmarks/search-quality-2026-09-24.json b/docs/benchmarks/search-quality-2026-09-24.json new file mode 100644 index 0000000..e6614e0 --- /dev/null +++ b/docs/benchmarks/search-quality-2026-09-24.json @@ -0,0 +1,25 @@ +{ + "cache": "existing", + "totals": { + "baseline": { + "searches": 28, + "fileTop5": 26, + "symbolCases": 11, + "symbolTop5": 8, + "medianMs": 2536, + "p95Ms": 3415, + "falseWeakWarnings": 0, + "negativesFlagged": 0 + }, + "candidate": { + "searches": 28, + "fileTop5": 26, + "symbolCases": 11, + "symbolTop5": 10, + "medianMs": 2828, + "p95Ms": 3615, + "falseWeakWarnings": 0, + "negativesFlagged": 3 + } + } +} \ No newline at end of file diff --git a/docs/plano-qualidade-busca-contexto.md b/docs/plano-qualidade-busca-contexto.md new file mode 100644 index 0000000..373f005 --- /dev/null +++ b/docs/plano-qualidade-busca-contexto.md @@ -0,0 +1,109 @@ +# Plano de melhoria da busca e seleção de contexto + +Status: implementado localmente; validação e resultados registrados em [avaliacao-qualidade-busca.md](avaliacao-qualidade-busca.md). Publicação não executada nesta etapa. + +Decisão de orçamento: `context` já tinha padrão de **12.000 tokens**, que foi mantido. 1.200 era o orçamento de estresse dos testes; a avaliação também cobre 4.000 e 8.000 tokens. A política final está em [search-quality.md](search-quality.md). + +Base: avaliação prática do Sensegrep 1.17.2 em 24/09/2026, com Ollama `qwen3-embedding:0.6b` no CuraAI. Foram feitas 22 perguntas: arquivo esperado no top 5 em 20 casos e no top 10 em 21. A mediana de busca foi 2,46 s. Esses números medem localização de arquivos, não suficiência das respostas. + +## Objetivo + +Recuperar as funções que implementam a regra perguntada e incluí-las no contexto disponível. Preservar diversidade, limites de saída e latência, sem tratar similaridade como garantia de resposta correta. + +## 1. Diversidade flexível por arquivo — prioridade alta + +**Problema observado:** a pergunta sobre liberar saldo pequeno antigo retornou `MAX_SMALL_BALANCE_AGE_MS`, mas o limite padrão de um resultado por arquivo excluiu `shouldIgnoreMinimumPayout`. Com três resultados por arquivo, a função apareceu em segundo. + +### Trabalho proposto + +- Separar a seleção do primeiro resultado por arquivo da inclusão de evidências complementares. +- Experimentar a inclusão de um segundo trecho quando ele acrescentar uma função ou regra relevante, sem repetir o conteúdo já selecionado. +- Considerar papel do símbolo, relevância à pergunta, sobreposição de conteúdo e contribuição adicional. +- Preservar limites explícitos informados pelo usuário; documentar qualquer mudança no comportamento padrão. +- Evitar preferência universal por funções: uma consulta sobre um valor de configuração pode ser melhor respondida por uma constante. + +### Critérios de aceitação + +- A pergunta sobre saldo mínimo recupera `shouldIgnoreMinimumPayout` no top 5 pelo modo padrão. +- Perguntas cujo alvo é uma constante continuam encontrando essa constante. +- A inclusão complementar não volta a concentrar os resultados em poucos arquivos nem produz trechos redundantes. + +## 2. Seleção de contexto pequeno — prioridade alta + +**Problema observado:** com 1.200 tokens, o contexto da pergunta sobre saldo mínimo incluiu preferências, uma constante e configuração de conta, mas omitiu a função que decide a exceção. + +### Trabalho proposto + +- Avaliar a seleção de contexto separadamente do ranking da busca. +- Favorecer evidências que expliquem a implementação da regra perguntada. +- Penalizar redundância e trechos periféricos; experimentar reserva de orçamento para helpers complementares. +- Preferir trechos completos quando forem suficientes e couberem, mantendo indicação explícita de qualquer corte. +- Comparar estratégias sob o mesmo orçamento; não usar apenas o aumento do limite de tokens como solução. + +### Critérios de aceitação + +- `shouldIgnoreMinimumPayout` entra no contexto de 1.200 tokens da pergunta correspondente. +- Os casos de autenticação e voz continuam incluindo seus helpers centrais. +- Os limites de tokens e bytes continuam sendo respeitados. +- Quando faltar evidência, a saída permanece explicitamente incompleta. + +## 3. Recuperação de helpers relacionados — prioridade seguinte + +**Problema observado:** a pergunta ampla sobre criptografia de dados de repasse encontrou o fluxo de pagamentos, mas não trouxe `payoutDestinationCrypto.ts` no top 10. Restrição de pasta e reformulação recuperaram o helper. + +### Trabalho proposto + +- Experimentar uma segunda etapa limitada, a partir dos resultados mais relevantes. +- Usar imports e chamadas resolvidas para localizar possíveis helpers relacionados. +- Pontuar os candidatos adicionais pela relação com a pergunta, não apenas pela proximidade no grafo. +- Deduplicar os resultados e registrar a origem da expansão nos diagnósticos. +- Definir limites de profundidade, candidatos e tempo; preservar o resultado inicial quando a expansão não puder ser concluída. +- Evitar expansão indiscriminada para dependências genéricas e não presumir que o grafo seja exaustivo. + +### Critérios de aceitação + +- A pergunta ampla de criptografia recupera o helper relevante sem exigir que o usuário conheça o nome do arquivo. +- Consultas de controle não ganham dependências irrelevantes no lugar de resultados úteis. +- O ganho de cobertura e o custo de latência são medidos separadamente. + +## 4. Indicação de evidência insuficiente — prioridade seguinte + +**Problema observado:** uma consulta sobre Kubernetes/Kafka/EKS retornou código de certificados de assinatura e outros resultados pouco pertinentes. `answerSufficiency: not-assessed` evita uma promessa de suficiência, mas não orienta bem o consumidor. + +### Trabalho proposto + +- Montar exemplos positivos, negativos e ambíguos antes de definir a regra de sinalização. +- Experimentar sinais combinados, como suporte lexical/estrutural, coerência dos resultados e separação entre candidatos. +- Calibrar o indicador com diferentes consultas e, quando viável, mais de um modelo de embeddings. +- Distinguir evidência fraca de ausência comprovada: busca semântica não prova inexistência. +- Começar por um aviso explícito e orientação de refinamento, avaliando separadamente uma eventual política de abstenção. +- Evitar um corte arbitrário de similaridade que silencie resultados úteis. + +### Critérios de aceitação + +- Consultas negativas conhecidas recebem sinalização útil de evidência fraca. +- Resultados positivos não são descartados indiscriminadamente, inclusive em português. +- A documentação deixa claro que o indicador não é uma probabilidade calibrada de correção, salvo se isso for efetivamente demonstrado. + +## Avaliação comum às quatro etapas + +- Transformar os casos observados em regressões por **símbolo e regra**, além de arquivo. +- Definir os alvos antes de executar cada consulta; aceitar implementações alternativas quando justificadas pelo código. +- Manter perguntas novas fora dos ajustes iniciais para reduzir o risco de adaptar as heurísticas apenas aos exemplos conhecidos. +- Comparar as versões sobre o mesmo snapshot de código, índice, modelo e configuração. +- Separar medições com cache aquecido das consultas sem cache. +- Medir cobertura no top 5/top 10, posição do símbolo central, presença da regra no contexto, redundância, falsos positivos e latência mediana/p95. +- Avaliar contextos de 1.200 tokens e orçamentos maiores, mantendo comparações com o mesmo limite de bytes. +- Definir tolerâncias de regressão e de latência antes de selecionar a implementação final. +- Executar testes de regressão, typecheck e testes reais da CLI; validar o pacote instalado antes de considerar uma futura release concluída. + +## Ordem de execução + +1. Consolidar a base de avaliação e os casos de controle. +2. Implementar e medir diversidade flexível. +3. Implementar e medir seleção de contexto pequeno. +4. Experimentar recuperação de helpers, mantendo-a apenas se o ganho justificar custo e ruído. +5. Calibrar a indicação de evidência insuficiente. +6. Revisar os resultados combinados, documentar limitações e preparar uma proposta de release. + +Os itens acima preservam os critérios de aceitação do plano original. A implementação recebeu testes isolados de comportamento e comparação A/B do conjunto; esta comparação não atribui separadamente o ganho de cada heurística. Consulte o relatório para os resultados e limitações da validação. diff --git a/docs/search-quality.md b/docs/search-quality.md new file mode 100644 index 0000000..0ae1a4a --- /dev/null +++ b/docs/search-quality.md @@ -0,0 +1,42 @@ +# Search evidence and context selection + +Search now keeps a primary result per file and may include one additional distinct symbol when it contributes a referenced rule/value or a separate requested operation. The first five file anchors remain diverse. Explicit `--max-per-file` remains a strict override, including `0` for no cap. Exact lookups retain their existing default of two. + +Behavior questions favor relevant executable implementations, including functions that consume a highly ranked constant. This changes the representative within a file while preserving the file's relevance. Queries explicitly asking for constants/default values do not receive this preference. Context selection combines relevance, symbol coverage, referenced helpers, novelty, and bounded size penalties; it prefers complete snippets and marks unavoidable partial source. Structurally linked helpers that cover another requested operation can survive the general relevance cutoff and receive a complementary-evidence bonus before a large peripheral chunk consumes the budget. + +## Context budget + +`context` and `audit` already default to **12,000 estimated output tokens**. That default is preserved. Use **4,000** for a smaller everyday pack or **8,000** for multi-file investigation. An explicit 1,200-token budget is still supported and tested as a stress case. + +```sh +sensegrep context "authentication role refresh" --max-tokens 4000 --json +sensegrep context "payout encryption and transfer" --max-tokens 8000 --json +``` + +These are output budgets, independent of the embedding model's input window and indexing chunk policy. JSON metadata also consumes the serialized byte budget, so increasing context tokens without increasing an explicit `--max-output-bytes` may still truncate results. + +## Local helper expansion + +Hybrid natural-language searches examine at most six source files among the first 24 candidates and collect at most 32 relevant call relationships. Only same-file and unambiguous relative-import calls in JS/TS are resolved; aliases in named imports are supported. Related symbols from the imported module can contribute a separately requested operation. They are annotated as module-related, not as direct calls. + +Expansion uses one hop, at most 256 indexed rows, source files up to 512 KB, and a 750 ms wall-time deadline. It verifies the source hash against indexed call-site ranges, preserves file/subdirectory/Git scope and structural filters, avoids embedding requests, and falls back to the initial candidates on failure. Diagnostics expose time, candidates, added results and truncation. Exact, pattern and non-hybrid searches do not expand helpers. + +This is retrieval assistance, not an exhaustive call graph. Package aliases, namespace imports, re-exports and non-JS/TS call resolution remain outside this bounded expansion. + +## Insufficient evidence + +`answerSufficiency: "weak-evidence"` is an advisory when returned candidates lack support for most meaningful query terms, or no candidates were found. It does not remove results or prove that a feature is absent. Scope errors remain separate. + +`evidenceAssessment` identifies the heuristic and matched/missing terms. It explicitly reports `calibrated: false`: this is not a calibrated probability of correctness. Positive lexical support remains `not-assessed`, never “answer proven.” Portuguese queries against English code are treated conservatively; lack of vocabulary overlap alone does not trigger the warning unless multiple named technical terms are also unsupported. Other cross-language cases need additional evaluation. + +## Reproducible comparison + +Build the candidate, retain a baseline CLI, and use an existing indexed code snapshot: + +```sh +node scripts/evaluate-search-quality.mjs --root /path/to/CuraAI --baseline /path/to/baseline/main.js --cases scripts/fixtures/search-quality-curaai.json --output /tmp/search-quality +``` + +The checked-in manifest contains historical regressions, symbol targets, new control questions, negative queries, and context budgets of 1,200/4,000/8,000 tokens. It is a CuraAI-specific evaluation set, not a general benchmark. Some historical file targets may become obsolete as that application evolves; inspect symbols and source before interpreting a miss. + +The runner alternates baseline/candidate execution, warms the query cache, records wall time and internal diagnostics, and requires a fresh, unchanged index snapshot. Use `--cache reuse` to reuse an already warmed cache and `--cache off` in a separate run to include embedding latency. Output includes per-command JSON, file/symbol ranks, context membership, median/p95, negative warnings and false warnings. Neither mode indexes or installs packages. Scope correctness, deadline fallback, strict caps and token ceilings also have provider-independent unit tests. The separate `search-quality-context-curaai.json` manifest validates content-bearing output and its serialized byte ceilings. diff --git a/package-lock.json b/package-lock.json index aa19f6c..0baaf29 100644 --- a/package-lock.json +++ b/package-lock.json @@ -8703,10 +8703,10 @@ }, "packages/cli": { "name": "@sensegrep/cli", - "version": "1.17.2", + "version": "1.17.3", "license": "Apache-2.0", "dependencies": { - "@sensegrep/core": "^1.17.2" + "@sensegrep/core": "^1.17.3" }, "bin": { "sensegrep": "dist/main.js" @@ -8717,7 +8717,7 @@ }, "packages/core": { "name": "@sensegrep/core", - "version": "1.17.2", + "version": "1.17.3", "license": "Apache-2.0", "dependencies": { "@aws-sdk/client-bedrock-runtime": "^3.1084.0", @@ -8738,13 +8738,13 @@ }, "packages/mcp": { "name": "@sensegrep/mcp", - "version": "1.17.2", + "version": "1.17.3", "license": "Apache-2.0", "dependencies": { "@modelcontextprotocol/node": "^2.0.0", "@modelcontextprotocol/sdk": "^1.29.0", "@modelcontextprotocol/server": "^2.0.0", - "@sensegrep/core": "^1.17.2", + "@sensegrep/core": "^1.17.3", "zod": "^4.4.3" }, "bin": { @@ -8760,7 +8760,7 @@ "version": "0.1.27", "license": "Apache-2.0", "dependencies": { - "@sensegrep/core": "^1.17.2" + "@sensegrep/core": "^1.17.3" }, "devDependencies": { "@types/node": "^20.19.43", diff --git a/packages/cli/CHANGELOG.md b/packages/cli/CHANGELOG.md index 311f953..8209017 100644 --- a/packages/cli/CHANGELOG.md +++ b/packages/cli/CHANGELOG.md @@ -1,5 +1,14 @@ # @sensegrep/cli +## 1.17.3 + +### Patch Changes + +- Improve implementation-aware ranking and token-bounded context selection, preserve complementary symbols with flexible default file diversity, and discover relevant local helpers through bounded call/import expansion. Add advisory weak-evidence diagnostics without discarding semantic results. Preserve the existing 12,000-token context default and explicit user limits. + +- Updated dependencies []: + - @sensegrep/core@1.17.3 + ## 1.17.2 ### Patch Changes diff --git a/packages/cli/package.json b/packages/cli/package.json index a206b37..a4e7971 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -1,6 +1,6 @@ { "name": "@sensegrep/cli", - "version": "1.17.2", + "version": "1.17.3", "type": "module", "bin": { "sensegrep": "dist/main.js" @@ -25,7 +25,7 @@ "node": ">=20" }, "dependencies": { - "@sensegrep/core": "^1.17.2" + "@sensegrep/core": "^1.17.3" }, "publishConfig": { "access": "public" diff --git a/packages/cli/src/usage.ts b/packages/cli/src/usage.ts index baef01b..1357ca2 100644 --- a/packages/cli/src/usage.ts +++ b/packages/cli/src/usage.ts @@ -40,7 +40,7 @@ Search options: --min-complexity Minimum cyclomatic complexity --max-complexity Maximum cyclomatic complexity --min-score Minimum relevance score 0-1 - --max-per-file Max results per file (search default: 1; --exact: 2) + --max-per-file Strict cap per file (default: primary + complementary symbol; --exact: 2) --max-per-symbol Max results per symbol (default: 2) --has-docs Require documentation --language typescript|javascript|python|java|vue (comma-separated for multiple) diff --git a/packages/core/CHANGELOG.md b/packages/core/CHANGELOG.md index e8df5cb..982c3a3 100644 --- a/packages/core/CHANGELOG.md +++ b/packages/core/CHANGELOG.md @@ -1,5 +1,11 @@ # @sensegrep/core +## 1.17.3 + +### Patch Changes + +- Improve implementation-aware ranking and token-bounded context selection, preserve complementary symbols with flexible default file diversity, and discover relevant local helpers through bounded call/import expansion. Add advisory weak-evidence diagnostics without discarding semantic results. Preserve the existing 12,000-token context default and explicit user limits. + ## 1.17.2 ### Patch Changes diff --git a/packages/core/package.json b/packages/core/package.json index fefc658..0a7d6b1 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -1,6 +1,6 @@ { "name": "@sensegrep/core", - "version": "1.17.2", + "version": "1.17.3", "type": "module", "main": "./dist/index.js", "types": "./dist/index.d.ts", diff --git a/packages/core/src/tool/agent-output.ts b/packages/core/src/tool/agent-output.ts index 0ab15fc..ce6e472 100644 --- a/packages/core/src/tool/agent-output.ts +++ b/packages/core/src/tool/agent-output.ts @@ -186,7 +186,8 @@ export function projectSearchAgentResponse(raw: any, options: AgentProjectionOpt : [] const response: Record = { ...baseEnvelope(raw, options), - answerSufficiency: "not-assessed", + answerSufficiency: raw.answerSufficiency ?? "not-assessed", + ...(raw.evidenceAssessment ? { evidenceAssessment: raw.evidenceAssessment } : {}), ...(raw.coverage ? { coverage: defined({ changedFiles: raw.coverage.changedFiles, diff --git a/packages/core/src/tool/helper-expansion.test.ts b/packages/core/src/tool/helper-expansion.test.ts new file mode 100644 index 0000000..9c78783 --- /dev/null +++ b/packages/core/src/tool/helper-expansion.test.ts @@ -0,0 +1,80 @@ +import { afterEach, describe, expect, it, vi } from "vitest" +import fs from "node:fs/promises" +import os from "node:os" +import path from "node:path" +import { VectorStore } from "../semantic/lancedb.js" +import { expandHelpers } from "./helper-expansion.js" +import type { SearchResources, WorkingResult } from "./sensegrep-pipeline.js" + +afterEach(() => vi.restoreAllMocks()) + +describe("bounded helper discovery", () => { + async function fixture(run: (root: string, anchor: WorkingResult) => Promise) { + const root = await fs.mkdtemp(path.join(os.tmpdir(), "sensegrep-helpers-")) + const content = 'import { encryptDestination as protect } from "./crypto.js"\nexport function saveAccount() { return protect(account) }\n' + await fs.writeFile(path.join(root, "account.ts"), content) + try { + await run(root, { file: "account.ts", content, startLine: 2, endLine: 2, semanticScore: 0.7, metadata: { symbolName: "saveAccount" } }) + } finally { await fs.rm(root, { recursive: true, force: true }) } + } + const row = { id: "helper", content: "export function encryptDestination(account) { return encrypt(account) }", distance: 0, + metadata: { file: "crypto.ts", symbolName: "encryptDestination", symbolType: "function", startLine: 1, endLine: 3 } } + it("resolves an aliased relative import and preserves structural filters", async () => { + await fixture(async (root, anchor) => { + const list = vi.spyOn(VectorStore, "listDocuments").mockResolvedValue([row] as any) + const filters = { all: [{ key: "language", operator: "equals" as const, value: "typescript" }] } + const expanded = await expandHelpers({ projectDirectory: root, collection: {} } as SearchResources, + [anchor], "Where are account destinations encrypted?", filters, new Set(["account.ts", "crypto.ts"])) + expect(expanded.results.some((r) => r.metadata.symbolName === "encryptDestination")).toBe(true) + expect(expanded.added).toBe(1) + expect(list.mock.calls[0][1]?.filters?.all).toEqual(expect.arrayContaining(filters.all)) + expect(expanded.results.at(-1)?.whyMatched?.[0]).toContain("helper expansion") + }) + }) + it("does not escape the caller's include/exclude or changed-file universe", async () => { + await fixture(async (root, anchor) => { + const list = vi.spyOn(VectorStore, "listDocuments") + const expanded = await expandHelpers({ projectDirectory: root } as SearchResources, [anchor], + "encrypted destination", {}, new Set(["account.ts"])) + expect(expanded.results).toEqual([anchor]) + expect(list).not.toHaveBeenCalled() + }) + }) + it("does not resolve current call sites using stale indexed line ranges", async () => { + await fixture(async (root, anchor) => { + const list = vi.spyOn(VectorStore, "listDocuments") + const expanded = await expandHelpers({ projectDirectory: root, meta: { files: { "account.ts": { hash: "outdated" } } } } as unknown as SearchResources, + [anchor], "encrypt destination", {}, new Set(["account.ts", "crypto.ts"])) + expect(expanded.results).toEqual([anchor]) + expect(list).not.toHaveBeenCalled() + }) + }) + it("does not expand unrelated calls or ambiguous module paths", async () => { + await fixture(async (root, anchor) => { + const list = vi.spyOn(VectorStore, "listDocuments") + for (const [query, files] of [ + ["notification delivery", ["account.ts", "crypto.ts"]], + ["encrypt destination", ["account.ts", "crypto.ts", "crypto/index.ts"]], + ] as const) { + const expanded = await expandHelpers({ projectDirectory: root } as SearchResources, [anchor], query, {}, new Set(files)) + expect(expanded.results).toEqual([anchor]) + } + expect(list).not.toHaveBeenCalled() + }) + }) + it("honors caller cancellation", async () => { + await fixture(async (root, anchor) => { + await expect(expandHelpers({ projectDirectory: root } as SearchResources, [anchor], "encrypt destination", {}, + new Set(["account.ts", "crypto.ts"]), AbortSignal.abort())).rejects.toThrow() + }) + }) + it("bounds a stalled database lookup and leaves the original candidates untouched", async () => { + await fixture(async (root, anchor) => { + vi.spyOn(VectorStore, "listDocuments").mockImplementation(() => new Promise(() => {})) + const original = structuredClone(anchor) + await expect(expandHelpers({ projectDirectory: root, collection: {} } as SearchResources, [anchor], + "encrypt destination", {}, new Set(["account.ts", "crypto.ts"]))).rejects.toThrow("750ms") + expect(anchor).toEqual(original) + }) + }) +}) diff --git a/packages/core/src/tool/helper-expansion.ts b/packages/core/src/tool/helper-expansion.ts new file mode 100644 index 0000000..4f9b6c8 --- /dev/null +++ b/packages/core/src/tool/helper-expansion.ts @@ -0,0 +1,111 @@ +import fs from "node:fs/promises" +import path from "node:path" +import { createHash } from "node:crypto" +import { TreeSitterChunking } from "../semantic/chunking-treesitter.js" +import { VectorStore } from "../semantic/lancedb.js" +import type { SearchResources, WorkingResult } from "./sensegrep-pipeline.js" +import { evidenceTerms } from "./search-quality.js" + +/** Bounded, one-hop expansion. Only same-file or relative-import calls are resolved. */ +export async function expandHelpers(resources: SearchResources, results: WorkingResult[], query: string, + filters: VectorStore.SearchFilters, allowedFiles: Set, signal?: AbortSignal) { + const controller = new AbortController() + const combined = signal ? AbortSignal.any([signal, controller.signal]) : controller.signal + let timeout: ReturnType | undefined + try { + return await Promise.race([ + expandWithinBudget(resources, results, query, filters, allowedFiles, combined), + new Promise((_, reject) => { + timeout = setTimeout(() => { controller.abort(); reject(new Error("Helper expansion exceeded 750ms")) }, 750) + }), + ]) + } finally { if (timeout) clearTimeout(timeout) } +} + +async function expandWithinBudget(resources: SearchResources, results: WorkingResult[], query: string, + filters: VectorStore.SearchFilters, allowedFiles: Set, signal?: AbortSignal) { + const started = Date.now() + const deadline = started + 750 + const terms = evidenceTerms(query) + const normalize = (file: string) => file.replace(/\\/g, "/") + const fileMap = new Map([...allowedFiles].map((file) => [normalize(file), file])) + const edges: Array<{ file: string; symbol: string; source: WorkingResult }> = [] + const files = new Map() + for (const result of results.slice(0, 24)) { + if (!/\.[cm]?[jt]sx?$/.test(result.file)) continue + if (!files.has(result.file) && files.size >= 6) continue + files.set(result.file, [...(files.get(result.file) ?? []), result]) + } + for (const [file, anchors] of files) { + signal?.throwIfAborted() + if (Date.now() >= deadline || edges.length >= 32) break + const absolute = path.resolve(resources.projectDirectory, file) + const relative = path.relative(resources.projectDirectory, absolute) + if (relative.startsWith("..") || path.isAbsolute(relative)) continue + try { + const stat = await fs.stat(absolute) + if (stat.size > 512_000) continue + const source = await fs.readFile(absolute, "utf8") + const indexedHash = resources.meta?.files?.[file]?.hash + if (indexedHash && createHash("sha1").update(source).digest("hex") !== indexedHash) continue + const calls = await TreeSitterChunking.graphCalls(source, file) + for (const call of calls) { + if (edges.length >= 32) break + const anchor = anchors.find((r) => call.line >= r.startLine && call.line <= r.endLine) + if (!anchor || !/^[\w$]+$/.test(call.target)) continue + if (!evidenceTerms(call.target).some((term) => terms.includes(term))) continue + let targetFile = file + if (call.module) { + if (!call.module.startsWith(".")) continue + const base = path.posix.normalize(path.posix.join(path.posix.dirname(normalize(file)), call.module)).replace(/\.[cm]?[jt]sx?$/, "") + const matches = [...fileMap.keys()].filter((candidate) => { + const stem = candidate.replace(/\.[cm]?[jt]sx?$/, "") + return stem === base || stem === `${base}/index` + }) + if (matches.length !== 1) continue + targetFile = fileMap.get(matches[0])! + } + if (!allowedFiles.has(targetFile)) continue + if (!edges.some((edge) => edge.file === targetFile && edge.symbol === call.target)) + edges.push({ file: targetFile, symbol: call.target, source: anchor }) + } + } catch (error) { + if (signal?.aborted) throw error + // Missing/unparseable sources cannot establish a resolved call. + } + } + if (!edges.length || Date.now() >= deadline) return { results, added: 0, considered: edges.length, elapsedMs: Date.now() - started, truncated: Date.now() >= deadline } + const rows = await VectorStore.listDocuments(resources.collection, { + filters: { ...filters, all: [...(filters.all ?? []), { key: "file", operator: "in", value: [...new Set(edges.map((e) => e.file))] }, + ], }, + limit: 256, excludeVector: true, signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(Math.max(1, deadline - Date.now()))]) : AbortSignal.timeout(Math.max(1, deadline - Date.now())), + }) + signal?.throwIfAborted() + if (Date.now() >= deadline) return { results, added: 0, considered: edges.length, elapsedMs: Date.now() - started, truncated: true } + const merged = new Map(results.map((r) => [`${r.file}:${r.startLine}:${r.endLine}`, r])) + let added = 0 + for (const row of rows) { + const direct = edges.find((e) => e.file === row.metadata.file && e.symbol === row.metadata.symbolName) + const edge = direct ?? edges.find((e) => e.file === row.metadata.file && e.file !== e.source.file) + if (!edge) continue + const symbol = String(row.metadata.symbolName ?? "") + const symbolHits = evidenceTerms(symbol).filter((term) => terms.includes(term)).length + if (symbolHits < Math.min(2, terms.length)) continue + const ownTerms = new Set(evidenceTerms(`${symbol} ${row.content}`)) + const coverage = terms.filter((term) => ownTerms.has(term)).length / Math.max(1, terms.length) + const symbolCoverage = symbolHits / Math.max(1, terms.length) + if (coverage < 0.2 || symbolCoverage === 0) continue + const key = `${edge.file}:${row.metadata.startLine}:${row.metadata.endLine}` + const existing = merged.get(key) + // A graph relationship alone is not enough: the helper supplies its own query support. + const score = Math.min(0.95, edge.source.semanticScore * 0.5 + coverage * 0.45 + symbolCoverage * 0.35 + 0.12) + merged.set(key, { ...existing, id: row.id, file: edge.file, content: row.content, + startLine: Number(row.metadata.startLine), endLine: Number(row.metadata.endLine), metadata: row.metadata, + semanticScore: existing?.semanticScore ?? score, + rerankScore: Math.max(existing?.rerankScore ?? existing?.semanticScore ?? 0, score), + whyMatched: [...(existing?.whyMatched ?? []), `helper expansion${direct ? "" : " (related module)"}: ${edge.source.file}:${edge.source.metadata.symbolName ?? edge.source.startLine} -> ${symbol}`], + }) + if (!existing) added++ + } + return { results: [...merged.values()], added, considered: edges.length, elapsedMs: Date.now() - started, truncated: rows.length === 256 } +} diff --git a/packages/core/src/tool/search-quality.test.ts b/packages/core/src/tool/search-quality.test.ts new file mode 100644 index 0000000..80bc256 --- /dev/null +++ b/packages/core/src/tool/search-quality.test.ts @@ -0,0 +1,92 @@ +import { describe, expect, it } from "vitest" +import { assessEvidence, flexibleDiversity, rankEvidence } from "./search-quality.js" +import { selectWithinTokenBudget, diversifyResults, type WorkingResult } from "./sensegrep-pipeline.js" +import { projectSearchAgentResponse } from "./agent-output.js" +import { SenseGrepContextParametersSchema } from "./sensegrep-context.js" + +const result = (symbol: string, type: string, score: number, content: string, startLine: number, file = "policy.ts"): WorkingResult => ({ + file, content, startLine, endLine: startLine + 3, semanticScore: score, + metadata: { symbolName: symbol, symbolType: type, fileRole: "implementation" }, +}) +const constant = result("MAX_SMALL_BALANCE_AGE_MS", "variable", 0.64, "export const MAX_SMALL_BALANCE_AGE_MS = 30 * Time.DAY", 1) +const helper = result("shouldIgnoreMinimumPayout", "function", 0.60, + "export function shouldIgnoreMinimumPayout(oldestAvailableAt: number, nowMs: number) { return nowMs - oldestAvailableAt >= MAX_SMALL_BALANCE_AGE_MS }", 10) +const query = "When can an old small doctor balance be paid even if below the minimum payout amount?" + +describe("evidence quality regressions", () => { + it("keeps the rule and its constant within top five without filling the page with one file", () => { + const other = result("configureAccount", "function", 0.59, "configure payout account", 20) + const rows = flexibleDiversity(rankEvidence(query, [constant, other, helper]), query) + expect(rows.map((r) => r.metadata.symbolName)).toEqual(["shouldIgnoreMinimumPayout", "MAX_SMALL_BALANCE_AGE_MS"]) + }) + it("retains constant-first ranking for a constant question and respects a strict file cap", () => { + const rows = rankEvidence("What is the default value of MAX_SMALL_BALANCE_AGE_MS?", [constant, helper]) + expect(rows[0]).toBe(constant) + expect(diversifyResults(rows, { maxPerFile: 1, maxPerSymbol: 2 })).toHaveLength(1) + }) + it.each([1200, 4000, 8000])("includes the central rule under a %i token ceiling", (budget) => { + const peripheral = result("updatePayoutPreferences", "function", 0.63, "minimum payout amount preferences " + "settings ".repeat(400), 1, "preferences.ts") + const rows = flexibleDiversity(rankEvidence(query, [constant, peripheral, helper]), query) + const pack = selectWithinTokenBudget(rows, budget, query, 5) + expect(pack.results.some((r) => r.metadata.symbolName === "shouldIgnoreMinimumPayout")).toBe(true) + expect(pack.estimatedTokens).toBeLessThanOrEqual(budget) + }) + it("keeps the already larger context default and accepts an explicit small budget", () => { + expect(SenseGrepContextParametersSchema.parse({ query }).maxTokens).toBe(12000) + expect(SenseGrepContextParametersSchema.parse({ query, maxTokens: 1200 }).maxTokens).toBe(1200) + }) + it("flags unrelated evidence without deleting it, and preserves the assessment in JSON", () => { + const rows = [result("signMedicalCertificate", "function", 0.8, "sign medical certificates", 1)] + const assessment = assessEvidence("Kubernetes operator reconciles Kafka partitions and rotates AWS EKS certificates", rows) + expect(assessment.status).toBe("weak-evidence") + const projected = projectSearchAgentResponse({ results: rows, answerSufficiency: assessment.status, evidenceAssessment: assessment }) + expect(projected.answerSufficiency).toBe("weak-evidence") + expect(projected.results).toHaveLength(1) + expect(projected.evidenceAssessment.calibrated).toBe(false) + }) + it.each(["Clerk token refresh", "atualizar token de sessao Clerk", "valor MAX_SMALL_BALANCE_AGE_MS"])("does not flag positive support: %s", (text) => { + expect(assessEvidence(text, [constant, result("refreshClerkToken", "function", 0.5, "Clerk session token refresh", 1)]).status).toBe("not-assessed") + }) + it("does not treat empty results as proof of repository-wide absence", () => { + expect(assessEvidence("missing implementation", [])).toMatchObject({ status: "weak-evidence", scope: "returned-candidates", calibrated: false }) + }) + it("does not infer weak evidence from a Portuguese/English vocabulary mismatch", () => { + const rows = [result("markContextConsumed", "function", 0.8, "return expectedLastBridgeMessageId === lastBridgeMessageId", 1)] + expect(assessEvidence("Nao apagar contexto de voz se outra mensagem chegou durante o processamento", rows).status).toBe("not-assessed") + }) + it("preserves five file anchors before admitting supplementary symbols", () => { + const others = Array.from({ length: 5 }, (_, i) => result(`rule${i}`, "function", 0.5, "minimum payout rule", 1, `rule${i}.ts`)) + const rows = flexibleDiversity(rankEvidence(query, [constant, helper, ...others]), query) + expect(new Set(rows.slice(0, 5).map((r) => r.file)).size).toBe(5) + expect(rows.some((r) => r.metadata.symbolName === "MAX_SMALL_BALANCE_AGE_MS")).toBe(true) + }) + it("retains two functions covering different requested operations in one file", () => { + const decrypt = result("decryptDestination", "function", 0.7, "function decryptDestination(encrypted) { return encrypted }", 1) + const encrypt = result("encryptDestination", "function", 0.68, "function encryptDestination(value) { return value }", 10) + expect(flexibleDiversity([decrypt, encrypt], "encrypt and decrypt destination")).toHaveLength(2) + }) + it("changes a file's representative without demoting that file below peripheral UI", () => { + const ui = result("MinimumPayoutScreen", "function", 0.59, "minimum payout amount screen", 1, "screen.ts") + const rows = rankEvidence(query, [constant, helper, ui]) + expect(rows[0].metadata.symbolName).toBe("shouldIgnoreMinimumPayout") + expect(rows[0].rerankScore).toBe(constant.semanticScore) + }) + it("keeps a structurally linked operation just below the general context cutoff", () => { + const wrapper = result("processTransfer", "function", 0.68, "return decryptPayoutDestination(value)", 1, "transfer.ts") + const crypto = result("encryptPayoutDestination", "function", 0.47, "function encryptPayoutDestination(value) { return cipher(value) }", 1, "crypto.ts") + crypto.whyMatched = ["helper expansion: configure.ts:configureTransfer -> encryptPayoutDestination"] + const noise = result("unrelatedOperation", "function", 0.47, "irrelevant", 1, "noise.ts") + const pack = selectWithinTokenBudget([wrapper, crypto, noise], 1200, "payout destination encrypted and decrypted before transfers") + expect(pack.results).toContain(crypto) + expect(pack.results).not.toContain(noise) + }) + it("selects a complementary helper before spending the remaining context on a large screen", () => { + const decrypt = result("decryptPayoutDestination", "function", 0.65, "function decryptPayoutDestination() { return decode() }", 1, "crypto.ts") + const encrypt = result("encryptPayoutDestination", "function", 0.47, "function encryptPayoutDestination() { return encode() }", 10, "crypto.ts") + encrypt.whyMatched = ["helper expansion: transfer.ts:configure -> encryptPayoutDestination"] + const ui = result("PayoutScreen", "function", 0.62, "payout destination encrypt decrypt " + "render screen ".repeat(100), 1, "screen.ts") + const pack = selectWithinTokenBudget([decrypt, ui, encrypt], 2000, "Where are payout destinations encrypted and decrypted?") + expect(pack.results.indexOf(encrypt)).toBeGreaterThanOrEqual(0) + expect(pack.results.indexOf(encrypt)).toBeLessThan(pack.results.indexOf(ui)) + }) +}) diff --git a/packages/core/src/tool/search-quality.ts b/packages/core/src/tool/search-quality.ts new file mode 100644 index 0000000..6268d9a --- /dev/null +++ b/packages/core/src/tool/search-quality.ts @@ -0,0 +1,112 @@ +import type { WorkingResult } from "./sensegrep-pipeline.js" + +// These signals rank evidence; they never assert that an answer is correct. +const STOP = new Set("a an the and or of to in on for with from by is are be can does do how where when what which before after even if than below above code function implementation logic old qual quais como onde quando para por com uma um que os as de da do dos das em no na nao se ser".split(" ")) +export function evidenceTerms(text: string): string[] { + const words = text.replace(/([a-z0-9])([A-Z])/g, "$1 $2").normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().match(/[a-z][a-z0-9]*/g) ?? [] + return [...new Set(words.filter((word) => word.length > 2 && !STOP.has(word)).map((word) => + word.length > 5 ? word.replace(/(?:ing|ed|s)$/, "") : word))] +} + +export function implementationIntent(query: string): boolean { + return /\b(when|how|where|ensure|prevent|apply|validate|quando|como|onde|garantir|impedir|validar)\b/i.test(query) + && !/\b(value|constant|default|configured|constante|padrao)\b/i.test(query) +} + +function executable(result: WorkingResult): boolean { + return ["function", "method"].includes(String(result.metadata.symbolType)) + || /=>|\bfunction\b/.test(result.content) +} + +function mentions(source: WorkingResult, target: WorkingResult): boolean { + const name = String(target.metadata.symbolName ?? "") + return name.length > 3 && /^[\w$]+$/.test(name) + && new RegExp(`(? 1) return 0 + const tokens = evidenceTerms(query) + const body = new Set(evidenceTerms(`${result.metadata.symbolName ?? ""} ${result.content}`)) + const coverage = tokens.filter((token) => body.has(token)).length / Math.max(1, tokens.length) + if (executable(result)) return Math.min(0.12, coverage * 0.2) + return ["variable", "constant"].includes(String(result.metadata.symbolType)) ? -0.06 : 0 +} + +export function rankEvidence(query: string, results: WorkingResult[]): WorkingResult[] { + const anchors = results.slice(0, 5).filter((r) => !executable(r) && ["variable", "constant"].includes(String(r.metadata.symbolType))) + const originalBest = new Map() + for (const r of results) originalBest.set(r.file, Math.max(originalBest.get(r.file) ?? 0, r.rerankScore ?? r.semanticScore)) + const adjusted = results.map((r) => { + if (r.semanticScore > 1) return r + const ruleReference = implementationIntent(query) && executable(r) && anchors.some((a) => a.file === r.file && mentions(r, a)) ? 0.1 : 0 + const adjustment = ruleReference ? Math.max(evidenceUtility(query, r), ruleReference) : evidenceUtility(query, r) + return adjustment === 0 ? r : { ...r, rerankScore: Math.max(0, Math.min(1, (r.rerankScore ?? r.semanticScore) + adjustment)), + whyMatched: [...(r.whyMatched ?? []), `implementation evidence adjustment: ${adjustment.toFixed(3)}`] } + }) + const adjustedBest = new Map() + for (const r of adjusted) adjustedBest.set(r.file, Math.max(adjustedBest.get(r.file) ?? 0, r.rerankScore ?? r.semanticScore)) + // Change which symbol represents a file without demoting the entire file: + // vocabulary-rich UI wrappers must not displace a relevant policy module. + return adjusted.map((r) => { + const original = originalBest.get(r.file)! + const shift = original - adjustedBest.get(r.file)! + return shift === 0 || original > 1 ? r : { ...r, rerankScore: Math.max(0, (r.rerankScore ?? r.semanticScore) + shift) } + }).sort((a, b) => (b.rerankScore ?? b.semanticScore) - (a.rerankScore ?? a.semanticScore)) +} + +/** One primary result plus at most one distinct, relevant, complementary symbol. */ +export function flexibleDiversity(results: WorkingResult[], query: string, maxPerSymbol = 2): WorkingResult[] { + const kept: WorkingResult[] = [] + const complements: WorkingResult[] = [] + const files = new Map() + const symbols = new Map() + const terms = evidenceTerms(query) + for (const r of results) { + const name = String(r.metadata.symbolName ?? "") + const siblings = files.get(r.file) ?? [] + if (maxPerSymbol > 0 && name && (symbols.get(name) ?? 0) >= maxPerSymbol) continue + if (siblings.length) { + const first = siblings[0] + if (siblings.length >= 2 || !name || name === first.metadata.symbolName) continue + if (first.startLine <= r.endLine && r.startLine <= first.endLine) continue + const relevant = evidenceTerms(`${name} ${r.content}`).filter((term) => terms.includes(term)).length >= 2 + const firstTerms = new Set(evidenceTerms(String(first.metadata.symbolName ?? ""))) + const addsFacet = executable(first) && executable(r) + && evidenceTerms(name).some((term) => terms.includes(term) && !firstTerms.has(term)) + const related = mentions(first, r) || mentions(r, first) || addsFacet + if (["type", "interface", "enum"].includes(String(r.metadata.symbolType)) + && !/\b(type|interface|schema|contract|tipo|contrato)\b/i.test(query)) continue + if (!relevant || !related || (r.rerankScore ?? r.semanticScore) < (first.rerankScore ?? first.semanticScore) * 0.7) continue + } + siblings.push(r) + files.set(r.file, siblings) + if (name) symbols.set(name, (symbols.get(name) ?? 0) + 1) + if (siblings.length === 1) kept.push(r) + else complements.push(r) + } + // Preserve the first five file anchors. Context selection still sees all complements. + return [...kept.slice(0, 5), ...complements, ...kept.slice(5)] +} + +export function assessEvidence(query: string, results: WorkingResult[]) { + const terms = evidenceTerms(query) + const supported = new Set(results.slice(0, 5).flatMap((r) => evidenceTerms(`${r.file} ${r.metadata.symbolName ?? ""} ${r.content}`))) + const matched = terms.filter((term) => supported.has(term)) + const portuguese = /[ãõçáéíóúâêô]/i.test(query) || (query.toLowerCase().match(/\b(onde|como|quando|nao|uma|para|por|com|que|dos|das|de|da|do|se|durante|depois|antes)\b/g)?.length ?? 0) >= 2 + const namedTerms = evidenceTerms((query.match(/\b(?:[A-Z]{2,}|[A-Z][a-zA-Z]{2,})\b/g) ?? []).join(" ")) + const crossLanguageUnsupported = !portuguese || namedTerms.filter((term) => !supported.has(term)).length >= 2 + // Conservative advisory only: several independent query terms must be missing. + // Never discard semantic matches, including cross-language matches. + const weak = results.length === 0 || (crossLanguageUnsupported && terms.length >= 4 && matched.length / terms.length < 0.25 + && !results.some((r) => r.semanticScore > 1)) + return { + status: weak ? "weak-evidence" as const : "not-assessed" as const, + method: "lexical-support-v1", + matchedTerms: matched, + missingTerms: terms.filter((term) => !supported.has(term)), + scope: "returned-candidates", + calibrated: false, + crossLanguageLimited: portuguese, + } +} diff --git a/packages/core/src/tool/search-schema.ts b/packages/core/src/tool/search-schema.ts index 6a567d0..64d9d73 100644 --- a/packages/core/src/tool/search-schema.ts +++ b/packages/core/src/tool/search-schema.ts @@ -51,7 +51,7 @@ export const SenseGrepParametersSchema = z.object({ ...CommonSearchShape, maxOutputBytes: z.number().int().min(256).max(100_000_000).optional().describe("Maximum serialized JSON response bytes (minimum 256)"), limit: z.number().int().positive().max(500).optional().describe("Maximum results (default: 10)"), - maxPerFile: z.number().int().nonnegative().optional().describe("Maximum results per file (search default: 1; exact lookup: 2)"), + maxPerFile: z.number().int().nonnegative().optional().describe("Strict maximum results per file; default admits one primary and one complementary symbol (exact: 2)"), maxPerSymbol: z.number().int().nonnegative().optional().describe("Maximum results per symbol (default: 2)"), }) diff --git a/packages/core/src/tool/sensegrep-pipeline.ts b/packages/core/src/tool/sensegrep-pipeline.ts index 74197fc..2165711 100644 --- a/packages/core/src/tool/sensegrep-pipeline.ts +++ b/packages/core/src/tool/sensegrep-pipeline.ts @@ -13,6 +13,8 @@ import { TreeShaker } from "../semantic/tree-shaker.js" import { expandSemanticKindFilter } from "../semantic/language/index.js" import { fileRoleBoost, type FileRole, type SearchPurpose } from "../semantic/file-role.js" import { createResultId, decodeResultId } from "./result-id.js" +import { expandHelpers } from "./helper-expansion.js" +import { evidenceTerms, evidenceUtility } from "./search-quality.js" export type ResultMetadata = Record @@ -685,7 +687,7 @@ export function rerankWorkingResults(query: string, results: WorkingResult[]): W const executablePenalty = seeksExecutableCode && exactSymbol === 0 && nonExecutableTypes.has(symbolType) ? 0.08 : 0 const rerankScore = Math.max(0, Math.min( 1, - result.semanticScore * 0.72 + lexical * 0.2 + structural * 0.08 + domainAdjustment - executablePenalty, + Math.max(result.semanticScore, result.rerankScore ?? 0) * 0.72 + lexical * 0.2 + structural * 0.08 + domainAdjustment - executablePenalty, )) return { ...result, @@ -711,9 +713,18 @@ export function selectWithinTokenBudget(results: WorkingResult[], maxTokens?: nu const selected: WorkingResult[] = [] let estimatedTokens = 0 const queryTokens = getQueryTokens(query) + const evidenceQueryTerms = evidenceTerms(query) const covered = new Set() const strongest = Math.max(0, ...results.map((r) => r.rerankScore ?? r.semanticScore)) - const remaining = results.filter((r) => (r.rerankScore ?? r.semanticScore) >= strongest * 0.7) + const remaining = results.filter((r) => { + const score = r.rerankScore ?? r.semanticScore + if (score >= strongest * 0.7) return true + // A separately requested operation can sit just below the general cutoff. + // Admit only bounded structural expansion with its own multi-term support. + const symbolHits = evidenceTerms(String(r.metadata.symbolName ?? "")).filter((term) => evidenceQueryTerms.includes(term)).length + return score >= strongest * 0.55 && symbolHits >= 2 + && r.whyMatched?.some((reason) => reason.startsWith("helper expansion:")) === true + }) const wantsTests = purpose === "test" || queryTokens.some((token) => ["test", "tests", "teste", "testes"].includes(token)) const wantsContracts = queryTokens.some((token) => ["type", "types", "interface", "interfaces", "schema", "contract", "contrato", "tipos"].includes(token)) const anchors = [...results].filter((r) => r.metadata.fileRole !== "test" && r.metadata.fileRole !== "contract") @@ -739,7 +750,15 @@ export function selectWithinTokenBudget(results: WorkingResult[], maxTokens?: nu const redundant = selected.some((s) => s.file === r.file && s.startLine <= r.endLine && r.startLine <= s.endLine) const contractPenalty = !wantsContracts && (r.metadata.fileRole === "contract" || ["type", "interface", "enum"].includes(String(r.metadata.symbolType))) ? 0.2 : 0 const testPenalty = !wantsTests && r.metadata.fileRole === "test" ? 0.18 : 0 - return (r.rerankScore ?? r.semanticScore) + novelty * 0.15 + (supported.has(r) ? 0.12 : 0) + const symbolHits = evidenceTerms(String(r.metadata.symbolName ?? "")).filter((term) => evidenceQueryTerms.includes(term)).length + const supportBonus = supported.has(r) && (evidenceQueryTerms.length === 0 || symbolHits >= Math.min(2, evidenceQueryTerms.length)) ? 0.12 : 0 + const symbolTerms = evidenceTerms(String(r.metadata.symbolName ?? "")) + const companionBonus = symbolHits >= 2 && r.whyMatched?.some((reason) => reason.startsWith("helper expansion:")) + && selected.some((s) => s.file === r.file && String(s.metadata.symbolName) !== String(r.metadata.symbolName) + && symbolTerms.some((term) => evidenceQueryTerms.includes(term) && !evidenceTerms(String(s.metadata.symbolName ?? "")).includes(term))) ? 0.18 : 0 + return (r.rerankScore ?? r.semanticScore) + evidenceUtility(query, r) + novelty * 0.15 + supportBonus + + companionBonus + + 0.2 * symbolHits / Math.max(1, evidenceQueryTerms.length) - contractPenalty - testPenalty - (redundant ? 0.3 : 0) - 0.12 * Math.sqrt(estimateResultTokens(r) / maxTokens) } fitting.sort((a, b) => utility(b) - utility(a) || compareWorkingResults(a, b)) @@ -1406,6 +1425,22 @@ export async function collectWorkingResults( const fallbackResults = [...exactSymbolResults, ...literalFallbackResults, ...patternFallbackResults] workingResults = fuseHybridResults(workingResults, [...hybridLexicalResults, ...fallbackResults]) + if (!params.exact && !params.pattern && params.hybrid !== false && !params.symbol && !params.name) { + try { + const expanded = await expandHelpers(resources, workingResults, params.query, filters, + new Set(lexicalCandidateFiles), options.signal) + workingResults = expanded.results + metrics.helperExpansionMs = expanded.elapsedMs + metrics.helperExpansionAdded = expanded.added + metrics.helperExpansionCandidates = expanded.considered + metrics.helperExpansionTruncated = expanded.truncated ? 1 : 0 + } catch (error) { + if (options.signal?.aborted) throw error + metrics.helperExpansionTruncated = 1 + warnings.push("Helper expansion unavailable or timed out; initial search results were preserved.") + } + } + workingResults = annotateWorkingResults(workingResults, params) workingResults = workingResults.map((result) => { const role = String(result.metadata.fileRole ?? "implementation") as FileRole diff --git a/packages/core/src/tool/sensegrep.ts b/packages/core/src/tool/sensegrep.ts index c1fd2cc..50568df 100644 --- a/packages/core/src/tool/sensegrep.ts +++ b/packages/core/src/tool/sensegrep.ts @@ -18,6 +18,7 @@ import { } from "./sensegrep-pipeline.js" import { SenseGrepParametersSchema } from "./search-schema.js" import { embeddingConfigFingerprint } from "../semantic/embedding-config.js" +import { assessEvidence, flexibleDiversity, rankEvidence } from "./search-quality.js" const DESCRIPTION = readFileSync(new URL("./sensegrep.txt", import.meta.url), "utf8") const MAX_LINE_LENGTH = 2000 @@ -111,7 +112,7 @@ export const SenseGrepTool = Tool.define("sensegrep", { subdirPrefix: resolved.subdirPrefix, freshness, schema, - }, params, { + }, { ...params, rerank: false }, { rawLimit: Math.max(200, params.pattern ? limit * 3 : limit * 2), diversify: false, signal: ctx.abort, @@ -125,7 +126,7 @@ export const SenseGrepTool = Tool.define("sensegrep", { let workingResults = collected.results // Sort by semantic score initially - workingResults.sort((a, b) => b.semanticScore - a.semanticScore) + workingResults.sort((a, b) => (b.rerankScore ?? b.semanticScore) - (a.rerankScore ?? a.semanticScore)) // Optional deterministic lexical/structural rerank on top-N candidates let rankedResults = workingResults @@ -143,6 +144,7 @@ export const SenseGrepTool = Tool.define("sensegrep", { ])), ) + rankedResults = rankEvidence(params.query, rankedResults) const minScore = typeof params.minScore === "number" ? params.minScore : undefined if (minScore !== undefined) { rankedResults = rankedResults.filter((r) => (r.rerankScore ?? r.semanticScore) >= minScore) @@ -154,7 +156,9 @@ export const SenseGrepTool = Tool.define("sensegrep", { // Enforce diversity across file/symbol to avoid repeating the same source const maxPerFile = typeof params.maxPerFile === "number" ? Math.max(0, params.maxPerFile) : (params.exact ? 2 : 1) const maxPerSymbol = typeof params.maxPerSymbol === "number" ? Math.max(0, params.maxPerSymbol) : 2 - const diversifiedResults = diversifyResults(dedupedResults, { maxPerFile, maxPerSymbol }) + const diversifiedResults = params.maxPerFile === undefined && !params.exact + ? flexibleDiversity(dedupedResults, params.query, maxPerSymbol) + : diversifyResults(dedupedResults, { maxPerFile, maxPerSymbol }) // Take top results const budgeted = selectWithinTokenBudget(diversifiedResults, params.maxTokens, params.query, limit, params.purpose) @@ -371,6 +375,11 @@ export const SenseGrepTool = Tool.define("sensegrep", { results: (result as any).results ?? [], output: (result as any).output ?? "", })) / 4)) + const evidence = assessEvidence(params.query, ((result as any).results ?? []).map((r: any) => ({ + ...r, semanticScore: r.score ?? 0, metadata: r.metadata ?? {}, + }))) + if ((result as any).status && !["complete", "incomplete"].includes((result as any).status)) evidence.status = "not-assessed" + const weakWarning = "Weak evidence: returned candidates have little lexical support for this query. Refine the query or inspect related symbols; this does not prove absence from the repository." return { schemaVersion: 1, command: params.commandName ?? "search", @@ -381,7 +390,10 @@ export const SenseGrepTool = Tool.define("sensegrep", { snapshotId: `${meta.tableName ?? "chunks"}:${meta.updatedAt}`, }, ...result, - answerSufficiency: "not-assessed", + answerSufficiency: evidence.status, + evidenceAssessment: evidence, + warnings: [...((result as any).warnings ?? []), ...(evidence.status === "weak-evidence" ? [weakWarning] : [])], + output: evidence.status === "weak-evidence" ? `${weakWarning}\n\n${result.output}` : result.output, budget: { maxOutputBytes: params.maxOutputBytes, maxBytes: params.maxOutputBytes, diff --git a/packages/core/src/tool/sensegrep.txt b/packages/core/src/tool/sensegrep.txt index 44c5ee6..68c573c 100644 --- a/packages/core/src/tool/sensegrep.txt +++ b/packages/core/src/tool/sensegrep.txt @@ -7,13 +7,14 @@ Usage: - `query` (required): Natural language description of what you're looking for - `pattern` (optional): Regex pattern to filter semantic results (case-insensitive) -- `limit` (optional): Maximum results (default: 20) +- `limit` (optional): Maximum results (default: 10) - `include` (optional): File pattern filter (e.g. "*.ts", "src/**/*.tsx") -- `rerank` (optional): Compatibility flag. Remote-only mode keeps semantic ranking as-is +- `rerank` (optional): Apply additional deterministic lexical/structural reranking - `minScore` (optional): Minimum relevance score 0-1 (filters low-confidence results) - `symbol` (optional): Filter by symbol name (e.g. "VectorStore") -- `maxPerFile` (optional): Max results per file (default: 1) -- `maxPerSymbol` (optional): Max results per symbol (default: 1) +- `maxPerFile` (optional): Strict cap; by default retain a primary and a complementary symbol, preserving the first five file anchors +- `maxPerSymbol` (optional): Max results per symbol (default: 2) +- `maxTokens` (optional): Output token budget; context defaults to 12000, independent of embedding input limits Semantic Metadata Filters (optional): - `symbolType`: Filter by code structure type - "function", "class", "method", "interface", "type", "variable", "namespace", "enum" @@ -28,8 +29,9 @@ Semantic Metadata Filters (optional): How it works: 1. Semantic search finds conceptually relevant code chunks using AI embeddings 2. If `pattern` provided, ripgrep filters to only chunks containing the pattern -3. If `rerank` enabled, the current remote-only build keeps semantic ranking unchanged -4. Top results returned with full code content +3. Hybrid searches may expand relevant local helper calls by one hop within a bounded budget +4. Optional reranking and evidence-aware diversity select results; context chooses complementary evidence within its token budget +5. Weak lexical support produces an advisory warning, not proof of repository-wide absence; cross-language evidence can remain unassessed When to use semantic search vs literal search: - semantic search first: Discovery and conceptual questions ("find error handling", "authentication logic", "database queries") diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md index 9dd2525..c334427 100644 --- a/packages/mcp/CHANGELOG.md +++ b/packages/mcp/CHANGELOG.md @@ -1,5 +1,14 @@ # @sensegrep/mcp +## 1.17.3 + +### Patch Changes + +- Improve implementation-aware ranking and token-bounded context selection, preserve complementary symbols with flexible default file diversity, and discover relevant local helpers through bounded call/import expansion. Add advisory weak-evidence diagnostics without discarding semantic results. Preserve the existing 12,000-token context default and explicit user limits. + +- Updated dependencies []: + - @sensegrep/core@1.17.3 + ## 1.17.2 ### Patch Changes diff --git a/packages/mcp/package.json b/packages/mcp/package.json index 4ee4766..a70f1fb 100644 --- a/packages/mcp/package.json +++ b/packages/mcp/package.json @@ -1,7 +1,7 @@ { "name": "@sensegrep/mcp", "mcpName": "io.github.Stahldavid/sensegrep", - "version": "1.17.2", + "version": "1.17.3", "type": "module", "main": "./dist/server.js", "bin": { @@ -33,7 +33,7 @@ "@modelcontextprotocol/node": "^2.0.0", "@modelcontextprotocol/sdk": "^1.29.0", "@modelcontextprotocol/server": "^2.0.0", - "@sensegrep/core": "^1.17.2", + "@sensegrep/core": "^1.17.3", "zod": "^4.4.3" }, "publishConfig": { diff --git a/packages/mcp/src/http-server.ts b/packages/mcp/src/http-server.ts index ccacffc..60a9a49 100644 --- a/packages/mcp/src/http-server.ts +++ b/packages/mcp/src/http-server.ts @@ -39,7 +39,7 @@ export function createSensegrepHttpHandler( return createMcpHandler(async (context) => { options.onServerCreated?.(context); const server = new Server( - { name: "sensegrep", version: "1.17.2" }, + { name: "sensegrep", version: "1.17.3" }, { capabilities: { tools: {} } }, ); diff --git a/packages/mcp/src/server.ts b/packages/mcp/src/server.ts index a394d4d..6121152 100644 --- a/packages/mcp/src/server.ts +++ b/packages/mcp/src/server.ts @@ -939,7 +939,7 @@ export function createStdioMcpServer(): Server { const server = new Server( { name: "sensegrep", - version: "1.17.2", + version: "1.17.3", }, { capabilities: { diff --git a/packages/vscode/package.json b/packages/vscode/package.json index 4d29ed0..876e696 100644 --- a/packages/vscode/package.json +++ b/packages/vscode/package.json @@ -602,6 +602,6 @@ "typescript": "^5.9.3" }, "dependencies": { - "@sensegrep/core": "^1.17.2" + "@sensegrep/core": "^1.17.3" } } diff --git a/plugin/sensegrep-cursor/.cursor-plugin/plugin.json b/plugin/sensegrep-cursor/.cursor-plugin/plugin.json index 71fbb83..e0cc606 100644 --- a/plugin/sensegrep-cursor/.cursor-plugin/plugin.json +++ b/plugin/sensegrep-cursor/.cursor-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.17.2", + "version": "1.17.3", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Replaces grep/ripgrep for 95% of code exploration tasks.", "author": { "name": "sensegrep" diff --git a/plugin/sensegrep-cursor/.mcp.json b/plugin/sensegrep-cursor/.mcp.json index 0bac553..77f8a76 100644 --- a/plugin/sensegrep-cursor/.mcp.json +++ b/plugin/sensegrep-cursor/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.17.2" + "@sensegrep/mcp@1.17.3" ] } } diff --git a/plugin/sensegrep-plugin/.claude-plugin/marketplace.json b/plugin/sensegrep-plugin/.claude-plugin/marketplace.json index 3eceacb..03cb552 100644 --- a/plugin/sensegrep-plugin/.claude-plugin/marketplace.json +++ b/plugin/sensegrep-plugin/.claude-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": ".", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Adds sensegrep MCP tools + smart usage instructions to Claude Code.", - "version": "1.17.2", + "version": "1.17.3", "author": { "name": "sensegrep" }, diff --git a/plugin/sensegrep-plugin/.claude-plugin/plugin.json b/plugin/sensegrep-plugin/.claude-plugin/plugin.json index eed2e7e..cf85b70 100644 --- a/plugin/sensegrep-plugin/.claude-plugin/plugin.json +++ b/plugin/sensegrep-plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.17.2", + "version": "1.17.3", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Replaces grep/ripgrep for 95% of code exploration tasks.", "author": { "name": "sensegrep", diff --git a/plugin/sensegrep-plugin/.mcp.json b/plugin/sensegrep-plugin/.mcp.json index 0bac553..77f8a76 100644 --- a/plugin/sensegrep-plugin/.mcp.json +++ b/plugin/sensegrep-plugin/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.17.2" + "@sensegrep/mcp@1.17.3" ] } } diff --git a/plugins/sensegrep/.codex-plugin/plugin.json b/plugins/sensegrep/.codex-plugin/plugin.json index a7c4fc9..be3781f 100644 --- a/plugins/sensegrep/.codex-plugin/plugin.json +++ b/plugins/sensegrep/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.17.2", + "version": "1.17.3", "description": "Semantic + structural code search for AI agents. Read the right code, not more code — combining semantic search, exact matching, and AST-aware retrieval.", "author": { "name": "sensegrep", diff --git a/plugins/sensegrep/.mcp.json b/plugins/sensegrep/.mcp.json index 0bac553..77f8a76 100644 --- a/plugins/sensegrep/.mcp.json +++ b/plugins/sensegrep/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.17.2" + "@sensegrep/mcp@1.17.3" ] } } diff --git a/scripts/evaluate-search-quality.mjs b/scripts/evaluate-search-quality.mjs new file mode 100644 index 0000000..6982f48 --- /dev/null +++ b/scripts/evaluate-search-quality.mjs @@ -0,0 +1,72 @@ +// Compare two built CLIs against the same existing index. Does not index or install. +// node scripts/evaluate-search-quality.mjs --root --baseline --cases --output +import { spawnSync } from 'node:child_process' +import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import path from 'node:path' +import { fileURLToPath } from 'node:url' + +const flags = Object.fromEntries(Array.from({ length: Math.floor((process.argv.length - 2) / 2) }, (_, i) => + [process.argv[2 + i * 2].replace(/^--/, ''), process.argv[3 + i * 2]])) +for (const flag of ['root', 'baseline', 'cases', 'output']) if (!flags[flag]) throw new Error(`Missing --${flag}`) +const root = path.resolve(flags.root) +const output = path.resolve(flags.output) +const candidate = path.resolve(flags.candidate ?? fileURLToPath(new URL('../packages/cli/dist/main.js', import.meta.url))) +const cases = JSON.parse(readFileSync(flags.cases, 'utf8')) +mkdirSync(output, { recursive: true }) +const rows = [] +const warmed = new Set() +const invoke = (cli, args) => { + const start = performance.now() + const p = spawnSync(process.execPath, [cli, ...args, '--root', root, '--json', '--log-format', 'none'], { + encoding: 'utf8', timeout: 30000, maxBuffer: 16 * 1024 * 1024, + env: { ...process.env, ...(flags.cache === 'off' ? { SENSEGREP_QUERY_CACHE: 'false' } : {}) }, + }) + if (p.status !== 0) throw new Error(`CLI failed: ${p.error?.message ?? p.stderr ?? p.status}`) + return { data: JSON.parse(p.stdout), ms: Math.round(performance.now() - start), bytes: Buffer.byteLength(p.stdout) } +} +const initialIndex = invoke(candidate, ['verify', '--strict']).data +for (let i = 0; i < cases.length; i++) { + const c = cases[i] + const row = { name: c.name, expected: c.expected, symbol: c.symbol, negative: c.negative, budget: c.budget } + // Alternate order to reduce systematic timing bias. A warmup makes cache mode explicit. + if (flags.cache !== 'off' && flags.cache !== 'reuse' && !warmed.has(c.query)) { + invoke(candidate, ['search', c.query, '--limit', '1']) + warmed.add(c.query) + } + const variants = [['baseline', path.resolve(flags.baseline)], ['candidate', candidate]] + if (i % 2) variants.reverse() + for (const [variant, cli] of variants) { + const { data, ms, bytes } = invoke(cli, c.args ?? ['search', c.query, '--limit', '10', '--diagnostic']) + const results = data.results ?? [] + if (c.budget && (data.budget?.usedTokens > c.budget || (data.budget?.maxBytes && bytes > data.budget.maxBytes))) + throw new Error(`Output budget exceeded for ${c.name}`) + row[variant] = { + ms, bytes, snapshot: data.diagnostic?.index?.snapshotId ?? data.index?.snapshotId, fileRank: c.expected ? results.findIndex((r) => r.file === c.expected) + 1 : null, + symbolRank: c.symbol ? results.findIndex((r) => r.file === c.expected && r.symbol === c.symbol) + 1 : null, + sufficiency: data.answerSufficiency, budget: data.budget, metrics: data.diagnostic?.metrics, + symbols: results.map((r) => `${r.file}:${r.symbol ?? ''}`), + } + writeFileSync(path.join(output, `${i}-${variant}.json`), JSON.stringify({ case: c, data, ms, bytes }, null, 2)) + } + if (!row.baseline.snapshot || !row.candidate.snapshot) throw new Error('Snapshot evidence missing; use --diagnostic in case args') + if (row.baseline.snapshot !== row.candidate.snapshot) throw new Error('Index snapshot changed during comparison') + if (row.candidate.snapshot !== initialIndex.snapshotId) throw new Error('Index snapshot changed since preflight') + rows.push(row) + writeFileSync(path.join(output, 'summary.json'), JSON.stringify(rows, null, 2)) + console.log(JSON.stringify({ name: c.name, before: row.baseline.symbolRank ?? row.baseline.fileRank, after: row.candidate.symbolRank ?? row.candidate.fileRank })) +} +const finalIndex = invoke(candidate, ['verify', '--strict']).data +if (finalIndex.snapshotId !== initialIndex.snapshotId) throw new Error('Index changed before final verification') +const searches = rows.filter((r) => !r.budget && r.expected) +const quantile = (xs, q) => [...xs].sort((a, b) => a - b)[Math.ceil(xs.length * q) - 1] +const totals = Object.fromEntries(['baseline', 'candidate'].map((variant) => [variant, { + searches: searches.length, + fileTop5: searches.filter((r) => r[variant].fileRank > 0 && r[variant].fileRank <= 5).length, + symbolCases: searches.filter((r) => r.symbol).length, + symbolTop5: searches.filter((r) => r.symbol && r[variant].symbolRank > 0 && r[variant].symbolRank <= 5).length, + medianMs: quantile(searches.map((r) => r[variant].ms), 0.5), p95Ms: quantile(searches.map((r) => r[variant].ms), 0.95), + falseWeakWarnings: searches.filter((r) => r[variant].sufficiency === 'weak-evidence').length, + negativesFlagged: rows.filter((r) => r.negative && r[variant].sufficiency === 'weak-evidence').length, +}])) +writeFileSync(path.join(output, 'totals.json'), JSON.stringify({ cache: flags.cache === 'off' ? 'disabled' : flags.cache === 'reuse' ? 'existing' : 'warmed', totals }, null, 2)) +console.log(JSON.stringify(totals)) diff --git a/scripts/fixtures/search-quality-context-curaai.json b/scripts/fixtures/search-quality-context-curaai.json new file mode 100644 index 0000000..f7cf4db --- /dev/null +++ b/scripts/fixtures/search-quality-context-curaai.json @@ -0,0 +1,200 @@ +[ + { + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?", + "expected": "convex/lib/payoutPolicy.ts", + "symbol": "shouldIgnoreMinimumPayout", + "args": [ + "context", + "When can an old small doctor balance be paid even if below the minimum payout amount?", + "--max-tokens", + "1200", + "--max-output-bytes", + "4800", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 1200, + "name": "context-payout-minimum-1200" + }, + { + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?", + "expected": "convex/lib/payoutPolicy.ts", + "symbol": "shouldIgnoreMinimumPayout", + "args": [ + "context", + "When can an old small doctor balance be paid even if below the minimum payout amount?", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 4000, + "name": "context-payout-minimum-4000" + }, + { + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?", + "expected": "convex/lib/payoutPolicy.ts", + "symbol": "shouldIgnoreMinimumPayout", + "args": [ + "context", + "When can an old small doctor balance be paid even if below the minimum payout amount?", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 8000, + "name": "context-payout-minimum-8000" + }, + { + "query": "Clerk session token refresh after onboarding role metadata", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "symbol": "syncClerkRoleSession", + "args": [ + "context", + "Clerk session token refresh after onboarding role metadata", + "--max-tokens", + "1200", + "--max-output-bytes", + "4800", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 1200, + "name": "context-auth-en-1200" + }, + { + "query": "Clerk session token refresh after onboarding role metadata", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "symbol": "syncClerkRoleSession", + "args": [ + "context", + "Clerk session token refresh after onboarding role metadata", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 4000, + "name": "context-auth-en-4000" + }, + { + "query": "Clerk session token refresh after onboarding role metadata", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "symbol": "syncClerkRoleSession", + "args": [ + "context", + "Clerk session token refresh after onboarding role metadata", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 8000, + "name": "context-auth-en-8000" + }, + { + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "expected": "convex/lib/agentBridgePrompt.ts", + "symbol": "formatDoctorRequestPayload", + "args": [ + "context", + "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "--max-tokens", + "1200", + "--max-output-bytes", + "4800", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 1200, + "name": "context-voice-draft-1200" + }, + { + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "expected": "convex/lib/agentBridgePrompt.ts", + "symbol": "formatDoctorRequestPayload", + "args": [ + "context", + "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 4000, + "name": "context-voice-draft-4000" + }, + { + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "expected": "convex/lib/agentBridgePrompt.ts", + "symbol": "formatDoctorRequestPayload", + "args": [ + "context", + "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--diagnostic", + "--json-detail", + "content" + ], + "budget": 8000, + "name": "context-voice-draft-8000" + }, + { + "symbol": "encryptPayoutDestination", + "args": [ + "context", + "Where are doctor payout bank details encrypted before storage and decrypted for transfers?", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--json-detail", + "content", + "--diagnostic" + ], + "expected": "convex/lib/payoutDestinationCrypto.ts", + "budget": 4000, + "name": "context-crypto-4000", + "query": "Where are doctor payout bank details encrypted before storage and decrypted for transfers?" + }, + { + "symbol": "encryptPayoutDestination", + "args": [ + "context", + "Where are doctor payout bank details encrypted before storage and decrypted for transfers?", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--json-detail", + "content", + "--diagnostic" + ], + "expected": "convex/lib/payoutDestinationCrypto.ts", + "budget": 8000, + "name": "context-crypto-8000", + "query": "Where are doctor payout bank details encrypted before storage and decrypted for transfers?" + } +] diff --git a/scripts/fixtures/search-quality-curaai.json b/scripts/fixtures/search-quality-curaai.json new file mode 100644 index 0000000..50d1936 --- /dev/null +++ b/scripts/fixtures/search-quality-curaai.json @@ -0,0 +1,334 @@ +[ + { + "symbol": null, + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "name": "auth-en", + "query": "Clerk session token refresh after onboarding role metadata" + }, + { + "symbol": null, + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "name": "auth-pt", + "query": "atualizar token de sessao Clerk depois de definir perfil no cadastro" + }, + { + "symbol": null, + "expected": "convex/actions/livekit.ts", + "name": "payment", + "query": "LiveKit prevent patient joining room until consultation payment confirmed" + }, + { + "symbol": null, + "expected": "convex/lib/patientConsent.ts", + "name": "consent", + "query": "patient consent timestamp validation clock skew future acceptedAt" + }, + { + "symbol": null, + "expected": "convex/lib/resend.ts", + "name": "email", + "query": "Resend notification email retry idempotency key" + }, + { + "symbol": null, + "expected": "convex/mutations/googleCalendarWatch.ts", + "name": "calendar", + "query": "Google Calendar webhook deduplicate message number" + }, + { + "symbol": null, + "expected": "convex/lib/agentBridgePrompt.ts", + "name": "bridge", + "query": "LiveKit doctor voice request untrusted draft only medical review" + }, + { + "symbol": null, + "expected": "convex/http.ts", + "name": "webhook", + "query": "Asaas webhook verify access token timing safe comparison" + }, + { + "symbol": null, + "expected": "convex/lib/gtiSms.ts", + "name": "sms-format", + "query": "Onde o telefone brasileiro e normalizado e a mensagem SMS perde acentos?" + }, + { + "symbol": null, + "expected": "convex/lib/gtiSms.ts", + "name": "sms-timeout", + "query": "GTI SMS avoid automatic retry when network timeout leaves delivery uncertain" + }, + { + "symbol": null, + "expected": "convex/mutations/agentBridge.ts", + "name": "voice-race", + "query": "Nao apagar contexto de voz se outra mensagem chegou durante o processamento" + }, + { + "symbol": null, + "expected": "convex/lib/prescriptionDelivery.ts", + "name": "signed-pdf", + "query": "Only release a prescription when its signed PDF matches the current canonical document version" + }, + { + "symbol": null, + "expected": "convex/mutations/doctors.ts", + "name": "calendar-token", + "query": "Preservar refresh token do Google Calendar quando atualizacao parcial omite campos" + }, + { + "symbol": null, + "expected": "convex/lib/patientConsent.ts", + "name": "consent-pt", + "query": "Aceite dos termos da teleconsulta rejeitado porque o relogio do celular esta adiantado" + }, + { + "symbol": null, + "expected": "apps/web/lib/analytics/posthog.ts", + "name": "privacy", + "query": "Remove personal identifiers from browser analytics event properties" + }, + { + "symbol": null, + "expected": "convex/actions/asaas.ts", + "name": "refund", + "query": "Asaas refund network failure must require manual review instead of automatic retry" + }, + { + "symbol": "encryptPayoutDestination", + "expected": "convex/lib/payoutDestinationCrypto.ts", + "name": "pix-crypto", + "query": "Where are doctor payout bank details encrypted before storage and decrypted for transfers?" + }, + { + "symbol": "getCommissionHoldMs", + "expected": "convex/lib/payoutPolicy.ts", + "name": "payout-hold", + "query": "Onde e definido quanto tempo a comissao medica fica retida antes de poder ser paga?" + }, + { + "symbol": "applyAvailableCreditForPatientConsultation", + "expected": "convex/lib/consultationCreditRedemption.ts", + "name": "credit-reuse", + "query": "Apply an available patient consultation credit before charging for another appointment" + }, + { + "symbol": null, + "expected": "convex/mutations/medicalStore.ts", + "name": "seed-auth", + "query": "Prevent anonymous users from seeding medical store products without a server secret" + }, + { + "symbol": "formatDoctorRequestPayload", + "expected": "convex/lib/agentBridgePrompt.ts", + "name": "voice-draft", + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending" + }, + { + "symbol": "shouldIgnoreMinimumPayout", + "expected": "convex/lib/payoutPolicy.ts", + "name": "payout-minimum", + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?" + }, + { + "symbol": "MAX_SMALL_BALANCE_AGE_MS", + "expected": "convex/lib/payoutPolicy.ts", + "name": "constant-control", + "query": "What is the value of MAX_SMALL_BALANCE_AGE_MS?" + }, + { + "symbol": "shouldIgnoreMinimumPayout", + "expected": "convex/lib/payoutPolicy.ts", + "name": "minimum-pt", + "query": "Quando um saldo pequeno pode ser pago apesar do valor minimo de repasse?" + }, + { + "symbol": "markContextConsumed", + "expected": "convex/mutations/agentBridge.ts", + "name": "voice-concurrency", + "query": "How does expectedLastBridgeMessageId prevent clearing voice context after another message arrives?" + }, + { + "symbol": "markContextConsumed", + "expected": "convex/mutations/agentBridge.ts", + "name": "heldout-cancel", + "query": "Which comparison rejects acknowledgment of voice context when the latest bridge message has changed?" + }, + { + "symbol": "decryptPayoutDestination", + "expected": "convex/lib/payoutDestinationCrypto.ts", + "name": "heldout-crypto", + "query": "How is a saved payout destination decoded for a bank transfer?" + }, + { + "symbol": "syncClerkRoleSession", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "name": "heldout-clerk", + "query": "After assigning a role, force Clerk to fetch a fresh session token" + }, + { + "name": "negative-0", + "negative": true, + "query": "Kubernetes operator reconciles Kafka partitions and rotates AWS EKS certificates" + }, + { + "name": "negative-1", + "negative": true, + "query": "Redis cluster sentinel failover replication slots" + }, + { + "name": "negative-2", + "negative": true, + "query": "Terraform provisioning Amazon Neptune graph clusters" + }, + { + "name": "negative-3", + "negative": true, + "query": "Kubernetes Kafka EKS certificados infraestrutura" + }, + { + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?", + "expected": "convex/lib/payoutPolicy.ts", + "symbol": "shouldIgnoreMinimumPayout", + "args": [ + "context", + "When can an old small doctor balance be paid even if below the minimum payout amount?", + "--max-tokens", + "1200", + "--max-output-bytes", + "4800", + "--diagnostic" + ], + "budget": 1200, + "name": "context-payout-minimum-1200" + }, + { + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?", + "expected": "convex/lib/payoutPolicy.ts", + "symbol": "shouldIgnoreMinimumPayout", + "args": [ + "context", + "When can an old small doctor balance be paid even if below the minimum payout amount?", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--diagnostic" + ], + "budget": 4000, + "name": "context-payout-minimum-4000" + }, + { + "query": "When can an old small doctor balance be paid even if below the minimum payout amount?", + "expected": "convex/lib/payoutPolicy.ts", + "symbol": "shouldIgnoreMinimumPayout", + "args": [ + "context", + "When can an old small doctor balance be paid even if below the minimum payout amount?", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--diagnostic" + ], + "budget": 8000, + "name": "context-payout-minimum-8000" + }, + { + "query": "Clerk session token refresh after onboarding role metadata", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "symbol": "syncClerkRoleSession", + "args": [ + "context", + "Clerk session token refresh after onboarding role metadata", + "--max-tokens", + "1200", + "--max-output-bytes", + "4800", + "--diagnostic" + ], + "budget": 1200, + "name": "context-auth-en-1200" + }, + { + "query": "Clerk session token refresh after onboarding role metadata", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "symbol": "syncClerkRoleSession", + "args": [ + "context", + "Clerk session token refresh after onboarding role metadata", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--diagnostic" + ], + "budget": 4000, + "name": "context-auth-en-4000" + }, + { + "query": "Clerk session token refresh after onboarding role metadata", + "expected": "apps/web/lib/auth/syncClerkRoleSession.ts", + "symbol": "syncClerkRoleSession", + "args": [ + "context", + "Clerk session token refresh after onboarding role metadata", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--diagnostic" + ], + "budget": 8000, + "name": "context-auth-en-8000" + }, + { + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "expected": "convex/lib/agentBridgePrompt.ts", + "symbol": "formatDoctorRequestPayload", + "args": [ + "context", + "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "--max-tokens", + "1200", + "--max-output-bytes", + "4800", + "--diagnostic" + ], + "budget": 1200, + "name": "context-voice-draft-1200" + }, + { + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "expected": "convex/lib/agentBridgePrompt.ts", + "symbol": "formatDoctorRequestPayload", + "args": [ + "context", + "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "--max-tokens", + "4000", + "--max-output-bytes", + "16000", + "--diagnostic" + ], + "budget": 4000, + "name": "context-voice-draft-4000" + }, + { + "query": "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "expected": "convex/lib/agentBridgePrompt.ts", + "symbol": "formatDoctorRequestPayload", + "args": [ + "context", + "Ensure a doctor voice request can only create a draft and cannot authorize signing or sending", + "--max-tokens", + "8000", + "--max-output-bytes", + "32000", + "--diagnostic" + ], + "budget": 8000, + "name": "context-voice-draft-8000" + } +] diff --git a/server.json b/server.json index cc530a8..37179b5 100644 --- a/server.json +++ b/server.json @@ -8,12 +8,12 @@ "url": "https://github.com/Stahldavid/sensegrep", "source": "github" }, - "version": "1.17.2", + "version": "1.17.3", "packages": [ { "registryType": "npm", "identifier": "@sensegrep/mcp", - "version": "1.17.2", + "version": "1.17.3", "transport": { "type": "stdio" },