diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index ca81349..de1d59b 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": "./plugin/sensegrep-plugin", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Adds sensegrep MCP tools + smart usage instructions to Claude Code.", - "version": "1.16.0", + "version": "1.17.0", "author": { "name": "sensegrep" }, diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json index 2d0aac3..58ddbb5 100644 --- a/.cursor-plugin/marketplace.json +++ b/.cursor-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": "plugin/sensegrep-cursor", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns.", - "version": "1.16.0" + "version": "1.17.0" } ] } diff --git a/.gitignore b/.gitignore index 01c360f..61f2aec 100644 --- a/.gitignore +++ b/.gitignore @@ -1,5 +1,6 @@ node_modules/ -bun.lock +bun.lock +.sensegrep-build.lock # build outputs dist/ diff --git a/docs/cli-reference.md b/docs/cli-reference.md index f47420b..73791ac 100644 --- a/docs/cli-reference.md +++ b/docs/cli-reference.md @@ -347,6 +347,36 @@ sensegrep selftest --strict --json Hybrid retrieval defaults to parallel lexical and semantic collection. Search investigates at least 200 candidates before selecting the requested number of results; broad queries may take longer than adaptive mode. Use --hybrid-mode adaptive to explicitly opt into lexical skipping. -Duplicate scans first detect exact/normalized copies. Small cosine-vector candidate sets (up to 512) are compared in memory. Continuations preserve previous pairs in temporary snapshot-bound checkpoints; changed snapshots/candidate sets require restarting without --resume-cursor. A max-candidates cap still means incomplete coverage even when that subset finishes. - -Context accepts --max-output-bytes in addition to --max-tokens. Compact, relevant complete symbols are preferred over truncating an oversized first result; truncation remains explicit in JSON. +Duplicate scans first detect exact/normalized copies. Cosine-vector candidate sets up to 4096 are compared in memory, with precomputed norms and cached symmetric distances; larger sets and other distance metrics use the vector store. Diagnostic output reports load, preparation, and neighbor-scan timings plus the selected strategy. Continuations preserve previous pairs in temporary snapshot-bound checkpoints; changed snapshots/candidate sets require restarting without --resume-cursor. A max-candidates cap still means incomplete coverage even when that subset finishes. + +Search and context accept `--max-output-bytes` in addition to `--max-tokens`. +For search, the byte limit must be at least 256. JSON byte limits include the envelope, +UTF-8 encoding, pretty printing, and the trailing newline. If evidence does not fit, +`status: "incomplete"` and explicit truncation are returned. Token counts remain estimates. + +Token-budget selection considers the candidate pool before applying the result limit. +It balances relevance, new query-term coverage, source roles, and references from strong +candidates. Tests and type contracts receive less context space unless requested; +`--purpose test` explicitly favors test evidence. These are retrieval heuristics, not +proof that all relevant behavior has been found. Whole symbols are preferred, with an +explicit partial snippet only when no complete candidate fits. + +Search/context output reports `answerSufficiency: "not-assessed"`. Diagnostic cards use +`rankingStrength` for the high/medium/low ranking heuristic. Full internal results retain +`confidence` as a deprecated compatibility alias; neither value is a calibrated answer +probability. Execution completion is distinct from answer sufficiency. + +Graph references include source/target locations, resolution method, and call line when +available. Calls inside nested indexed symbols belong to the smallest containing symbol. +`graphCoverage` is the fraction of resolved extracted edges (including synthetic imports +and table references), not recall against all real references. Use compiler-aware tools +for exhaustive refactoring. Dynamic calls and unsupported alias/re-export forms may remain +unresolved. + +Clusters require similarity to every member to prevent transitive similarity chains. +Titles prefer symbol/file terms over common testing/framework imports. Colliding cluster +titles include a source location for disambiguation. + +Chunking policy version 6 preserves structural metadata even below the old minimum file +size. `sensegrep index --no-watch` detects the old policy and rebuilds the index atomically; +existing indexes remain readable until that explicit indexing command runs. diff --git a/docs/evaluation-fixes-2026-09-24.md b/docs/evaluation-fixes-2026-09-24.md new file mode 100644 index 0000000..2b80df5 --- /dev/null +++ b/docs/evaluation-fixes-2026-09-24.md @@ -0,0 +1,71 @@ +# Correções da avaliação de 24/09/2026 + +Implementação local na branch `works/fix-evaluation-regressions`, baseada na versão +1.16.0. Estes resultados se referem à CLI compilada do repositório, não ao pacote +global publicado. + +## Mudanças + +- Arquivos pequenos passam pelo parser e preservam linguagem, símbolos e chamadas. + O fallback também conserva linguagem e trechos curtos. Política de chunking 6: + a próxima indexação explícita reconhece a necessidade de reconstrução. +- O contexto aplica o orçamento antes de limitar o número final de resultados. + A seleção considera relevância, cobertura adicional de termos, papel do arquivo + e referências textuais em candidatos relevantes. O ganho por tamanho é limitado. + Testes e contratos continuam disponíveis e podem ser priorizados explicitamente. +- `rankingStrength` substitui a apresentação ambígua de `confidence` nos cards + diagnósticos; a saída interna completa preserva o alias para compatibilidade. + `answerSufficiency: "not-assessed"` separa conclusão da execução de suficiência. +- O grafo atribui chamadas ao menor símbolo indexado que as contém. Referências + expõem origem, destino, método de resolução e linha da chamada quando disponível. + O cache foi versionado e o denominador da cobertura está documentado. +- Duplicatas com até 4096 candidatos e distância cosseno usam vetores normalizados + e distâncias simétricas em memória. Há verificação de prazo, cancelamento, + continuação e tempos separados de leitura, preparação e busca de vizinhos. +- Clusters exigem compatibilidade com todos os membros, evitando agrupamentos + gigantes formados por uma cadeia de semelhanças. Nomes preferem símbolos/arquivos; + imports comuns de frameworks e testes têm menos influência. +- `search --max-output-bytes` limita o JSON completo, incluindo formatação e UTF-8. + O mínimo aceito é 256 bytes; limites inválidos retornam JSON de erro e exit code 2. + +## Validação + +- Build e type-check dos workspaces passaram; versões dos pacotes consistentes. +- Suíte completa: 44 arquivos, 232 testes aprovados. Depois, duas regressões de + seleção por papel/referência foram adicionadas; a execução direcionada passou + com 25 testes. Total de testes na árvore final: 234. +- Fixture real indexada com Ollama, incluindo arquivos menores que 200 caracteres: + filtros Python, Java e Vue retornaram resultados; o grafo encontrou + `purchase -> authorizeDebit -> readBalance` via alias importado e distinguiu + chamada direta de chamada agendada. +- Contexto de concorrência de voz com 1200 tokens passou a incluir + `markContextConsumed`, ausente na avaliação anterior. Contexto de autenticação + manteve os helpers de refresh de sessão. Ambos indicam orçamento incompleto. +- Busca de duplicatas com os mesmos parâmetros da avaliação anterior: + 1628 candidatos, limite 1500, prazo 5 segundos. Antes: 44 processados; depois: + 873 processados (aproximadamente 20 vezes mais nesta medição). +- Continuação: 627 candidatos restantes processados, sem timeout. Os 24 grupos + acumulados foram iguais aos 24 da execução única de 1500 candidatos. + O resultado continua incompleto devido aos 128 candidatos excluídos pelo limite. +- Escopo pequeno de duplicatas: 219/219 candidatos e os mesmos quatro grupos; + 1,524 s de processo, contra 1,511 s na avaliação anterior. O ganho é no escopo amplo. +- Consulta ampla de pagamentos: maior cluster retornado caiu de 52 para 18 membros; + os títulos distinguem processamento de webhook, reconciliação e mutações de status. +- JSON minimal/content/diagnostic/full com pretty printing respeitou 1800 bytes; + content respeitou também 1500 bytes. Quando os diagnósticos não cabem, a saída + degrada para um envelope incompleto e válido. +- A consulta negativa sobre Kubernetes continua podendo retornar um vizinho + semântico; a saída agora declara explicitamente que suficiência não foi avaliada. + +## Limites e aplicação + +Os tempos são observações locais, não um benchmark controlado de hardware. +O ganho de duplicatas vem do algoritmo de comparação, não de uso adicional da GPU. +Ranking e contexto continuam heurísticos; não garantem encontrar toda regra de negócio. +O grafo não substitui resolução pelo compilador para chamadas dinâmicas, re-exports +e outras formas que não resolve. A contagem de tokens de saída continua estimada. + +O pacote npm, a CLI global e a skill instalada permanecem na versão publicada. +Depois de disponibilizar esta implementação, execute `sensegrep index --no-watch` +nos projetos para reconstruir índices com a política antiga. Não houve reconstrução +do índice principal durante estes testes; apenas a fixture temporária foi indexada. diff --git a/package-lock.json b/package-lock.json index e52648e..c255bd8 100644 --- a/package-lock.json +++ b/package-lock.json @@ -8703,10 +8703,10 @@ }, "packages/cli": { "name": "@sensegrep/cli", - "version": "1.16.0", + "version": "1.17.0", "license": "Apache-2.0", "dependencies": { - "@sensegrep/core": "^1.16.0" + "@sensegrep/core": "^1.17.0" }, "bin": { "sensegrep": "dist/main.js" @@ -8717,7 +8717,7 @@ }, "packages/core": { "name": "@sensegrep/core", - "version": "1.16.0", + "version": "1.17.0", "license": "Apache-2.0", "dependencies": { "@aws-sdk/client-bedrock-runtime": "^3.1084.0", @@ -8738,13 +8738,13 @@ }, "packages/mcp": { "name": "@sensegrep/mcp", - "version": "1.16.0", + "version": "1.17.0", "license": "Apache-2.0", "dependencies": { "@modelcontextprotocol/node": "^2.0.0", "@modelcontextprotocol/sdk": "^1.29.0", "@modelcontextprotocol/server": "^2.0.0", - "@sensegrep/core": "^1.16.0", + "@sensegrep/core": "^1.17.0", "zod": "^4.4.3" }, "bin": { @@ -8760,7 +8760,7 @@ "version": "0.1.27", "license": "Apache-2.0", "dependencies": { - "@sensegrep/core": "^1.16.0" + "@sensegrep/core": "^1.17.0" }, "devDependencies": { "@types/node": "^20.19.43", diff --git a/packages/cli/CHANGELOG.md b/packages/cli/CHANGELOG.md index b8222d0..6ea9c6e 100644 --- a/packages/cli/CHANGELOG.md +++ b/packages/cli/CHANGELOG.md @@ -1,5 +1,16 @@ # @sensegrep/cli +## 1.17.0 + +### Minor Changes + +- [`bde06a7`](https://github.com/Stahldavid/sensegrep/commit/bde06a75e8327c4a5c0a4c90db9b03f06c00a3b6) Thanks [@Stahldavid](https://github.com/Stahldavid)! - Preserve structural metadata in small files, improve token-budget context selection and cluster coherence, expose graph resolution evidence, accelerate bounded duplicate scans, and support serialized byte budgets for search. Distinguish ranking strength from answer sufficiency. Chunking policy changes trigger an atomic rebuild on the next explicit indexing command. + +### Patch Changes + +- Updated dependencies [[`bde06a7`](https://github.com/Stahldavid/sensegrep/commit/bde06a75e8327c4a5c0a4c90db9b03f06c00a3b6)]: + - @sensegrep/core@1.17.0 + ## 1.16.0 ### Minor Changes diff --git a/packages/cli/README.md b/packages/cli/README.md index 11aee8f..4874a87 100644 --- a/packages/cli/README.md +++ b/packages/cli/README.md @@ -33,7 +33,9 @@ sensegrep semantic-kinds --json `--json` writes parseable JSON to stdout; progress and warnings are written to stderr. Argument errors under `--json` also use stdout JSON and exit code 2. `--max-output-bytes` -is enforced against the complete serialized payload. +is supported by search/context/audit/literal and enforced against the complete serialized JSON payload. Search requires at least 256 bytes. +Search/context explicitly report `answerSufficiency: "not-assessed"`; diagnostic +`rankingStrength` describes retrieval ranking, not proof that the question is answered. Hybrid retrieval runs semantic and lexical work concurrently and uses one batched index read for lexical matches. `--hybrid-mode parallel` is the default; `--no-hybrid` is available for latency-sensitive semantic-only discovery. Deterministic query embeddings are cached locally diff --git a/packages/cli/package.json b/packages/cli/package.json index 0412c08..d4dbd21 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -1,6 +1,6 @@ { "name": "@sensegrep/cli", - "version": "1.16.0", + "version": "1.17.0", "type": "module", "bin": { "sensegrep": "dist/main.js" @@ -25,7 +25,7 @@ "node": ">=20" }, "dependencies": { - "@sensegrep/core": "^1.16.0" + "@sensegrep/core": "^1.17.0" }, "publishConfig": { "access": "public" diff --git a/packages/cli/src/args.test.ts b/packages/cli/src/args.test.ts index 9eef06c..51eeb33 100644 --- a/packages/cli/src/args.test.ts +++ b/packages/cli/src/args.test.ts @@ -28,6 +28,7 @@ describe("CLI arguments", () => { }) it("accepts embedding timeout and global audit budgets", () => { + expect(validateKnownFlags("search", { "max-output-bytes": "4800" })).toBeUndefined() expect(validateKnownFlags("context", { "max-output-bytes": "4800" })).toBeUndefined() expect(validateKnownFlags("search", { "embedding-timeout": "1000" })).toBeUndefined() expect(validateKnownFlags("audit", { diff --git a/packages/cli/src/args.ts b/packages/cli/src/args.ts index 4843aa3..9e76c51 100644 --- a/packages/cli/src/args.ts +++ b/packages/cli/src/args.ts @@ -50,7 +50,7 @@ const ALLOWED_FLAGS_BY_COMMAND: Record> = { ]), verify: new Set([...GLOBAL_FLAGS, "strict"]), status: new Set([...GLOBAL_FLAGS, "verbose", "verify"]), - search: new Set([...GLOBAL_FLAGS, ...EMBEDDING_FLAGS, ...INDEX_RUN_FLAGS, ...SEARCH_FILTER_FLAGS]), + search: new Set([...GLOBAL_FLAGS, ...EMBEDDING_FLAGS, ...INDEX_RUN_FLAGS, ...SEARCH_FILTER_FLAGS, "max-output-bytes", "maxOutputBytes"]), literal: new Set([...GLOBAL_FLAGS, "query", "include", "exclude", "limit", "regex", "ignore-case", "ignoreCase", "filesystem", "max-output-bytes", "maxOutputBytes", "include-rendered-output", "dry-run"]), context: new Set([...GLOBAL_FLAGS, ...EMBEDDING_FLAGS, ...INDEX_RUN_FLAGS, ...SEARCH_FILTER_FLAGS, "require-coverage", "requireCoverage", "max-output-bytes", "maxOutputBytes"]), audit: new Set([ diff --git a/packages/cli/src/search-commands.ts b/packages/cli/src/search-commands.ts index 9193526..08ce7b9 100644 --- a/packages/cli/src/search-commands.ts +++ b/packages/cli/src/search-commands.ts @@ -143,6 +143,7 @@ export function buildCommonSearchParams(query: string, flags: Flags, defaults: O if (flags["no-shake"] !== undefined) params.shake = false assignNumberParam(params, flags, "minScore", ["min-score", "minScore"]) assignNumberParam(params, flags, "maxTokens", ["max-tokens", "maxTokens"]) + assignNumberParam(params, flags, "maxOutputBytes", ["max-output-bytes", "maxOutputBytes"]) if (flags.hybrid !== undefined) params.hybrid = toBool(flags.hybrid) ?? true if (flags["no-hybrid"] !== undefined) params.hybrid = false assignStringParam(params, flags, "hybridMode", ["hybrid-mode", "hybridMode"]) @@ -193,6 +194,9 @@ export async function executeSearchLikeTool(input: { includeRendered, includeFilterExplanations: input.params.explainFilters === true, }) + if (typeof input.params.maxOutputBytes === "number") { + payload.budget = { ...payload.budget, maxBytes: input.params.maxOutputBytes } + } const finalPayload = enforceActualOutputBudget(payload) if (input.params.requireCoverage === true && finalPayload.coverageSatisfied === false) process.exitCode = 2 writeJson(finalPayload) diff --git a/packages/cli/src/usage.ts b/packages/cli/src/usage.ts index 298505a..3dd7b84 100644 --- a/packages/cli/src/usage.ts +++ b/packages/cli/src/usage.ts @@ -77,7 +77,7 @@ Search options: --continue-uncovered Add token-bounded batches until changed-file textual coverage is complete --batch-tokens Per-batch audit budget (default: 4000) --max-total-tokens Global audit token budget, including continuation batches - --max-output-bytes Global serialized audit evidence budget + --max-output-bytes Serialized JSON budget for search/context/audit (search minimum: 256) --max-batches Maximum number of continuation batches --profile Select a side-by-side named index profile --embed-model Override remote embedding model diff --git a/packages/core/CHANGELOG.md b/packages/core/CHANGELOG.md index 396c4bc..83bae0a 100644 --- a/packages/core/CHANGELOG.md +++ b/packages/core/CHANGELOG.md @@ -1,5 +1,11 @@ # @sensegrep/core +## 1.17.0 + +### Minor Changes + +- [`bde06a7`](https://github.com/Stahldavid/sensegrep/commit/bde06a75e8327c4a5c0a4c90db9b03f06c00a3b6) Thanks [@Stahldavid](https://github.com/Stahldavid)! - Preserve structural metadata in small files, improve token-budget context selection and cluster coherence, expose graph resolution evidence, accelerate bounded duplicate scans, and support serialized byte budgets for search. Distinguish ranking strength from answer sufficiency. Chunking policy changes trigger an atomic rebuild on the next explicit indexing command. + ## 1.16.0 ### Minor Changes diff --git a/packages/core/package.json b/packages/core/package.json index df7bc97..229cd61 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -1,6 +1,6 @@ { "name": "@sensegrep/core", - "version": "1.16.0", + "version": "1.17.0", "type": "module", "main": "./dist/index.js", "types": "./dist/index.d.ts", diff --git a/packages/core/src/semantic/chunk-limits.ts b/packages/core/src/semantic/chunk-limits.ts index cfc85e2..169bad8 100644 --- a/packages/core/src/semantic/chunk-limits.ts +++ b/packages/core/src/semantic/chunk-limits.ts @@ -2,7 +2,7 @@ import { tokenCounterIdentity } from "./token-count.js" import { getEmbeddingConfig, type EmbeddingConfig } from "./embedding-config.js" const CHARS_PER_TOKEN = 4 -const CHUNKING_SIGNATURE_VERSION = 5 +const CHUNKING_SIGNATURE_VERSION = 6 export type GeneralChunkLimits = { max: number diff --git a/packages/core/src/semantic/chunking.test.ts b/packages/core/src/semantic/chunking.test.ts index bd00e7a..cff4d97 100644 --- a/packages/core/src/semantic/chunking.test.ts +++ b/packages/core/src/semantic/chunking.test.ts @@ -2,6 +2,23 @@ import { describe, expect, it } from "vitest" import { getGeneralChunkLimits } from "./chunk-limits.js" import { Chunking } from "./chunking.js" +describe("small source files", () => { + const cases = [ + ["caller.ts", "import { debit as allowDebit } from './rules';\nexport function purchase() { return allowDebit(); }", "typescript", "purchase"], + ["mail.py", "class MailQueue:\n def retry_failed_delivery(self, attempts):\n return 'retry' if attempts < 3 else 'manual_review'\n", "python", "retry_failed_delivery"], + ["Invoice.java", "class Invoice { public boolean isOverdue(int days) { return days > 30; } }", "java", "Invoice"], + ["Status.vue", '\n', "vue", undefined], + ] as const + it.each(cases)("preserves metadata in %s without padding", async (file, content, language, symbol) => { + expect(content.length).toBeLessThan(200) + for (const chunks of [await Chunking.chunkAsync(content, file), (await Chunking.analyzeAsync(content, file)).chunks]) { + expect(chunks.length).toBeGreaterThan(0) + expect(chunks.some((chunk) => chunk.language === language)).toBe(true) + if (symbol) expect(chunks.some((chunk) => chunk.symbolName === symbol)).toBe(true) + } + }) +}) + describe("Chunking oversized content", () => { it("splits very large single-line code into safe chunks", () => { const content = `const payload = "${"a".repeat(getGeneralChunkLimits().max * 2)}";` diff --git a/packages/core/src/semantic/chunking.ts b/packages/core/src/semantic/chunking.ts index 9e47e20..2a25e67 100644 --- a/packages/core/src/semantic/chunking.ts +++ b/packages/core/src/semantic/chunking.ts @@ -335,7 +335,7 @@ export namespace Chunking { if (isBoundary && currentChunk.length > 0 && !inBlock) { // Save current chunk const chunkContent = currentChunk.join("\n") - if (chunkContent.length >= getMinChunkSize()) { + if (chunkContent.trim().length > 0) { chunks.push({ content: chunkContent, startLine: chunkStartLine + 1, @@ -368,7 +368,7 @@ export namespace Chunking { // Don't forget the last chunk if (currentChunk.length > 0) { const chunkContent = currentChunk.join("\n") - if (chunkContent.length >= getMinChunkSize()) { + if (chunkContent.trim().length > 0) { chunks.push({ content: chunkContent, startLine: chunkStartLine + 1, @@ -378,7 +378,10 @@ export namespace Chunking { } } - const normalizedChunks = enforceMaxChunkSize(chunks) + const normalizedChunks = enforceMaxChunkSize(chunks.map((chunk) => ({ + ...chunk, + language: getLanguageForFile(filePath)?.id, + }))) log.info("chunked code file", { filePath, chunks: normalizedChunks.length }) return normalizedChunks } @@ -453,7 +456,7 @@ export namespace Chunking { * Chunk a file into semantic pieces (synchronous wrapper) */ export function chunk(content: string, filePath: string): Chunk[] { - if (content.length < getMinChunkSize()) { + if (!isCodeFile(filePath) && content.length < getMinChunkSize()) { return [ { content, @@ -476,7 +479,7 @@ export namespace Chunking { * Async version that uses tree-sitter when possible */ export async function chunkAsync(content: string, filePath: string): Promise { - if (content.length < getMinChunkSize()) { + if (!isCodeFile(filePath) && content.length < getMinChunkSize()) { return [ { content, @@ -497,16 +500,9 @@ export namespace Chunking { if (TreeSitterChunking.isSupported(filePath)) { try { const parsed = await TreeSitterChunking.analyze(content, filePath) - const chunks = content.length < getMinChunkSize() - ? [{ - content, - startLine: 1, - endLine: parsed.lines.length, - type: "code" as const, - }] - : enforceMaxChunkSize( - parsed.chunks.length > 0 ? parsed.chunks : chunkCodeRegex(content, filePath), - ) + const chunks = enforceMaxChunkSize( + parsed.chunks.length > 0 ? parsed.chunks : chunkCodeRegex(content, filePath), + ) const collapsibleRegions = parsed.tree ? TreeShaker.findCollapsibleRegions(parsed.tree, parsed.lines) : [] diff --git a/packages/core/src/semantic/code-graph.test.ts b/packages/core/src/semantic/code-graph.test.ts index 18dfbb6..6d61768 100644 --- a/packages/core/src/semantic/code-graph.test.ts +++ b/packages/core/src/semantic/code-graph.test.ts @@ -25,6 +25,17 @@ function row(symbolName: string, content: string, startLine: number) { } describe("CodeGraph", () => { + it("attributes nested calls to the smallest enclosing symbol", async () => { + listDocuments.mockResolvedValue([ + { content: "class Service {\n run() {\n target()\n }\n}", metadata: { file: "src/app.ts", startLine: 1, endLine: 5, symbolName: "Service" } }, + { content: "run() {\n target()\n }", metadata: { file: "src/app.ts", startLine: 2, endLine: 4, symbolName: "run", parentScope: "Service" } }, + row("target", "return true", 20), + ]) + const { CodeGraph } = await import("./code-graph.js") + const references = (await CodeGraph.findReferences("target")).references + expect(references).toHaveLength(1) + expect(references[0]).toMatchObject({ from: "run", callLine: 3, resolution: "ast-name" }) + }) it("classifies only scheduler targets, ignoring comments and strings", async () => { listDocuments.mockResolvedValue([ row("run", 'validate(); ctx.scheduler.runAfter(0, internal.jobs.deliver, {}); // cron validate()\nconst note = "scheduler validate()";', 1), diff --git a/packages/core/src/semantic/code-graph.ts b/packages/core/src/semantic/code-graph.ts index e179cab..b15fe55 100644 --- a/packages/core/src/semantic/code-graph.ts +++ b/packages/core/src/semantic/code-graph.ts @@ -14,6 +14,8 @@ export namespace CodeGraph { toId: string kind: "call" | "import" | "inheritance" | "component-usage" | "hook-usage" | "convex-api" | "route-invocation" | "scheduled-function" | "schema-table" confidence: "high" | "medium" + resolution?: "ast-relative-import" | "ast-name" | "metadata-name" | "synthetic" | "syntax-inheritance" + callLine?: number location: Location targetLocation: Location } @@ -31,7 +33,7 @@ export namespace CodeGraph { "if", "for", "while", "switch", "catch", "function", "return", "typeof", "new", "super", "import", ]) let cached: { key: string; snapshot: Snapshot } | undefined - const CACHE_VERSION = 3 + const CACHE_VERSION = 4 type PersistedSnapshot = { version: number @@ -163,6 +165,7 @@ export namespace CodeGraph { const symbols = new Map() const nodes = new Map() const nodesByName = new Map() + const nodesByFile = new Map() const nodeByDocument = new Map() const symbolGroups = new Map() @@ -193,6 +196,7 @@ export namespace CodeGraph { const location = createLocation(metadata, symbol) const node = { id: location.id, name: symbol, location } nodes.set(node.id, node) + nodesByFile.set(location.file, [...(nodesByFile.get(location.file) ?? []), node]) nodesByName.set(symbol, [...(nodesByName.get(symbol) ?? []), node]) symbols.set(symbol, [...(symbols.get(symbol) ?? []), location]) for (const row of cluster) nodeByDocument.set(row, node) @@ -205,7 +209,7 @@ export namespace CodeGraph { const seenReferences = new Set() let unresolvedEdges = 0 let ambiguousEdges = 0 - const addReference = (source: Node, target: Node, kind: Reference["kind"], confidence: "high" | "medium") => { + const addReference = (source: Node, target: Node, kind: Reference["kind"], confidence: "high" | "medium", resolution?: Reference["resolution"], callLine?: number) => { if (target.id === source.id) return const key = `${source.id}\0${target.id}\0${kind}` if (seenReferences.has(key)) return @@ -220,6 +224,8 @@ export namespace CodeGraph { toId: target.id, kind, confidence, + resolution, + callLine, location: source.location, targetLocation: target.location, }) @@ -235,7 +241,7 @@ export namespace CodeGraph { if (!source) continue const file = source.location.file - let calls: Array<{ target: string; scheduled: boolean; module?: string }> + let calls: Array<{ target: string; scheduled: boolean; module?: string; line?: number }> if (TreeSitterChunking.isSupported(file)) { if (!callsByFile.has(file)) { const absolute = path.resolve(resolved.root, file) @@ -247,11 +253,16 @@ export namespace CodeGraph { const fullCalls = await callsByFile.get(file) calls = fullCalls ? fullCalls.filter((call) => call.line >= Number(row.metadata.startLine) && call.line <= Number(row.metadata.endLine)) - : await TreeSitterChunking.graphCalls(row.content, file) + : (await TreeSitterChunking.graphCalls(row.content, file)).map((call) => ({ ...call, line: call.line + Number(row.metadata.startLine) - 1 })) } else { calls = extractPersistedCalls(row.content, row.metadata.calls, sourceName).map((target) => ({ target, scheduled: false })) } - for (const { target: targetName, scheduled, module } of calls) { + for (const { target: targetName, scheduled, module, line } of calls) { + // A containing class/file chunk must not also own its method's calls. + if (line !== undefined && (nodesByFile.get(file) ?? []).some((node) => node.id !== source.id + && node.location.startLine >= source.location.startLine && node.location.endLine <= source.location.endLine + && node.location.endLine - node.location.startLine < source.location.endLine - source.location.startLine + && node.location.startLine <= line && node.location.endLine >= line)) continue let targetCandidates = nodesByName.get(targetName) ?? nodesByName.get(targetName.split(".").at(-1) ?? "") ?? [] if (module) { const modulePath = path.posix.normalize(path.posix.join(path.posix.dirname(file.replace(/\\/g, "/")), module)).replace(/\.[cm]?[jt]sx?$/, "") @@ -266,12 +277,13 @@ export namespace CodeGraph { else unresolvedEdges++ continue } - addReference(source, resolvedTarget.target, scheduled ? "scheduled-function" : classifyEdge(targetName, row.content), resolvedTarget.confidence!) + addReference(source, resolvedTarget.target, scheduled ? "scheduled-function" : classifyEdge(targetName, row.content), resolvedTarget.confidence!, + module ? "ast-relative-import" : TreeSitterChunking.isSupported(file) ? "ast-name" : "metadata-name", line) } for (const match of row.content.matchAll(/\bextends\s+([A-Za-z_$][A-Za-z0-9_$]*)/g)) { const target = resolveTarget(source, nodesByName.get(match[1]) ?? []) - if (target.target) addReference(source, target.target, "inheritance", target.confidence!) + if (target.target) addReference(source, target.target, "inheritance", target.confidence!, "syntax-inheritance") else if (target.ambiguous) ambiguousEdges++ else unresolvedEdges++ } @@ -290,7 +302,7 @@ export namespace CodeGraph { target = { id, name: synthetic.name, location } nodes.set(id, target) } - addReference(source, target, synthetic.kind, "high") + addReference(source, target, synthetic.kind, "high", "synthetic") } } diff --git a/packages/core/src/semantic/duplicate-detector.test.ts b/packages/core/src/semantic/duplicate-detector.test.ts index fb3df7b..a0d072d 100644 --- a/packages/core/src/semantic/duplicate-detector.test.ts +++ b/packages/core/src/semantic/duplicate-detector.test.ts @@ -82,6 +82,22 @@ vi.mock("./lancedb.js", () => ({ })) describe("DuplicateDetector", () => { + it("uses memory for 513 candidates and resumes to the same groups as a full run", async () => { + const { DuplicateDetector } = await import("./duplicate-detector.js") + const { VectorStore } = await import("./lancedb.js") + rows = Array.from({ length: 513 }, (_, i) => ({ ...baseRows[0], id: `memory-${i}`, metadata: { ...baseRows[0].metadata, file: `src/memory-${i}.ts` } })) + vi.mocked(VectorStore.searchByVector).mockClear() + const controller = new AbortController() + const options = { path: process.cwd(), minLines: 1, ignoreAcceptablePatterns: true } + const partial = await DuplicateDetector.detect({ ...options, signal: controller.signal, + onProgress: ({ current }) => { if (current === 17) controller.abort() } }) + expect(partial.summary.resumeCursor).toBe(17) + expect(partial.metrics?.strategy).toBe("memory-cosine") + const resumed = await DuplicateDetector.detect({ ...options, resumeCursor: partial.summary.resumeCursor }) + const full = await DuplicateDetector.detect(options) + expect(resumed.duplicates).toEqual(full.duplicates) + expect(VectorStore.searchByVector).not.toHaveBeenCalled() + }) beforeEach(() => { rows = [...baseRows] searchDelayMs = 0 diff --git a/packages/core/src/semantic/duplicate-detector.ts b/packages/core/src/semantic/duplicate-detector.ts index f2e7a5d..01e7c68 100644 --- a/packages/core/src/semantic/duplicate-detector.ts +++ b/packages/core/src/semantic/duplicate-detector.ts @@ -107,6 +107,7 @@ export namespace DuplicateDetector { processedCandidates?: number elapsedMs?: number } + metrics?: { loadMs: number; prepareMs: number; neighborsMs: number; strategy: "memory-cosine" | "vector-store" } duplicates: DuplicateGroup[] acceptableDuplicates?: DuplicateGroup[] } @@ -525,24 +526,6 @@ export namespace DuplicateDetector { return files } - /** - * Calcular similaridade entre dois vetores (cosine similarity) - */ - function cosineSimilarity(vec1: number[], vec2: number[]): number { - let dotProduct = 0 - let norm1 = 0 - let norm2 = 0 - - for (let i = 0; i < vec1.length; i++) { - dotProduct += vec1[i] * vec2[i] - norm1 += vec1[i] * vec1[i] - norm2 += vec2[i] * vec2[i] - } - - const denominator = Math.sqrt(norm1) * Math.sqrt(norm2) - return denominator > 0 ? dotProduct / denominator : 0 - } - const TOKENIZE_REGEX = /[A-Za-z_$][A-Za-z0-9_$]*|\d+|[^\s]/g function tokenize(text: string): string[] { @@ -814,6 +797,7 @@ export namespace DuplicateDetector { ], }) + const loadedAt = Date.now() const subdirPrefix = resolvedIndex.subdirPrefix const isInScopedPath = (file: string) => { if (!subdirPrefix) return true @@ -955,7 +939,7 @@ export namespace DuplicateDetector { const maxNeighbors = Math.min(30, Math.max(5, candidates.length)) const resumeCursor = Math.min(candidates.length, Math.max(0, options.resumeCursor ?? 0)) const fingerprint = createHash("sha256").update(JSON.stringify({ - version: 1, root: resolvedIndex.root, updatedAt: meta.updatedAt, embeddings: meta.embeddings, + version: 2, root: resolvedIndex.root, updatedAt: meta.updatedAt, embeddings: meta.embeddings, candidates: candidates.map((candidate) => candidate.id), thresholds, normalizeIdentifiers, crossFileOnly: options.crossFileOnly, crossLanguage: options.crossLanguage, })).digest("hex") @@ -972,6 +956,17 @@ export namespace DuplicateDetector { pairs.set(key, pair) } } + // Bound quadratic work to moderate candidate sets. Normalize once instead of + // recomputing both vector norms for every pair, and reuse symmetric distances. + const memoryCosine = candidates.length <= 4096 && VectorStore.getDistanceMetric(meta) === "cosine" + const unitVectors = memoryCosine ? candidates.map(({ vector }) => { + const norm = Math.sqrt(vector.reduce((sum, value) => sum + value * value, 0)) + return Float64Array.from(vector, (value) => norm ? value / norm : 0) + }) : [] + const distances = memoryCosine ? new Float64Array(candidates.length * (candidates.length - 1) / 2).fill(NaN) : undefined + const preparedAt = Date.now() + const metrics = { loadMs: loadedAt - startTime, prepareMs: preparedAt - loadedAt, neighborsMs: 0, + strategy: memoryCosine ? "memory-cosine" as const : "vector-store" as const } const deadline = options.timeoutMs ? startTime + options.timeoutMs : Number.POSITIVE_INFINITY let processedCandidates = 0 let timedOut = false @@ -979,6 +974,7 @@ export namespace DuplicateDetector { options.onProgress?.({ phase: "neighbors", current: resumeCursor, total: candidates.length, elapsedMs: Date.now() - startTime }) for (let candidateIndex = resumeCursor; candidateIndex < candidates.length; candidateIndex++) { + if (memoryCosine && (candidateIndex - resumeCursor) % 16 === 0) await new Promise((resolve) => setImmediate(resolve)) if (options.signal?.aborted) { aborted = true break @@ -995,11 +991,33 @@ export namespace DuplicateDetector { : options.signal ?? timeoutSignal let neighbors: Awaited> try { - neighbors = candidates.length <= 512 && VectorStore.getDistanceMetric(meta) === "cosine" - ? candidates.filter((other) => eligible(candidate, other)).map((other) => ({ - id: other.id, distance: 1 - cosineSimilarity(candidate.vector, other.vector), - })).sort((a, b) => a.distance - b.distance || a.id.localeCompare(b.id)).slice(0, maxNeighbors) as typeof neighbors - : await VectorStore.searchByVector(collection, candidate.vector, { + if (memoryCosine) { + const nearest: Array<{ id: string; distance: number }> = [] + for (let otherIndex = 0; otherIndex < candidates.length; otherIndex++) { + if (otherIndex % 256 === 0 && (options.signal?.aborted || Date.now() >= deadline)) { + throw new Error("Duplicate neighbor scan interrupted") + } + const other = candidates[otherIndex] + if (!eligible(candidate, other)) continue + const high = Math.max(candidateIndex, otherIndex) + const low = Math.min(candidateIndex, otherIndex) + const offset = high * (high - 1) / 2 + low + let distance = distances![offset] + if (Number.isNaN(distance)) { + const left = unitVectors[candidateIndex], right = unitVectors[otherIndex] + let dot = 0 + if (left.length === right.length) for (let i = 0; i < left.length; i++) dot += left[i] * right[i] + distance = 1 - Math.max(-1, Math.min(1, dot)) + distances![offset] = distance + } + const entry = { id: other.id, distance } + const position = nearest.findIndex((value) => distance < value.distance || (distance === value.distance && entry.id.localeCompare(value.id) < 0)) + if (position >= 0) nearest.splice(position, 0, entry) + else if (nearest.length < maxNeighbors) nearest.push(entry) + if (nearest.length > maxNeighbors) nearest.pop() + } + neighbors = nearest as typeof neighbors + } else neighbors = await VectorStore.searchByVector(collection, candidate.vector, { limit: maxNeighbors, filters: { ...neighborFilters, @@ -1063,6 +1081,7 @@ export namespace DuplicateDetector { } } + metrics.neighborsMs = Date.now() - preparedAt const nextCursor = resumeCursor + processedCandidates < candidates.length ? resumeCursor + processedCandidates : undefined @@ -1100,6 +1119,7 @@ export namespace DuplicateDetector { processedCandidates, elapsedMs: Date.now() - startTime, }, + metrics, duplicates: [], } } @@ -1255,6 +1275,7 @@ export namespace DuplicateDetector { processedCandidates, elapsedMs: elapsed, }, + metrics, duplicates, acceptableDuplicates: acceptableDuplicates.length > 0 ? acceptableDuplicates : undefined, } diff --git a/packages/core/src/semantic/indexer.test.ts b/packages/core/src/semantic/indexer.test.ts index 6bb52e0..d2262b2 100644 --- a/packages/core/src/semantic/indexer.test.ts +++ b/packages/core/src/semantic/indexer.test.ts @@ -31,7 +31,7 @@ const chunkAsync = vi.fn() const analyzeAsync = vi.fn() const addOverlap = vi.fn((chunks) => chunks) const testChunkingSignature = { - version: 5, provider: "openai", model: "test-model", dimension: 3, + version: 6, provider: "openai", model: "test-model", dimension: 3, modelMaxTokens: 8192, usableModelTokens: 8028, maxChars: 28000, minChars: 200, overlapChars: 512, simpleChars: 16384, mediumChars: 16384, complexChars: 16384, @@ -285,7 +285,7 @@ describe("Indexer incremental updates", () => { it("requires a rebuild instead of mixing chunk policies during watched updates", async () => { const meta = await readIndexMeta() - readIndexMeta.mockResolvedValue({ ...meta, chunking: { ...testChunkingSignature, version: 4 } }) + readIndexMeta.mockResolvedValue({ ...meta, chunking: { ...testChunkingSignature, version: 5 } }) const { Indexer } = await import("./indexer.js") await expect(Indexer.updateFile("src/a.ts")).rejects.toThrow("Chunking policy changed") expect(embedDocumentsReusingFile).not.toHaveBeenCalled() diff --git a/packages/core/src/tool/agent-output.test.ts b/packages/core/src/tool/agent-output.test.ts index dd0a178..b94ef1b 100644 --- a/packages/core/src/tool/agent-output.test.ts +++ b/packages/core/src/tool/agent-output.test.ts @@ -1,7 +1,23 @@ import { describe, expect, it } from "vitest" -import { projectAgentResponse } from "./agent-output.js" +import { projectAgentResponse, enforceAgentOutputBudget } from "./agent-output.js" describe("shared agent output projection", () => { + it("keeps ranking strength separate from answer sufficiency", () => { + const response = projectAgentResponse({ command: "search", results: [{ file: "test.ts", score: 0.9, confidence: "high" }] }, { detail: "diagnostic" }) + expect(response.answerSufficiency).toBe("not-assessed") + expect(response.results[0].diagnostic.rankingStrength).toBe("high") + expect(response.results[0].diagnostic).not.toHaveProperty("confidence") + }) + + it.each([false, true])("enforces serialized bytes, including UTF-8 and pretty=%s", (pretty) => { + const response = enforceAgentOutputBudget({ command: "search", status: "complete", budget: { maxBytes: 512 }, + results: [{ file: "ação.ts", lines: [1, 100], content: "ação 🐈\n".repeat(100) }] }, { pretty, trailingNewline: true }) + const serialized = JSON.stringify(response, null, pretty ? 2 : undefined) + "\n" + expect(Buffer.byteLength(serialized)).toBeLessThanOrEqual(512) + expect(response.status).toBe("incomplete") + expect(response.budget.usedBytes).toBe(Buffer.byteLength(serialized)) + expect(JSON.parse(serialized)).toEqual(response) + }) it("makes grouped summaries compact and directly expandable", () => { const projected = projectAgentResponse({ command: "survey", diff --git a/packages/core/src/tool/agent-output.ts b/packages/core/src/tool/agent-output.ts index 14b453f..5e14ac1 100644 --- a/packages/core/src/tool/agent-output.ts +++ b/packages/core/src/tool/agent-output.ts @@ -150,7 +150,8 @@ export function projectAgentResult(entry: any, rank: number, options: AgentProje semanticKind: entry.semanticKind, framework: entry.framework, fileRole: entry.fileRole ?? entry.metadata?.fileRole, - confidence: entry.confidence, + rankingStrength: entry.rankingStrength ?? entry.confidence, + answerSufficiency: entry.answerSufficiency ?? "not-assessed", weakMatch: entry.isWeakMatch, why: entry.whyMatched, filterMatches: entry.filterMatches, @@ -185,6 +186,7 @@ export function projectSearchAgentResponse(raw: any, options: AgentProjectionOpt : [] const response: Record = { ...baseEnvelope(raw, options), + answerSufficiency: "not-assessed", ...(raw.coverage ? { coverage: defined({ changedFiles: raw.coverage.changedFiles, @@ -336,6 +338,7 @@ export function projectDuplicateAgentResponse(raw: any, options: AgentProjection status: raw.status ?? "complete", warnings: projectWarnings(raw.warnings), summary, + ...(diagnostics ? { diagnostic: { metrics: raw.metrics } } : {}), duplicates: (raw.duplicates ?? []).map((group: any) => defined({ level: group.level, similarity: group.similarity, @@ -373,8 +376,13 @@ export function projectGraphAgentResponse(raw: any, options: AgentProjectionOpti definitions: (raw.definitions ?? []).map(projectGraphLocation), references: (raw.references ?? []).map((reference: any) => defined({ fromId: reference.fromId, + toId: reference.toId, + source: reference.location ? projectGraphLocation(reference.location) : undefined, + target: reference.targetLocation ? projectGraphLocation(reference.targetLocation) : undefined, kind: reference.kind, confidence: reference.confidence, + resolution: reference.resolution, + callLine: reference.callLine, })), truncated: raw.truncated ?? false, graphCoverage: raw.metrics?.graphCoverage, @@ -402,6 +410,7 @@ export function projectGraphAgentResponse(raw: any, options: AgentProjectionOpti if (diagnostics) { response.diagnostic = defined({ metrics: raw.metrics, + coverageDefinition: "resolved edges / (resolved + unresolved + ambiguous); includes synthetic imports/table edges, not repository completeness", path: Array.isArray(raw.pathNodes) ? raw.pathNodes.map(projectGraphLocation) : undefined, }) } @@ -479,6 +488,7 @@ function truncateContent(entry: any, maxCharacters: number): any { content, contentTruncated: content.length < original.length, integrity: content.length < original.length ? "partial" : entry.integrity, + ...(entry.snippetIntegrity !== undefined ? { snippetIntegrity: content.length < original.length ? "partial" : entry.snippetIntegrity } : {}), } } diff --git a/packages/core/src/tool/search-schema.ts b/packages/core/src/tool/search-schema.ts index 01ee233..3c9a856 100644 --- a/packages/core/src/tool/search-schema.ts +++ b/packages/core/src/tool/search-schema.ts @@ -49,6 +49,7 @@ export const CommonSearchShape = { export const SenseGrepParametersSchema = z.object({ ...CommonSearchShape, + maxOutputBytes: z.number().int().min(256).max(100_000_000).optional().describe("Maximum serialized JSON response bytes (minimum 256)"), limit: z.number().int().positive().max(500).optional().describe("Maximum results (default: 10)"), maxPerFile: z.number().int().nonnegative().optional().describe("Maximum results per file (default: 2)"), maxPerSymbol: z.number().int().nonnegative().optional().describe("Maximum results per symbol (default: 2)"), diff --git a/packages/core/src/tool/sensegrep-cluster.test.ts b/packages/core/src/tool/sensegrep-cluster.test.ts index 5bdf42f..4537d51 100644 --- a/packages/core/src/tool/sensegrep-cluster.test.ts +++ b/packages/core/src/tool/sensegrep-cluster.test.ts @@ -48,6 +48,19 @@ vi.mock("../project/instance.js", () => ({ })) describe("SenseGrepClusterTool", () => { + it("does not merge a similarity chain into one cluster", async () => { + const { buildInitialClusters } = await import("./sensegrep-cluster.js") + const nodes = [0, Math.PI / 6, Math.PI / 3].map((angle, index) => ({ + file: `src/${index}.ts`, startLine: 1, endLine: 3, content: "", semanticScore: 1, + metadata: { symbolType: "function" }, vector: [Math.cos(angle), Math.sin(angle)], + importHints: [], symbolHints: [], domainLabel: "business logic", + })) + const groups = buildInitialClusters(nodes, 0.8) + expect(groups.length).toBeGreaterThan(1) + expect(groups.flat()).toHaveLength(3) + expect(groups.some((group) => group.includes(nodes[0]) && group.includes(nodes[2]))).toBe(false) + expect(buildInitialClusters([...nodes].reverse(), 0.8)).toEqual(groups) + }) beforeEach(() => { vi.clearAllMocks() readIndexMeta.mockResolvedValue({ diff --git a/packages/core/src/tool/sensegrep-cluster.ts b/packages/core/src/tool/sensegrep-cluster.ts index 45537aa..4b0675c 100644 --- a/packages/core/src/tool/sensegrep-cluster.ts +++ b/packages/core/src/tool/sensegrep-cluster.ts @@ -9,7 +9,7 @@ import { deriveDomainLabel, formatGroupedResultHeader, formatRepresentativeSnippets, - getDominantSymbolPhrases, + getGroupTitleSignal, getGroupingReasons, getImportHints, getQueryTokens, @@ -65,33 +65,7 @@ type ClusterGroup = { dominantSymbolTypes: string[] } -const GENERIC_TITLE_SIGNALS = new Set(["api", "client", "clients", "service", "services", "types", "contracts", "model", "models"]) - -class UnionFind { - private parent = new Map() - - constructor(size: number) { - for (let index = 0; index < size; index += 1) { - this.parent.set(index, index) - } - } - - find(value: number): number { - const parent = this.parent.get(value) - if (parent === undefined || parent === value) return value - const root = this.find(parent) - this.parent.set(value, root) - return root - } - - union(a: number, b: number) { - const rootA = this.find(a) - const rootB = this.find(b) - if (rootA !== rootB) { - this.parent.set(rootB, rootA) - } - } -} +const GENERIC_TITLE_SIGNALS = new Set(["api", "client", "clients", "service", "services", "types", "contracts", "model", "models", "react", "convex", "convex-test", "vitest", "jest", "errorhelpers"]) function combinedSimilarity(a: ClusterNode, b: ClusterNode): number { const weighted: Array<[number, number]> = [] @@ -115,34 +89,22 @@ function combinedSimilarity(a: ClusterNode, b: ClusterNode): number { return weighted.reduce((sum, [weight, score]) => sum + weight * score, 0) / totalWeight } -function averagePairwiseSimilarity(cluster: ClusterNode[], candidate: ClusterNode): number { - if (cluster.length === 0) return 0 - let total = 0 - for (const member of cluster) total += combinedSimilarity(member, candidate) - return total / cluster.length -} - -function buildInitialClusters(nodes: ClusterNode[], threshold: number): ClusterNode[][] { - if (nodes.length === 0) return [] - const unionFind = new UnionFind(nodes.length) - - for (let a = 0; a < nodes.length; a += 1) { - for (let b = a + 1; b < nodes.length; b += 1) { - if (combinedSimilarity(nodes[a], nodes[b]) >= threshold) { - unionFind.union(a, b) - } +export function buildInitialClusters(nodes: ClusterNode[], threshold: number): ClusterNode[][] { + // Complete-link admission prevents A~B and B~C from implying A~C. + // Stable relevance/source ordering makes ties reproducible across index reads. + const ordered = [...nodes].sort((a, b) => b.semanticScore - a.semanticScore || a.file.localeCompare(b.file) || a.startLine - b.startLine) + const groups: ClusterNode[][] = [] + for (const node of ordered) { + let best: ClusterNode[] | undefined + let bestScore = -Infinity + for (const group of groups) { + const similarity = Math.min(...group.map((member) => combinedSimilarity(member, node))) + if (similarity >= threshold && similarity > bestScore) { best = group; bestScore = similarity } } + if (best) best.push(node) + else groups.push([node]) } - - const groups = new Map() - for (let index = 0; index < nodes.length; index += 1) { - const root = unionFind.find(index) - const cluster = groups.get(root) ?? [] - cluster.push(nodes[index]) - groups.set(root, cluster) - } - - return [...groups.values()] + return groups } function attachSmallClusters( @@ -157,7 +119,7 @@ function attachSmallClusters( if (largeClusters.length === 0 || smallClusters.length === 0) return clusters - const attachThreshold = Math.max(0.55, threshold - 0.08) + const attachThreshold = threshold const leftovers: ClusterNode[][] = [] for (const cluster of smallClusters) { @@ -171,7 +133,7 @@ function attachSmallClusters( let bestScore = 0 for (let index = 0; index < largeClusters.length; index += 1) { - const score = averagePairwiseSimilarity(largeClusters[index], candidate) + const score = Math.min(...largeClusters[index].map((member) => combinedSimilarity(member, candidate))) if (score > bestScore) { bestScore = score bestClusterIndex = index @@ -189,15 +151,9 @@ function attachSmallClusters( } function chooseClusterTitle(cluster: ClusterNode[], query: string): string { - const queryTokenSet = new Set(getQueryTokens(query)) const domainHints = topCounts(cluster.map((member) => member.domainLabel), 2) const strongestDomain = domainHints.find((value) => value !== "related code") - const importHints = topCounts(cluster.flatMap((member) => member.importHints), 2) - const symbolHints = topCounts(cluster.flatMap((member) => member.symbolHints), 3, queryTokenSet) - const symbolPhrases = getDominantSymbolPhrases(cluster, query, 2, false) - const importSignal = importHints.find((hint) => !GENERIC_TITLE_SIGNALS.has(hint)) ?? importHints[0] - const strongestSignal = symbolPhrases[0] ?? symbolHints[0] ?? - (importSignal && !GENERIC_TITLE_SIGNALS.has(importSignal) ? importSignal : undefined) + const strongestSignal = getGroupTitleSignal(cluster, query) if (strongestDomain && !strongestDomain.startsWith("domain /")) { if (strongestSignal) return `${strongestDomain} / ${strongestSignal}` @@ -254,7 +210,7 @@ function buildClusterGroups( const queryTokenSet = new Set(getQueryTokens(query)) const nodes: ClusterNode[] = results.map((result) => ({ ...result, - importHints: getImportHints(result.metadata), + importHints: getImportHints(result.metadata).filter((hint) => !GENERIC_TITLE_SIGNALS.has(hint)), symbolHints: getSymbolTokens(result.metadata).filter((token) => !queryTokenSet.has(token)), domainLabel: deriveDomainLabel(result), })) @@ -262,6 +218,12 @@ function buildClusterGroups( const initial = buildInitialClusters(nodes, threshold) const normalized = attachSmallClusters(initial, threshold, minClusterSize) const groups = normalized.map((cluster) => summarizeCluster(cluster, query)) + const titles = new Map() + for (const group of groups) titles.set(group.title, (titles.get(group.title) ?? 0) + 1) + for (const group of groups) if (titles.get(group.title)! > 1) { + const first = [...group.members].sort((a, b) => a.file.localeCompare(b.file) || a.startLine - b.startLine)[0] + group.title += ` - ${first.file}:${first.startLine}` + } return groups.sort((a, b) => { const rankA = a.score * Math.log2(a.members.length + 1) diff --git a/packages/core/src/tool/sensegrep-context.ts b/packages/core/src/tool/sensegrep-context.ts index 7171de8..9a17de1 100644 --- a/packages/core/src/tool/sensegrep-context.ts +++ b/packages/core/src/tool/sensegrep-context.ts @@ -249,7 +249,7 @@ export const SenseGrepContextTool = Tool.define("sensegrep-context", { ...result, schemaVersion: 1, command: params.commandName ?? "context", - status: finalCoverage?.exhaustive === false ? "incomplete" : "complete", + status: finalCoverage?.exhaustive === false || (result as any).status === "incomplete" ? "incomplete" : "complete", title: `Context: ${params.query}`, metadata: { ...result.metadata, diff --git a/packages/core/src/tool/sensegrep-pipeline.test.ts b/packages/core/src/tool/sensegrep-pipeline.test.ts index c4e6fd5..d379089 100644 --- a/packages/core/src/tool/sensegrep-pipeline.test.ts +++ b/packages/core/src/tool/sensegrep-pipeline.test.ts @@ -15,6 +15,7 @@ import { decodeResultId, toStructuredSearchResult, deriveDomainLabel, + getGroupTitleSignal, } from "./sensegrep-pipeline.js" describe("sensegrep pipeline result metadata", () => { @@ -164,6 +165,46 @@ describe("hybrid retrieval ranking", () => { expect(selected.estimatedTokens).toBeLessThanOrEqual(150) }) + it("selects a fitting implementation beyond the result limit", () => { + const selected = selectWithinTokenBudget([ + result("large.ts", 0.95, "x".repeat(6000)), + result("implementation.ts", 0.9, "refresh token"), + ], 100, "refresh token", 1) + expect(selected.results.map((r) => r.file)).toEqual(["implementation.ts"]) + }) + + it("does not let tiny wrappers displace a substantially stronger implementation", () => { + const selected = selectWithinTokenBudget([ + result("rules.ts", 0.95, "check balance " + "x".repeat(1200)), + result("wrapper.ts", 0.7, "check balance"), + ], 350, "check balance") + expect(selected.results[0].file).toBe("rules.ts") + }) + + it("never exceeds a budget smaller than metadata alone", () => { + expect(selectWithinTokenBudget([result("a.ts", 0.9, "x")], 1).results).toEqual([]) + }) + + it("reserves context for implementations but respects an explicit test purpose", () => { + const test = { ...result("rule.test.ts", 0.9, "x".repeat(250)), metadata: { fileRole: "test" } } + const implementation = { ...result("rule.ts", 0.8, "x".repeat(250)), metadata: { fileRole: "implementation" } } + expect(selectWithinTokenBudget([test, implementation], 100).results[0].file).toBe("rule.ts") + expect(selectWithinTokenBudget([test, implementation], 100, "", 2, "test").results[0].file).toBe("rule.test.ts") + }) + + it("retains a helper referenced by relevant code that is too large for the budget", () => { + const selected = selectWithinTokenBudget([ + result("workflow.ts", 0.9, "return preservePendingContext();" + "x".repeat(3000), "processWorkflow"), + result("format.ts", 0.8, "x".repeat(150), "formatPayload"), + result("guard.ts", 0.75, "if (expected !== actual) return false", "preservePendingContext"), + ], 100, "", 1) + expect(selected.results[0].file).toBe("guard.ts") + }) + + it("uses a descriptive filename rather than shared test framework imports", () => { + expect(getGroupTitleSignal([{ ...result("tests/calendarWebhook.test.ts", 0.9, ""), metadata: { imports: "convex-test,vitest" } }], "calendar")).toBe("calendar webhook") + }) + it("keeps compact complete implementations instead of exhausting context on a large prefix", () => { const selected = selectWithinTokenBudget([ result("large.ts", 0.95, "x".repeat(6000), "SessionProvider"), diff --git a/packages/core/src/tool/sensegrep-pipeline.ts b/packages/core/src/tool/sensegrep-pipeline.ts index a6620e5..3d17f07 100644 --- a/packages/core/src/tool/sensegrep-pipeline.ts +++ b/packages/core/src/tool/sensegrep-pipeline.ts @@ -54,7 +54,10 @@ export type StructuredSearchResult = { semanticKind?: string framework?: string fileRole?: string + /** @deprecated Ranking heuristic only; use rankingStrength. */ confidence: "high" | "medium" | "low" + rankingStrength: "high" | "medium" | "low" + answerSufficiency: "not-assessed" isWeakMatch: boolean whyMatched: string[] filterMatches?: Record @@ -700,52 +703,67 @@ export function estimateResultTokens(result: Pick sum + estimateResultTokens(result), 0) } + const selected = results.slice(0, maxResults) + return { results: selected, estimatedTokens: selected.reduce((sum, result) => sum + estimateResultTokens(result), 0) } } const selected: WorkingResult[] = [] let estimatedTokens = 0 - // Prefer useful complete evidence over a large prefix that exhausts the pack. - // Preserve relevance, while rewarding compact, query-specific implementations. const queryTokens = getQueryTokens(query) - const ordered = [...results].sort((a, b) => { + const covered = new Set() + const strongest = Math.max(0, ...results.map((r) => r.rerankScore ?? r.semanticScore)) + const remaining = results.filter((r) => (r.rerankScore ?? r.semanticScore) >= strongest * 0.7) + const wantsTests = purpose === "test" || queryTokens.some((token) => ["test", "tests", "teste", "testes"].includes(token)) + const wantsContracts = queryTokens.some((token) => ["type", "types", "interface", "interfaces", "schema", "contract", "contrato", "tipos"].includes(token)) + const anchors = [...results].filter((r) => r.metadata.fileRole !== "test" && r.metadata.fileRole !== "contract") + .sort((a, b) => (b.rerankScore ?? b.semanticScore) - (a.rerankScore ?? a.semanticScore)).slice(0, 5) + const supported = new Set(results.filter((r) => { + const name = String(r.metadata.symbolName ?? "") + if (name.length < 5 || !/^[\w$]+$/.test(name)) return false + const reference = new RegExp(`\\b${name.replaceAll("$", "\\$")}\\b`) + return anchors.some((anchor) => anchor !== r && reference.test(anchor.content)) + })) + const terms = (r: WorkingResult) => { + const text = `${r.file} ${r.metadata.symbolName ?? ""} ${r.content}`.toLowerCase() + return queryTokens.filter((token) => text.includes(token)) + } + // Relevance remains the main signal. A bounded size penalty cannot let tiny, + // redundant wrappers displace the implementation merely because they are cheap. + while (remaining.length && selected.length < maxResults) { + const available = maxTokens - estimatedTokens + const fitting = remaining.filter((r) => estimateResultTokens(r) <= available) + if (!fitting.length) break const utility = (r: WorkingResult) => { - const coverage = queryTokens.length ? lexicalRelevance(queryTokens, r) : 0 - const executableWeight = ["type", "interface", "enum"].includes(String(r.metadata.symbolType)) ? 0.75 : 1 - return ((r.rerankScore ?? r.semanticScore) + coverage * 0.3) * executableWeight / Math.sqrt(Math.max(120, estimateResultTokens(r))) + const novelty = queryTokens.length ? terms(r).filter((token) => !covered.has(token)).length / queryTokens.length : 0 + const redundant = selected.some((s) => s.file === r.file && s.startLine <= r.endLine && r.startLine <= s.endLine) + const contractPenalty = !wantsContracts && (r.metadata.fileRole === "contract" || ["type", "interface", "enum"].includes(String(r.metadata.symbolType))) ? 0.2 : 0 + const testPenalty = !wantsTests && r.metadata.fileRole === "test" ? 0.18 : 0 + return (r.rerankScore ?? r.semanticScore) + novelty * 0.15 + (supported.has(r) ? 0.12 : 0) + - contractPenalty - testPenalty - (redundant ? 0.3 : 0) - 0.12 * Math.sqrt(estimateResultTokens(r) / maxTokens) } - return utility(b) - utility(a) || compareWorkingResults(a, b) - }) - // Only truncate if no complete candidate fits. Never sacrifice complete evidence - // just because the first ranked symbol is larger than the entire budget. - const strongest = Math.max(0, ...results.map((r) => r.rerankScore ?? r.semanticScore)) - const fitting = ordered.filter((r) => estimateResultTokens(r) <= maxTokens && (r.rerankScore ?? r.semanticScore) >= strongest * 0.7) - for (const result of fitting.length ? fitting : ordered.slice(0, 1)) { - let selectedResult = result - let tokens = estimateResultTokens(selectedResult) - if (selected.length === 0 && tokens > maxTokens) { - const metadataOverhead = selectedResult.file.length + String(selectedResult.metadata.symbolName ?? "").length + 80 - const maxContentCharacters = Math.max(0, maxTokens * 4 - metadataOverhead) - let content = selectedResult.content.slice(0, maxContentCharacters) + fitting.sort((a, b) => utility(b) - utility(a) || compareWorkingResults(a, b)) + const winner = fitting[0] + selected.push(winner) + estimatedTokens += estimateResultTokens(winner) + terms(winner).forEach((token) => covered.add(token)) + remaining.splice(remaining.indexOf(winner), 1) + } + // Only return partial source when no complete candidate fits. Preserve source + // bounds/IDs for expansion and explicitly mark the evidence as partial. + if (!selected.length && results.length && maxResults > 0) { + const first = results[0] + const overhead = first.file.length + String(first.metadata.symbolName ?? "").length + 80 + const characters = maxTokens * 4 - overhead + if (characters > 0) { + let content = first.content.slice(0, characters) const newline = content.lastIndexOf("\n") - if (newline >= Math.floor(maxContentCharacters / 2)) content = content.slice(0, newline) - selectedResult = { - ...selectedResult, - content, - contentTruncated: content.length < selectedResult.content.length, - metadata: { - ...selectedResult.metadata, - snippetIntegrity: "partial", - contentTruncated: true, - }, - } - tokens = estimateResultTokens(selectedResult) + if (newline >= characters / 2) content = content.slice(0, newline) + const partial = { ...first, content, contentTruncated: true, + metadata: { ...first.metadata, snippetIntegrity: "partial", contentTruncated: true } } + const tokens = estimateResultTokens(partial) + if (tokens <= maxTokens) { selected.push(partial); estimatedTokens = tokens } } - if (selected.length > 0 && estimatedTokens + tokens > maxTokens) continue - selected.push(selectedResult) - estimatedTokens += tokens - if (estimatedTokens >= maxTokens) break } return { results: selected, estimatedTokens } } @@ -1580,6 +1598,16 @@ export function getDominantSymbolPhrases( .map(([phrase]) => phrase) } +/** Prefer source/domain terms over ubiquitous framework and testing imports. */ +export function getGroupTitleSignal(results: WorkingResult[], query: string): string | undefined { + const generic = new Set(["api", "client", "service", "types", "contracts", "model", "models", "test", "tests", "convex", "react", "vitest", "jest", "errorhelpers", "helpers", "utils", "index"]) + const meaningful = (phrase: string) => splitIdentifier(phrase).some((token) => !generic.has(token)) + const symbols = getDominantSymbolPhrases(results, query, 10, false).filter(meaningful) + if (symbols.length) return symbols[0] + const filenames = topCounts(results.map((result) => splitIdentifier(path.basename(result.file).replace(/\.(?:test|spec)?\.?[cm]?[jt]sx?$/, "")).filter((token) => !generic.has(token)).join(" ")).filter(Boolean), 1) + return filenames[0] ?? topCounts(results.flatMap((r) => getImportHints(r.metadata)).filter(meaningful), 1)[0] +} + function scoreToConfidence(score: number): "high" | "medium" | "low" { if (score >= 0.55) return "high" if (score >= 0.25) return "medium" @@ -2175,6 +2203,8 @@ export function toStructuredSearchResult(result: WorkingResult): StructuredSearc framework: typeof metadata.framework === "string" && metadata.framework ? metadata.framework : undefined, fileRole: typeof metadata.fileRole === "string" && metadata.fileRole ? metadata.fileRole : undefined, confidence: result.confidence ?? scoreToConfidence(result.rerankScore ?? result.semanticScore), + rankingStrength: result.confidence ?? scoreToConfidence(result.rerankScore ?? result.semanticScore), + answerSufficiency: "not-assessed", isWeakMatch: result.isWeakMatch ?? (result.rerankScore ?? result.semanticScore) < 0.25, whyMatched: result.whyMatched ?? [], filterMatches: result.filterMatches, diff --git a/packages/core/src/tool/sensegrep-survey.ts b/packages/core/src/tool/sensegrep-survey.ts index 629b860..98b2a33 100644 --- a/packages/core/src/tool/sensegrep-survey.ts +++ b/packages/core/src/tool/sensegrep-survey.ts @@ -8,7 +8,7 @@ import { deriveDomainLabel, formatGroupedResultHeader, formatRepresentativeSnippets, - getDominantSymbolPhrases, + getGroupTitleSignal, getGroupingReasons, getImportHints, getQueryTokens, @@ -51,7 +51,6 @@ type SurveyGroup = { dominantSymbolTypes: string[] } -const GENERIC_TITLE_SIGNALS = new Set(["api", "client", "clients", "service", "services", "types", "contracts", "model", "models"]) function buildSurveyGroups(results: WorkingResult[], query: string): SurveyGroup[] { const queryTokenSet = new Set(getQueryTokens(query)) @@ -100,12 +99,7 @@ function getSurveyWhyGrouped(group: SurveyGroup): string[] { } function chooseSurveyTitle(group: SurveyGroup, query: string): string { - const symbolPhrases = getDominantSymbolPhrases(group.members, query, 2, false) - const symbolHints = topCounts(group.symbolHints, 2, new Set(getQueryTokens(query))) - const importHints = topCounts(group.importHints, 2) - const importSignal = importHints.find((hint) => !GENERIC_TITLE_SIGNALS.has(hint)) ?? importHints[0] - const strongestSignal = symbolPhrases[0] ?? symbolHints[0] ?? - (importSignal && !GENERIC_TITLE_SIGNALS.has(importSignal) ? importSignal : undefined) + const strongestSignal = getGroupTitleSignal(group.members, query) if (!strongestSignal) return group.title if (group.title.includes(strongestSignal)) return group.title diff --git a/packages/core/src/tool/sensegrep.ts b/packages/core/src/tool/sensegrep.ts index a74dedb..46eae0a 100644 --- a/packages/core/src/tool/sensegrep.ts +++ b/packages/core/src/tool/sensegrep.ts @@ -157,13 +157,13 @@ export const SenseGrepTool = Tool.define("sensegrep", { const diversifiedResults = diversifyResults(dedupedResults, { maxPerFile, maxPerSymbol }) // Take top results - const limitedResults = diversifiedResults.slice(0, limit) - const budgeted = selectWithinTokenBudget(limitedResults, params.maxTokens, params.query) + const budgeted = selectWithinTokenBudget(diversifiedResults, params.maxTokens, params.query, limit, params.purpose) const finalResults = budgeted.results finalResults.forEach((result, index) => { result.rankScore = Number(((finalResults.length - index) / Math.max(1, finalResults.length)).toFixed(6)) }) metrics.estimatedOutputTokens = budgeted.estimatedTokens + metrics.tokenBudgetTruncated = params.maxTokens && (finalResults.length < Math.min(limit, diversifiedResults.length) || finalResults.some((r) => r.contentTruncated)) ? 1 : 0 if (finalResults.length === 0) { metrics.totalMs = Date.now() - startedAt @@ -295,7 +295,7 @@ export const SenseGrepTool = Tool.define("sensegrep", { // Always show relevance score metaParts.push(`Relevance: ${(Math.max(0, Math.min(1, result.semanticScore)) * 100).toFixed(1)}%`) if (result.confidence) { - metaParts.push(`Confidence: ${result.confidence}`) + metaParts.push(`Ranking strength: ${result.confidence}`) } if (result.rerankScore !== undefined) { metaParts.push(`Rerank: ${result.rerankScore.toFixed(3)}`) @@ -374,14 +374,17 @@ export const SenseGrepTool = Tool.define("sensegrep", { return { schemaVersion: 1, command: params.commandName ?? "search", - status: "complete", + status: metrics.tokenBudgetTruncated ? "incomplete" : "complete", index: { fresh: freshness ? !freshness.isStale : null, schemaCompatible: schema.schemaCompatible, snapshotId: `${meta.tableName ?? "chunks"}:${meta.updatedAt}`, }, ...result, + answerSufficiency: "not-assessed", budget: { + maxOutputBytes: params.maxOutputBytes, + maxBytes: params.maxOutputBytes, tokensRequested: params.maxTokens, tokensUsed: contextTokens, inputTokens: retrievalTokens, diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md index fcd54a1..2f21cb1 100644 --- a/packages/mcp/CHANGELOG.md +++ b/packages/mcp/CHANGELOG.md @@ -1,5 +1,16 @@ # @sensegrep/mcp +## 1.17.0 + +### Minor Changes + +- [`bde06a7`](https://github.com/Stahldavid/sensegrep/commit/bde06a75e8327c4a5c0a4c90db9b03f06c00a3b6) Thanks [@Stahldavid](https://github.com/Stahldavid)! - Preserve structural metadata in small files, improve token-budget context selection and cluster coherence, expose graph resolution evidence, accelerate bounded duplicate scans, and support serialized byte budgets for search. Distinguish ranking strength from answer sufficiency. Chunking policy changes trigger an atomic rebuild on the next explicit indexing command. + +### Patch Changes + +- Updated dependencies [[`bde06a7`](https://github.com/Stahldavid/sensegrep/commit/bde06a75e8327c4a5c0a4c90db9b03f06c00a3b6)]: + - @sensegrep/core@1.17.0 + ## 1.16.0 ### Minor Changes diff --git a/packages/mcp/package.json b/packages/mcp/package.json index fe73e05..198fc14 100644 --- a/packages/mcp/package.json +++ b/packages/mcp/package.json @@ -1,7 +1,7 @@ { "name": "@sensegrep/mcp", "mcpName": "io.github.Stahldavid/sensegrep", - "version": "1.16.0", + "version": "1.17.0", "type": "module", "main": "./dist/server.js", "bin": { @@ -33,7 +33,7 @@ "@modelcontextprotocol/node": "^2.0.0", "@modelcontextprotocol/sdk": "^1.29.0", "@modelcontextprotocol/server": "^2.0.0", - "@sensegrep/core": "^1.16.0", + "@sensegrep/core": "^1.17.0", "zod": "^4.4.3" }, "publishConfig": { diff --git a/packages/mcp/src/http-server.ts b/packages/mcp/src/http-server.ts index 2d6cc8a..0975aef 100644 --- a/packages/mcp/src/http-server.ts +++ b/packages/mcp/src/http-server.ts @@ -39,7 +39,7 @@ export function createSensegrepHttpHandler( return createMcpHandler(async (context) => { options.onServerCreated?.(context); const server = new Server( - { name: "sensegrep", version: "1.16.0" }, + { name: "sensegrep", version: "1.17.0" }, { capabilities: { tools: {} } }, ); diff --git a/packages/mcp/src/server.ts b/packages/mcp/src/server.ts index d75c836..028cebb 100644 --- a/packages/mcp/src/server.ts +++ b/packages/mcp/src/server.ts @@ -939,7 +939,7 @@ export function createStdioMcpServer(): Server { const server = new Server( { name: "sensegrep", - version: "1.16.0", + version: "1.17.0", }, { capabilities: { diff --git a/packages/vscode/package.json b/packages/vscode/package.json index e154e0e..62f6c12 100644 --- a/packages/vscode/package.json +++ b/packages/vscode/package.json @@ -602,6 +602,6 @@ "typescript": "^5.9.3" }, "dependencies": { - "@sensegrep/core": "^1.16.0" + "@sensegrep/core": "^1.17.0" } } diff --git a/plugin/sensegrep-cursor/.cursor-plugin/plugin.json b/plugin/sensegrep-cursor/.cursor-plugin/plugin.json index 3c2a087..7883f9c 100644 --- a/plugin/sensegrep-cursor/.cursor-plugin/plugin.json +++ b/plugin/sensegrep-cursor/.cursor-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.16.0", + "version": "1.17.0", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Replaces grep/ripgrep for 95% of code exploration tasks.", "author": { "name": "sensegrep" diff --git a/plugin/sensegrep-cursor/.mcp.json b/plugin/sensegrep-cursor/.mcp.json index d5e6d1b..751bfdd 100644 --- a/plugin/sensegrep-cursor/.mcp.json +++ b/plugin/sensegrep-cursor/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.16.0" + "@sensegrep/mcp@1.17.0" ] } } diff --git a/plugin/sensegrep-plugin/.claude-plugin/marketplace.json b/plugin/sensegrep-plugin/.claude-plugin/marketplace.json index bb5b45f..81ee1c1 100644 --- a/plugin/sensegrep-plugin/.claude-plugin/marketplace.json +++ b/plugin/sensegrep-plugin/.claude-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": ".", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Adds sensegrep MCP tools + smart usage instructions to Claude Code.", - "version": "1.16.0", + "version": "1.17.0", "author": { "name": "sensegrep" }, diff --git a/plugin/sensegrep-plugin/.claude-plugin/plugin.json b/plugin/sensegrep-plugin/.claude-plugin/plugin.json index f525d4e..74a150d 100644 --- a/plugin/sensegrep-plugin/.claude-plugin/plugin.json +++ b/plugin/sensegrep-plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.16.0", + "version": "1.17.0", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Replaces grep/ripgrep for 95% of code exploration tasks.", "author": { "name": "sensegrep", diff --git a/plugin/sensegrep-plugin/.mcp.json b/plugin/sensegrep-plugin/.mcp.json index d5e6d1b..751bfdd 100644 --- a/plugin/sensegrep-plugin/.mcp.json +++ b/plugin/sensegrep-plugin/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.16.0" + "@sensegrep/mcp@1.17.0" ] } } diff --git a/plugins/sensegrep/.codex-plugin/plugin.json b/plugins/sensegrep/.codex-plugin/plugin.json index 348cc0b..803e05e 100644 --- a/plugins/sensegrep/.codex-plugin/plugin.json +++ b/plugins/sensegrep/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.16.0", + "version": "1.17.0", "description": "Semantic + structural code search for AI agents. Read the right code, not more code — combining semantic search, exact matching, and AST-aware retrieval.", "author": { "name": "sensegrep", diff --git a/plugins/sensegrep/.mcp.json b/plugins/sensegrep/.mcp.json index d5e6d1b..751bfdd 100644 --- a/plugins/sensegrep/.mcp.json +++ b/plugins/sensegrep/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.16.0" + "@sensegrep/mcp@1.17.0" ] } } diff --git a/server.json b/server.json index 7fe9f24..258bb7f 100644 --- a/server.json +++ b/server.json @@ -8,12 +8,12 @@ "url": "https://github.com/Stahldavid/sensegrep", "source": "github" }, - "version": "1.16.0", + "version": "1.17.0", "packages": [ { "registryType": "npm", "identifier": "@sensegrep/mcp", - "version": "1.16.0", + "version": "1.17.0", "transport": { "type": "stdio" },