diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index de1d59b..32c4a71 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": "./plugin/sensegrep-plugin", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Adds sensegrep MCP tools + smart usage instructions to Claude Code.", - "version": "1.17.0", + "version": "1.17.1", "author": { "name": "sensegrep" }, diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json index 58ddbb5..437c319 100644 --- a/.cursor-plugin/marketplace.json +++ b/.cursor-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": "plugin/sensegrep-cursor", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns.", - "version": "1.17.0" + "version": "1.17.1" } ] } diff --git a/docs/reliability-release-1.17.1.md b/docs/reliability-release-1.17.1.md new file mode 100644 index 0000000..3215ac5 --- /dev/null +++ b/docs/reliability-release-1.17.1.md @@ -0,0 +1,18 @@ +# Reliability fixes validated on Windows + +## Behavior + +- Incremental CLI indexing copies existing vectors in Arrow batches into a staging generation, applies replacements/removals, verifies its count, and atomically switches the metadata pointer. Failure before activation leaves the previous generation intact. No-change runs do not copy the table. This adds local disk work for changed indexes, but does not re-embed unchanged chunks. Single-file watcher updates are outside this change. +- CLI duplicate JSON now applies `--limit` in minimal, diagnostic and full projections. `summary.total` is the number found; `summary.returned` is the number emitted (`totalDuplicates`/`returnedDuplicates` in full). Output truncation sets `status: incomplete`, `truncated` and `outputTruncated`; it does not invent a continuation cursor. A scan cursor, when present, is preserved. Raise `--limit` to see more groups from the same scan. +- Default search returns at most one result per file. Explicit `--max-per-file` overrides this, and `--exact` retains the two-per-file default. Context selection is unchanged and may select several relevant helpers per file. +- Group labels retain up to seven symbol words instead of four and remove dangling English prepositions/conjunctions. + +## Validation + +- Typecheck passed; full suite passed with 238 tests, followed by the added no-op regression test (14 indexer tests passed). Release CI runs the complete final suite of 239 tests. +- A real child CLI using Ollama was terminated during incremental persistence. Strict verification found stale source files but **no chunk mismatch**, and the active table name was unchanged. The next incremental run activated a new table; strict verification and exact lookup of the added symbol passed. +- Repeated the 16 CuraAI query cases in default and explicitly diverse modes. Default expected-file top-five coverage improved from 13/16 to 15/16; top-ten remained 16/16. The voice concurrency case improved from rank 10 to 6, so broad semantic retrieval is still not exhaustive. +- Exercised all three duplicate JSON projections with `--limit 1`, checking one emitted group and explicit output truncation. +- Rechecked small context budgets, real call references, cluster labels and index health using the built CLI before release. The installed npm CLI is checked again after publication. + +The initial source corpus changed during earlier testing; use these measurements as targeted acceptance evidence, not a universal accuracy claim. `--exact` still prefers exact symbols rather than imposing a strict filter; use `literal` to establish exhaustive textual absence. diff --git a/package-lock.json b/package-lock.json index c255bd8..4ff1d34 100644 --- a/package-lock.json +++ b/package-lock.json @@ -8703,10 +8703,10 @@ }, "packages/cli": { "name": "@sensegrep/cli", - "version": "1.17.0", + "version": "1.17.1", "license": "Apache-2.0", "dependencies": { - "@sensegrep/core": "^1.17.0" + "@sensegrep/core": "^1.17.1" }, "bin": { "sensegrep": "dist/main.js" @@ -8717,7 +8717,7 @@ }, "packages/core": { "name": "@sensegrep/core", - "version": "1.17.0", + "version": "1.17.1", "license": "Apache-2.0", "dependencies": { "@aws-sdk/client-bedrock-runtime": "^3.1084.0", @@ -8738,13 +8738,13 @@ }, "packages/mcp": { "name": "@sensegrep/mcp", - "version": "1.17.0", + "version": "1.17.1", "license": "Apache-2.0", "dependencies": { "@modelcontextprotocol/node": "^2.0.0", "@modelcontextprotocol/sdk": "^1.29.0", "@modelcontextprotocol/server": "^2.0.0", - "@sensegrep/core": "^1.17.0", + "@sensegrep/core": "^1.17.1", "zod": "^4.4.3" }, "bin": { @@ -8760,7 +8760,7 @@ "version": "0.1.27", "license": "Apache-2.0", "dependencies": { - "@sensegrep/core": "^1.17.0" + "@sensegrep/core": "^1.17.1" }, "devDependencies": { "@types/node": "^20.19.43", diff --git a/packages/cli/CHANGELOG.md b/packages/cli/CHANGELOG.md index 6ea9c6e..773871c 100644 --- a/packages/cli/CHANGELOG.md +++ b/packages/cli/CHANGELOG.md @@ -1,5 +1,14 @@ # @sensegrep/cli +## 1.17.1 + +### Patch Changes + +- [`063d867`](https://github.com/Stahldavid/sensegrep/commit/063d867e85b36f7e3153f376fe9fceba0d1e50ad) Thanks [@Stahldavid](https://github.com/Stahldavid)! - Keep the active index consistent when incremental indexing is interrupted by staging vector changes before atomically switching metadata. Preserve the no-change fast path and reuse existing embeddings. Honor duplicate result limits in every CLI JSON projection with explicit output truncation. Improve default search file diversity and retain meaningful words in cluster labels. + +- Updated dependencies [[`063d867`](https://github.com/Stahldavid/sensegrep/commit/063d867e85b36f7e3153f376fe9fceba0d1e50ad)]: + - @sensegrep/core@1.17.1 + ## 1.17.0 ### Minor Changes diff --git a/packages/cli/README.md b/packages/cli/README.md index 4874a87..2a5974c 100644 --- a/packages/cli/README.md +++ b/packages/cli/README.md @@ -51,3 +51,11 @@ structured warning instead of failing silently. - CLI reference: https://github.com/Stahldavid/sensegrep/blob/main/docs/cli-reference.md - Getting started: https://github.com/Stahldavid/sensegrep/blob/main/docs/getting-started.md - Issues: https://github.com/Stahldavid/sensegrep/issues + +### Reliability in 1.17.1 + +Search defaults to one result per file; use `--max-per-file 2` for more snippets from each file. `--exact` preserves its two-per-file default and prefers exact symbols rather than excluding approximate matches. + +Duplicate JSON respects `--limit`. `summary.outputTruncated` means more groups were found than emitted; raise `--limit` to expose them. A continuation cursor advances the candidate scan, not output pagination. Candidate caps still require raising `--max-candidates` for full coverage. + +Incremental `index --no-watch` stages updates before atomically activating the new snapshot, preserving the old index if the process is interrupted. No-change runs avoid copying vectors. This protection does not yet extend to individual watcher updates. diff --git a/packages/cli/package.json b/packages/cli/package.json index d4dbd21..b8c497b 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -1,6 +1,6 @@ { "name": "@sensegrep/cli", - "version": "1.17.0", + "version": "1.17.1", "type": "module", "bin": { "sensegrep": "dist/main.js" @@ -25,7 +25,7 @@ "node": ">=20" }, "dependencies": { - "@sensegrep/core": "^1.17.0" + "@sensegrep/core": "^1.17.1" }, "publishConfig": { "access": "public" diff --git a/packages/cli/src/main.ts b/packages/cli/src/main.ts index 243e0d6..21b4411 100644 --- a/packages/cli/src/main.ts +++ b/packages/cli/src/main.ts @@ -860,7 +860,7 @@ async function run() { if (flags.json) { const detail = resolveJsonProjection(flags["json-detail"] ?? flags.jsonDetail, flags.diagnostic === true) - writeJson(projectDuplicateResponse(result, detail, showCode || fullCode)) + writeJson(projectDuplicateResponse(result, detail, showCode || fullCode, limit)) return } diff --git a/packages/cli/src/search-commands.test.ts b/packages/cli/src/search-commands.test.ts index 434d799..959bf8a 100644 --- a/packages/cli/src/search-commands.test.ts +++ b/packages/cli/src/search-commands.test.ts @@ -120,6 +120,19 @@ describe("search command agent JSON contracts", () => { expect(projectDuplicateResponse(result, "minimal", true).duplicates[0].instances[0]).toHaveProperty("content", "secret code") }) + it("limits every duplicate JSON projection without losing scan continuation", () => { + const raw = { status: "complete", summary: { totalDuplicates: 3, returnedDuplicates: 3, resumeCursor: "next" }, duplicates: [{ instances: [] }, { instances: [] }, { instances: [] }] } + for (const detail of ["minimal", "diagnostic", "full"] as const) { + const result = projectDuplicateResponse(raw, detail, false, 1) + expect(result.duplicates).toHaveLength(1) + expect(result.status).toBe("incomplete") + expect(result.summary.outputTruncated).toBe(true) + expect(detail === "full" ? result.summary.returnedDuplicates : result.summary.returned).toBe(1) + expect(detail === "full" ? result.summary.resumeCursor : result.continuation.cursor).toBe("next") + } + expect(raw.duplicates).toHaveLength(3) + }) + it("projects literal, graph, and show with canonical locations", () => { const literal = projectLiteralResponse({ command: "literal", diff --git a/packages/cli/src/search-commands.ts b/packages/cli/src/search-commands.ts index 08ce7b9..7710de1 100644 --- a/packages/cli/src/search-commands.ts +++ b/packages/cli/src/search-commands.ts @@ -41,7 +41,11 @@ export function projectSearchResponse(res: any, detail: JsonProjection, includeR }) } -export function projectDuplicateResponse(res: any, detail: JsonProjection, includeCode: boolean): any { +export function projectDuplicateResponse(res: any, detail: JsonProjection, includeCode: boolean, limit?: number): any { + if (limit !== undefined && res.duplicates.length > limit) { + res = { ...res, status: "incomplete", duplicates: res.duplicates.slice(0, limit), + summary: { ...res.summary, returnedDuplicates: limit, truncated: true, outputTruncated: true } } + } return projectDuplicateAgentResponse(res, { detail, diagnostics: detail === "diagnostic", includeCode }) } diff --git a/packages/cli/src/usage.ts b/packages/cli/src/usage.ts index 3dd7b84..baef01b 100644 --- a/packages/cli/src/usage.ts +++ b/packages/cli/src/usage.ts @@ -40,7 +40,7 @@ Search options: --min-complexity Minimum cyclomatic complexity --max-complexity Maximum cyclomatic complexity --min-score Minimum relevance score 0-1 - --max-per-file Max results per file (default: 2) + --max-per-file Max results per file (search default: 1; --exact: 2) --max-per-symbol Max results per symbol (default: 2) --has-docs Require documentation --language typescript|javascript|python|java|vue (comma-separated for multiple) diff --git a/packages/core/CHANGELOG.md b/packages/core/CHANGELOG.md index 83bae0a..93ad9bc 100644 --- a/packages/core/CHANGELOG.md +++ b/packages/core/CHANGELOG.md @@ -1,5 +1,11 @@ # @sensegrep/core +## 1.17.1 + +### Patch Changes + +- [`063d867`](https://github.com/Stahldavid/sensegrep/commit/063d867e85b36f7e3153f376fe9fceba0d1e50ad) Thanks [@Stahldavid](https://github.com/Stahldavid)! - Keep the active index consistent when incremental indexing is interrupted by staging vector changes before atomically switching metadata. Preserve the no-change fast path and reuse existing embeddings. Honor duplicate result limits in every CLI JSON projection with explicit output truncation. Improve default search file diversity and retain meaningful words in cluster labels. + ## 1.17.0 ### Minor Changes diff --git a/packages/core/package.json b/packages/core/package.json index 229cd61..f2e37b7 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -1,6 +1,6 @@ { "name": "@sensegrep/core", - "version": "1.17.0", + "version": "1.17.1", "type": "module", "main": "./dist/index.js", "types": "./dist/index.d.ts", diff --git a/packages/core/src/semantic/indexer.test.ts b/packages/core/src/semantic/indexer.test.ts index d2262b2..d519d4d 100644 --- a/packages/core/src/semantic/indexer.test.ts +++ b/packages/core/src/semantic/indexer.test.ts @@ -1,5 +1,5 @@ import path from "node:path" -import { mkdir, rm, writeFile } from "node:fs/promises" +import { mkdir, rm, writeFile, stat } from "node:fs/promises" import { afterEach, beforeEach, describe, expect, it, vi } from "vitest" const TEST_DIR = path.join(process.cwd(), ".test-indexer") @@ -18,6 +18,7 @@ const replaceFileDocuments = vi.fn() const updateDocuments = vi.fn() const writeIndexMeta = vi.fn() const deleteCollection = vi.fn() +const copyCollection = vi.fn() const createStagingCollection = vi.fn() const openCollectionReadOnly = vi.fn() const dropCollectionTable = vi.fn() @@ -57,6 +58,7 @@ vi.mock("./lancedb.js", () => ({ writeIndexMeta, deleteCollection, createStagingCollection, + copyCollection, openCollectionReadOnly, dropCollectionTable, cleanupInactiveTables, @@ -210,11 +212,42 @@ describe("Indexer incremental updates", () => { expect(updateDocuments).not.toHaveBeenCalled() expect(embedDocumentsReusingFile).toHaveBeenCalledTimes(1) expect(embedDocuments).toHaveBeenCalledTimes(1) - expect(replaceFileDocuments).toHaveBeenCalledWith({}, "src/a.ts", expect.any(Array)) + expect(replaceFileDocuments).toHaveBeenCalledWith({ staging: true }, "src/a.ts", expect.any(Array)) expect(writeIndexMeta).toHaveBeenCalledTimes(1) expect(writeIndexMeta.mock.calls[0][1].chunking).toEqual(testChunkingSignature) }) + it("keeps the active snapshot untouched when staged persistence fails", async () => { + replaceFileDocuments.mockRejectedValueOnce(new Error("disk failure")) + const { Indexer } = await import("./indexer.js") + await expect(Indexer.indexProjectIncremental()).rejects.toThrow("disk failure") + expect(copyCollection).toHaveBeenCalledWith({}, { staging: true }, expect.anything()) + expect(replaceFileDocuments).toHaveBeenCalledWith({ staging: true }, "src/a.ts", expect.any(Array)) + expect(writeIndexMeta).not.toHaveBeenCalled() + expect(dropCollectionTable).toHaveBeenCalledWith(TEST_DIR, "chunks_staging") + }) + + it("does not activate a generation if its metadata cannot be committed", async () => { + writeIndexMeta.mockRejectedValueOnce(new Error("metadata failure")) + const { Indexer } = await import("./indexer.js") + await expect(Indexer.indexProjectIncremental()).rejects.toThrow("metadata failure") + expect(dropCollectionTable).toHaveBeenCalledWith(TEST_DIR, "chunks_staging") + expect(clearProjectCache).not.toHaveBeenCalled() + }) + + it("keeps no-op indexing on the active generation without copying vectors", async () => { + const meta = await readIndexMeta() + const current = await stat(path.join(TEST_DIR, "src/a.ts")) + meta.files["src/a.ts"].size = current.size + meta.files["src/a.ts"].mtimeMs = current.mtimeMs + const { Indexer } = await import("./indexer.js") + const result = await Indexer.indexProjectIncremental() + expect(result).toMatchObject({ files: 0, skipped: 1 }) + expect(createStagingCollection).not.toHaveBeenCalled() + expect(copyCollection).not.toHaveBeenCalled() + expect(replaceFileDocuments).not.toHaveBeenCalled() + }) + it("plans embedding work without calling the provider or mutating the index", async () => { const { Indexer } = await import("./indexer.js") diff --git a/packages/core/src/semantic/indexer.ts b/packages/core/src/semantic/indexer.ts index 67b7395..7fed5f1 100644 --- a/packages/core/src/semantic/indexer.ts +++ b/packages/core/src/semantic/indexer.ts @@ -1325,84 +1325,98 @@ export namespace Indexer { const embeddedRows = preparedReplacements.flatMap((prepared) => prepared.rows) const estimatedTokens = docsToAdd.reduce((total, document) => total + estimateEmbeddingTokens(document.content), 0) - // Replace each changed file with rollback support. This keeps the previous - // rows intact if a LanceDB append fails after deletion. - if (embeddedRows.length > 0) { - let chunksPersisted = 0 - for (const file of filesToReplace) { - run.assertNotTimedOut("persist") - const rows = embeddedRows.filter((row) => normalizeIndexedFilePath(row.file) === file) - await VectorStore.replaceFileDocuments(collection, file, rows) + // Stage mutations so a killed process cannot damage the active snapshot. + const hasChanges = filesToReplace.length > 0 || filesToRemove.length > 0 || (!truncatedByMaxFiles && remaining.size > 0) + const staging = hasChanges ? await VectorStore.createStagingCollection(Instance.directory, dimension) : null + const target = staging?.collection ?? collection + let activated = false + try { + if (staging) await VectorStore.copyCollection(collection, target, run.signal) + if (embeddedRows.length > 0) { + let chunksPersisted = 0 + for (const file of filesToReplace) { + run.assertNotTimedOut("persist") + const rows = embeddedRows.filter((row) => normalizeIndexedFilePath(row.file) === file) + await VectorStore.replaceFileDocuments(target, file, rows) + run.assertNotTimedOut("persist") + chunksPersisted += rows.length + run.emit({ + phase: "persist", + current: chunksPersisted, + total: embeddedRows.length, + message: `Persisted ${chunksPersisted}/${embeddedRows.length} changed chunks`, + chunksPrepared: docsToAdd.length, + chunksEmbedded: newlyEmbeddedChunks, + reusedChunks, + estimatedTokens, + requests: estimatedRequests, + batches: embeddingBatches, + chunksPersisted, + skipped, + failed, + }) + } + } + + for (const file of filesToRemove) { run.assertNotTimedOut("persist") - chunksPersisted += rows.length - run.emit({ - phase: "persist", - current: chunksPersisted, - total: embeddedRows.length, - message: `Persisted ${chunksPersisted}/${embeddedRows.length} changed chunks`, - chunksPrepared: docsToAdd.length, - chunksEmbedded: newlyEmbeddedChunks, - reusedChunks, - estimatedTokens, - requests: estimatedRequests, - batches: embeddingBatches, - chunksPersisted, - skipped, - failed, - }) + await VectorStore.deleteByFile(target, file) + removed++ } - } - for (const file of filesToRemove) { - run.assertNotTimedOut("persist") - await VectorStore.deleteByFile(collection, file) - removed++ - } + // Remove files that no longer exist + if (remaining.size > 0 && !truncatedByMaxFiles) { + for (const file of remaining) { + await VectorStore.deleteByFile(target, file) + removed++ + } + } else if (truncatedByMaxFiles) { + for (const file of remaining) { + const prev = previous[file] + if (prev) newStats[file] = prev + } + } - // Remove files that no longer exist - if (remaining.size > 0 && !truncatedByMaxFiles) { - for (const file of remaining) { - await VectorStore.deleteByFile(collection, file) - removed++ + let nextExpectedChunks = 0 + for (const fileStat of Object.values(newStats)) { + nextExpectedChunks += fileStat.chunks?.length ?? 0 } - } else if (truncatedByMaxFiles) { - for (const file of remaining) { - const prev = previous[file] - if (prev) newStats[file] = prev + const nextStats = await VectorStore.getStats(target) + if (nextStats.count !== nextExpectedChunks) { + log.warn("chunk mismatch after incremental update, falling back to full rebuild", { + expectedChunks: nextExpectedChunks, + actualChunks: nextStats.count, + }) + const full = await indexProjectUnlocked(options) + return { ...full, skipped, removed, mode: "full" } } - } + await VectorStore.optimizeForSearch(target, nextStats.count) - let nextExpectedChunks = 0 - for (const fileStat of Object.values(newStats)) { - nextExpectedChunks += fileStat.chunks?.length ?? 0 - } - const nextStats = await VectorStore.getStats(collection) - if (nextExpectedChunks > 0 && nextStats.count !== nextExpectedChunks) { - log.warn("chunk mismatch after incremental update, falling back to full rebuild", { - expectedChunks: nextExpectedChunks, - actualChunks: nextStats.count, + await VectorStore.writeIndexMeta(Instance.directory, { + version: 1, + root: Instance.directory, + profile: Instance.profile, + tableName: staging?.tableName ?? meta.tableName, + embeddings: { + provider, + model, + dimension, + distanceMetric: VectorStore.DEFAULT_DISTANCE_METRIC, + configFingerprint: embeddingConfigFingerprint(config), + }, + chunking, + files: newStats, + updatedAt: Date.now(), }) - const full = await indexProjectUnlocked(options) - return { ...full, skipped, removed, mode: "full" } + + activated = true + if (staging) { + VectorStore.clearProjectCache(Instance.directory) + await VectorStore.cleanupInactiveTables(Instance.directory, staging.tableName) + } + } finally { + if (staging && !activated) await VectorStore.dropCollectionTable(Instance.directory, staging.tableName).catch(() => {}) } - await VectorStore.optimizeForSearch(collection, nextStats.count) - - await VectorStore.writeIndexMeta(Instance.directory, { - version: 1, - root: Instance.directory, - profile: Instance.profile, - tableName: meta.tableName, - embeddings: { - provider, - model, - dimension, - distanceMetric: VectorStore.DEFAULT_DISTANCE_METRIC, - configFingerprint: embeddingConfigFingerprint(config), - }, - chunking, - files: newStats, - updatedAt: Date.now(), - }) const duration = Date.now() - start run.emit({ diff --git a/packages/core/src/semantic/lancedb.ts b/packages/core/src/semantic/lancedb.ts index f2e5985..835bd55 100644 --- a/packages/core/src/semantic/lancedb.ts +++ b/packages/core/src/semantic/lancedb.ts @@ -835,6 +835,15 @@ export namespace VectorStore { return { collection, tableName } } + /** Copy raw Arrow batches without embedding or changing the active generation. */ + export async function copyCollection(source: LanceTable, target: LanceTable, signal?: AbortSignal): Promise { + for await (const batch of source.query()) { + signal?.throwIfAborted() + const rows = batch.toArray().map((row: any) => ({ ...row.toJSON(), vector: Array.from(row.vector ?? [], Number) })) + await addEmbeddedDocuments(target, rows) + } + } + export async function openCollectionTable( projectPath: string, tableName: string, diff --git a/packages/core/src/tool/agent-output.ts b/packages/core/src/tool/agent-output.ts index 5e14ac1..0ab15fc 100644 --- a/packages/core/src/tool/agent-output.ts +++ b/packages/core/src/tool/agent-output.ts @@ -331,6 +331,7 @@ export function projectDuplicateAgentResponse(raw: any, options: AgentProjection processed: raw.summary?.processedCandidates ?? raw.summary?.analyzedCandidates, truncated: raw.summary?.truncated, timedOut: raw.summary?.timedOut, + outputTruncated: raw.summary?.outputTruncated, }) return { schemaVersion: AGENT_SCHEMA_VERSION, diff --git a/packages/core/src/tool/search-schema.ts b/packages/core/src/tool/search-schema.ts index 3c9a856..6a567d0 100644 --- a/packages/core/src/tool/search-schema.ts +++ b/packages/core/src/tool/search-schema.ts @@ -51,7 +51,7 @@ export const SenseGrepParametersSchema = z.object({ ...CommonSearchShape, maxOutputBytes: z.number().int().min(256).max(100_000_000).optional().describe("Maximum serialized JSON response bytes (minimum 256)"), limit: z.number().int().positive().max(500).optional().describe("Maximum results (default: 10)"), - maxPerFile: z.number().int().nonnegative().optional().describe("Maximum results per file (default: 2)"), + maxPerFile: z.number().int().nonnegative().optional().describe("Maximum results per file (search default: 1; exact lookup: 2)"), maxPerSymbol: z.number().int().nonnegative().optional().describe("Maximum results per symbol (default: 2)"), }) diff --git a/packages/core/src/tool/sensegrep-cluster.test.ts b/packages/core/src/tool/sensegrep-cluster.test.ts index 4537d51..69eabce 100644 --- a/packages/core/src/tool/sensegrep-cluster.test.ts +++ b/packages/core/src/tool/sensegrep-cluster.test.ts @@ -243,3 +243,14 @@ describe("SenseGrepClusterTool", () => { expect(listDocuments).toHaveBeenCalledTimes(1) }) }) + +it("keeps domain words in long labels and removes dangling prepositions", async () => { + const { getDominantSymbolPhrases } = await import("./sensegrep-pipeline.js") + const rows = [ + { metadata: { symbolName: "updateOrderStatusByAsaasPayment" } }, + { metadata: { symbolName: "recordPendingPaymentEventFor" } }, + ] as any + const labels = getDominantSymbolPhrases(rows, "", 3, false) + expect(labels).toContain("update order status by asaas payment") + expect(labels.every(label => !/\b(?:by|for)$/.test(label))).toBe(true) +}) diff --git a/packages/core/src/tool/sensegrep-pipeline.ts b/packages/core/src/tool/sensegrep-pipeline.ts index 3d17f07..74197fc 100644 --- a/packages/core/src/tool/sensegrep-pipeline.ts +++ b/packages/core/src/tool/sensegrep-pipeline.ts @@ -1583,7 +1583,8 @@ export function getDominantSymbolPhrases( for (const result of results) { const tokens = getSymbolTokens(result.metadata) .filter((token) => !excludeQueryTokens || !queryTokens.has(token)) - .slice(0, 4) + .slice(0, 7) + while (tokens.length && /^(?:by|for|with|from|to|of|in|on|at|and|or|the|a|an)$/.test(tokens[tokens.length - 1])) tokens.pop() if (tokens.length < 2) continue const phrase = tokens.join(" ") counts.set(phrase, (counts.get(phrase) ?? 0) + 1) diff --git a/packages/core/src/tool/sensegrep.ts b/packages/core/src/tool/sensegrep.ts index 46eae0a..c1fd2cc 100644 --- a/packages/core/src/tool/sensegrep.ts +++ b/packages/core/src/tool/sensegrep.ts @@ -152,7 +152,7 @@ export const SenseGrepTool = Tool.define("sensegrep", { const dedupedResults = dedupeOverlapping(rankedResults) // Enforce diversity across file/symbol to avoid repeating the same source - const maxPerFile = typeof params.maxPerFile === "number" ? Math.max(0, params.maxPerFile) : 2 + const maxPerFile = typeof params.maxPerFile === "number" ? Math.max(0, params.maxPerFile) : (params.exact ? 2 : 1) const maxPerSymbol = typeof params.maxPerSymbol === "number" ? Math.max(0, params.maxPerSymbol) : 2 const diversifiedResults = diversifyResults(dedupedResults, { maxPerFile, maxPerSymbol }) diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md index 2f21cb1..1faee07 100644 --- a/packages/mcp/CHANGELOG.md +++ b/packages/mcp/CHANGELOG.md @@ -1,5 +1,14 @@ # @sensegrep/mcp +## 1.17.1 + +### Patch Changes + +- [`063d867`](https://github.com/Stahldavid/sensegrep/commit/063d867e85b36f7e3153f376fe9fceba0d1e50ad) Thanks [@Stahldavid](https://github.com/Stahldavid)! - Keep the active index consistent when incremental indexing is interrupted by staging vector changes before atomically switching metadata. Preserve the no-change fast path and reuse existing embeddings. Honor duplicate result limits in every CLI JSON projection with explicit output truncation. Improve default search file diversity and retain meaningful words in cluster labels. + +- Updated dependencies [[`063d867`](https://github.com/Stahldavid/sensegrep/commit/063d867e85b36f7e3153f376fe9fceba0d1e50ad)]: + - @sensegrep/core@1.17.1 + ## 1.17.0 ### Minor Changes diff --git a/packages/mcp/package.json b/packages/mcp/package.json index 198fc14..e2dab04 100644 --- a/packages/mcp/package.json +++ b/packages/mcp/package.json @@ -1,7 +1,7 @@ { "name": "@sensegrep/mcp", "mcpName": "io.github.Stahldavid/sensegrep", - "version": "1.17.0", + "version": "1.17.1", "type": "module", "main": "./dist/server.js", "bin": { @@ -33,7 +33,7 @@ "@modelcontextprotocol/node": "^2.0.0", "@modelcontextprotocol/sdk": "^1.29.0", "@modelcontextprotocol/server": "^2.0.0", - "@sensegrep/core": "^1.17.0", + "@sensegrep/core": "^1.17.1", "zod": "^4.4.3" }, "publishConfig": { diff --git a/packages/mcp/src/http-server.ts b/packages/mcp/src/http-server.ts index 0975aef..8dc7a50 100644 --- a/packages/mcp/src/http-server.ts +++ b/packages/mcp/src/http-server.ts @@ -39,7 +39,7 @@ export function createSensegrepHttpHandler( return createMcpHandler(async (context) => { options.onServerCreated?.(context); const server = new Server( - { name: "sensegrep", version: "1.17.0" }, + { name: "sensegrep", version: "1.17.1" }, { capabilities: { tools: {} } }, ); diff --git a/packages/mcp/src/server.ts b/packages/mcp/src/server.ts index 028cebb..8de6454 100644 --- a/packages/mcp/src/server.ts +++ b/packages/mcp/src/server.ts @@ -939,7 +939,7 @@ export function createStdioMcpServer(): Server { const server = new Server( { name: "sensegrep", - version: "1.17.0", + version: "1.17.1", }, { capabilities: { diff --git a/packages/vscode/package.json b/packages/vscode/package.json index 62f6c12..19d3680 100644 --- a/packages/vscode/package.json +++ b/packages/vscode/package.json @@ -602,6 +602,6 @@ "typescript": "^5.9.3" }, "dependencies": { - "@sensegrep/core": "^1.17.0" + "@sensegrep/core": "^1.17.1" } } diff --git a/plugin/sensegrep-cursor/.cursor-plugin/plugin.json b/plugin/sensegrep-cursor/.cursor-plugin/plugin.json index 7883f9c..4f4ecac 100644 --- a/plugin/sensegrep-cursor/.cursor-plugin/plugin.json +++ b/plugin/sensegrep-cursor/.cursor-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.17.0", + "version": "1.17.1", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Replaces grep/ripgrep for 95% of code exploration tasks.", "author": { "name": "sensegrep" diff --git a/plugin/sensegrep-cursor/.mcp.json b/plugin/sensegrep-cursor/.mcp.json index 751bfdd..e6925cb 100644 --- a/plugin/sensegrep-cursor/.mcp.json +++ b/plugin/sensegrep-cursor/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.17.0" + "@sensegrep/mcp@1.17.1" ] } } diff --git a/plugin/sensegrep-cursor/skills/sensegrep/SKILL.md b/plugin/sensegrep-cursor/skills/sensegrep/SKILL.md index f118801..7ef701b 100644 --- a/plugin/sensegrep-cursor/skills/sensegrep/SKILL.md +++ b/plugin/sensegrep-cursor/skills/sensegrep/SKILL.md @@ -60,7 +60,7 @@ sensegrep_search({ strictImports: true, // strict AST import metadata validation hasDocumentation: true, // require docs minScore: 0.5, // relevance threshold - maxPerFile: 2, // dedup per file (default: 2) + maxPerFile: 2, // dedup per file (search default: 1; exact: 2) maxPerSymbol: 2, // dedup per symbol (default: 2) shake: false, // disable tree-shaking if collapsed output hides the target limit: 10 // max results (default: 10) diff --git a/plugin/sensegrep-plugin/.claude-plugin/marketplace.json b/plugin/sensegrep-plugin/.claude-plugin/marketplace.json index 81ee1c1..e08f7d8 100644 --- a/plugin/sensegrep-plugin/.claude-plugin/marketplace.json +++ b/plugin/sensegrep-plugin/.claude-plugin/marketplace.json @@ -13,7 +13,7 @@ "name": "sensegrep", "source": ".", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Adds sensegrep MCP tools + smart usage instructions to Claude Code.", - "version": "1.17.0", + "version": "1.17.1", "author": { "name": "sensegrep" }, diff --git a/plugin/sensegrep-plugin/.claude-plugin/plugin.json b/plugin/sensegrep-plugin/.claude-plugin/plugin.json index 74a150d..494464e 100644 --- a/plugin/sensegrep-plugin/.claude-plugin/plugin.json +++ b/plugin/sensegrep-plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.17.0", + "version": "1.17.1", "description": "Semantic code search for AI agents. Search code by meaning, not text patterns. Replaces grep/ripgrep for 95% of code exploration tasks.", "author": { "name": "sensegrep", diff --git a/plugin/sensegrep-plugin/.mcp.json b/plugin/sensegrep-plugin/.mcp.json index 751bfdd..e6925cb 100644 --- a/plugin/sensegrep-plugin/.mcp.json +++ b/plugin/sensegrep-plugin/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.17.0" + "@sensegrep/mcp@1.17.1" ] } } diff --git a/plugin/sensegrep-plugin/skills/sensegrep/SKILL.md b/plugin/sensegrep-plugin/skills/sensegrep/SKILL.md index 83b1438..80f3d72 100644 --- a/plugin/sensegrep-plugin/skills/sensegrep/SKILL.md +++ b/plugin/sensegrep-plugin/skills/sensegrep/SKILL.md @@ -60,7 +60,7 @@ sensegrep_search({ strictImports: true, // strict AST import metadata validation hasDocumentation: true, // require docs minScore: 0.5, // relevance threshold - maxPerFile: 2, // dedup per file (default: 2) + maxPerFile: 2, // dedup per file (search default: 1; exact: 2) maxPerSymbol: 2, // dedup per symbol (default: 2) shake: false, // disable tree-shaking if collapsed output hides the target limit: 10 // max results (default: 10) diff --git a/plugins/sensegrep/.codex-plugin/plugin.json b/plugins/sensegrep/.codex-plugin/plugin.json index 803e05e..fcd6169 100644 --- a/plugins/sensegrep/.codex-plugin/plugin.json +++ b/plugins/sensegrep/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sensegrep", - "version": "1.17.0", + "version": "1.17.1", "description": "Semantic + structural code search for AI agents. Read the right code, not more code — combining semantic search, exact matching, and AST-aware retrieval.", "author": { "name": "sensegrep", diff --git a/plugins/sensegrep/.mcp.json b/plugins/sensegrep/.mcp.json index 751bfdd..e6925cb 100644 --- a/plugins/sensegrep/.mcp.json +++ b/plugins/sensegrep/.mcp.json @@ -4,7 +4,7 @@ "command": "npx", "args": [ "-y", - "@sensegrep/mcp@1.17.0" + "@sensegrep/mcp@1.17.1" ] } } diff --git a/plugins/sensegrep/skills/sensegrep/SKILL.md b/plugins/sensegrep/skills/sensegrep/SKILL.md index f118801..7ef701b 100644 --- a/plugins/sensegrep/skills/sensegrep/SKILL.md +++ b/plugins/sensegrep/skills/sensegrep/SKILL.md @@ -60,7 +60,7 @@ sensegrep_search({ strictImports: true, // strict AST import metadata validation hasDocumentation: true, // require docs minScore: 0.5, // relevance threshold - maxPerFile: 2, // dedup per file (default: 2) + maxPerFile: 2, // dedup per file (search default: 1; exact: 2) maxPerSymbol: 2, // dedup per symbol (default: 2) shake: false, // disable tree-shaking if collapsed output hides the target limit: 10 // max results (default: 10) diff --git a/server.json b/server.json index 258bb7f..46e6ea6 100644 --- a/server.json +++ b/server.json @@ -8,12 +8,12 @@ "url": "https://github.com/Stahldavid/sensegrep", "source": "github" }, - "version": "1.17.0", + "version": "1.17.1", "packages": [ { "registryType": "npm", "identifier": "@sensegrep/mcp", - "version": "1.17.0", + "version": "1.17.1", "transport": { "type": "stdio" },