Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
75 changes: 75 additions & 0 deletions src/__tests__/orchestrate.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ import {
mapFindingsToReview,
type ReviewComment,
} from "../review/comment-mapping.js"
import { filterNonFindings } from "../review/filter-non-findings.js"
import { selectFindings } from "../review/select-findings.js"
import { renderCostSummary } from "../openrouter/cost-summary.js"
import {
Expand All @@ -29,6 +30,7 @@ import {
type GenerateFindings,
type ReviewContext,
} from "../orchestrate.js"
import { makeFinding } from "../review/__tests__/make-finding.js"
import { createTestLogger } from "./test-logger.js"

const sampleDiff = readFileSync(
Expand Down Expand Up @@ -652,6 +654,79 @@ describe("orchestrate", () => {

expect(stubs.fetchPullRequestCalls).toHaveLength(0)
})

it("filters non-findings from LLM output", async () => {
const nonFinding = makeFinding({
line: 3,
failure_scenario: "N/A — this is correct behavior.",
})
const realFinding = makeFinding({
line: 145,
failure_scenario:
'register(" ", "value") succeeds and the entry is orphaned.',
})
const mixedResponse: ReviewResponse = {
analysis: "checked",
findings: [nonFinding, realFinding],
}

const { findings: mixedFiltered } = filterNonFindings(
mixedResponse.findings,
)
const mixedSelection = selectFindings({
findings: mixedFiltered,
severityThreshold: "low",
maxFindings: undefined,
})
const mixedMapped = mapFindingsToReview({
findings: mixedSelection.selected,
commentableByPath: fixtureCommentableByPath,
})
const mixedBody = buildReviewBody({
bodyFindings: mixedMapped.bodyFindings,
droppedByCap: mixedSelection.droppedByCap,
model: "test/model",
inlineCommentCount: mixedMapped.comments.length,
})
const mixedFallbackBody = buildReviewBody({
bodyFindings: mixedSelection.selected,
droppedByCap: mixedSelection.droppedByCap,
model: "test/model",
bodyFindingsHeading: "Findings",
bodyFindingsDescription:
"Inline comments were unavailable; all findings are listed here:",
})

const stubs = makeOrchestrateDeps({
fixtureResult: { review: mixedResponse },
})
const logger = createTestLogger()

const result = await orchestrate(stubs.deps, logger)
Comment thread
aliasunder marked this conversation as resolved.
Comment thread
aliasunder marked this conversation as resolved.

expect(result).toEqual({
findingsCount: 1,
reviewUrl: "https://github.com/test/review/1",
modelUsed: "test/model",
skippedReason: "",
costSummaryMarkdown: expectedCostSummary,
})

expect(stubs.submitReviewCalls).toHaveLength(1)
expect(first(stubs.submitReviewCalls)).toEqual({
prNumber: fixturePrContext.prNumber,
commitId: fixturePrContext.headSha,
body: mixedBody,
comments: mixedMapped.comments,
fallbackBody: mixedFallbackBody,
})

expect(logger.messages).toContainEqual({
level: "info",
message: "filtered non-findings",
data: { droppedAsNonFinding: 1 },
})
})
})
})

Expand Down
12 changes: 10 additions & 2 deletions src/orchestrate.ts
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ import {
generateDelimiterNonce,
type PromptFile,
} from "./review/prompt.js"
import { filterNonFindings } from "./review/filter-non-findings.js"
import { selectFindings } from "./review/select-findings.js"

export type ReviewContext = {
Expand Down Expand Up @@ -245,9 +246,16 @@ export const orchestrate = async (

const { modelUsed, attempts } = structuredResult

// Step 12: select findings
// Step 12: filter non-findings before selection so cap slots aren't wasted
const { findings: realFindings, droppedAsNonFinding } = filterNonFindings(
structuredResult.review.findings,
)
if (droppedAsNonFinding > 0) {
logger.info("filtered non-findings", { droppedAsNonFinding })
}

const { selected, droppedByCap } = selectFindings({
findings: structuredResult.review.findings,
findings: realFindings,
severityThreshold,
maxFindings: config.maxFindings,
})
Expand Down
37 changes: 34 additions & 3 deletions src/review/__tests__/comment-mapping.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -215,11 +215,23 @@ The guard rejects only the exact empty string.
})

expect(mapped.comments[0]?.body).toContain(
"```diff\n-old line\n+new line\n```",
"<details>\n<summary>Suggested fix</summary>\n\n```diff\n-old line\n+new line\n```\n\n</details>",
Comment thread
aliasunder marked this conversation as resolved.
Comment thread
aliasunder marked this conversation as resolved.
)
expect(mapped.comments[1]?.body).not.toContain("```diff")
})

it("omits the suggestion block for an empty-string suggestion", () => {
const finding = makeFinding({ line: 145, suggestion: "" })

const mapped = mapFindingsToReview({
findings: [finding],
commentableByPath: makeCommentableByPath(),
})

expect(mapped.comments[0]?.body).not.toContain("<details>")
expect(mapped.comments[0]?.body).not.toContain("```diff")
})

it("sizes the suggestion fence beyond any backtick run inside the suggestion", () => {
const fencedSuggestion = "```md\n-old fence\n+new fence\n```"
const finding = makeFinding({ line: 145, suggestion: fencedSuggestion })
Expand All @@ -230,7 +242,7 @@ The guard rejects only the exact empty string.
})

expect(mapped.comments[0]?.body).toContain(
`\`\`\`\`diff\n${fencedSuggestion}\n\`\`\`\``,
`<details>\n<summary>Suggested fix</summary>\n\n\`\`\`\`diff\n${fencedSuggestion}\n\`\`\`\`\n\n</details>`,
)
})
})
Expand Down Expand Up @@ -282,7 +294,26 @@ _1 lower-severity finding(s) omitted by the max_findings cap: \`src/greeter.ts:5
model: "anthropic/claude-sonnet-4-6",
})

expect(body).toContain("```diff\n-old line\n+new line\n```")
expect(body).toBe(`### Findings beyond the diff

These are in code the changes touch or depend on, outside the diff's line ranges:

- **[medium/correctness]** Whitespace-only keys pass the empty-key guard — \`src/untouched.ts:30\` _(confidence: high)_
The guard rejects only the exact empty string.
**Failure scenario:** register(" ", "value") succeeds and the entry is orphaned.

<details>
<summary>Suggested fix</summary>

\`\`\`diff
-old line
+new line
\`\`\`

</details>

---
*umm-actually · anthropic/claude-sonnet-4-6*`)
})

it("renders only attribution when there is nothing beyond the diff and no cap drops", () => {
Expand Down
183 changes: 183 additions & 0 deletions src/review/__tests__/filter-non-findings.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,183 @@
import { describe, expect, it } from "vitest"
import { filterNonFindings } from "../filter-non-findings.js"
import { makeFinding } from "./make-finding.js"

describe("filterNonFindings", () => {
// --- Prefix patterns that should be dropped ---

it("drops a finding whose failure_scenario starts with 'N/A'", () => {
const finding = makeFinding({
failure_scenario: "N/A — tests are valid.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 1 })
})

it("drops a finding whose failure_scenario starts with 'n/a' (case-insensitive)", () => {
const finding = makeFinding({
failure_scenario: "n/a — Zod validation catches malformed responses.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 1 })
})

it("drops a finding whose failure_scenario starts with 'NA' (no slash)", () => {
const finding = makeFinding({
failure_scenario: "NA — the code handles this edge case.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 1 })
})

it("drops a finding whose failure_scenario starts with 'Not applicable'", () => {
const finding = makeFinding({
failure_scenario: "Not applicable — the code handles this edge case.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 1 })
})

it("drops a finding whose failure_scenario starts with 'Placeholder'", () => {
const finding = makeFinding({
failure_scenario: "Placeholder — re-evaluating.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 1 })
})

// --- Passthrough: real findings must not be dropped ---

it("keeps a finding with a concrete failure scenario", () => {
const finding = makeFinding({
failure_scenario:
"A CI step that gates deployment reads the env var, but the value is never set in the matrix.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [finding], droppedAsNonFinding: 0 })
})

it("keeps a finding that mentions 'by design' as a qualifier", () => {
const finding = makeFinding({
failure_scenario:
"The function returns null by design, but the caller never null-checks.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [finding], droppedAsNonFinding: 0 })
})

it("keeps a finding that mentions 'this is correct' as a qualifier", () => {
const finding = makeFinding({
failure_scenario:
"This is correct for ASCII but breaks on multi-byte UTF-8.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [finding], droppedAsNonFinding: 0 })
})

it("keeps a finding starting with 'None of'", () => {
const finding = makeFinding({
failure_scenario: "None of the guards catch this input.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [finding], droppedAsNonFinding: 0 })
})

it("keeps all findings when none are non-findings", () => {
const findings = [
makeFinding({
line: 1,
failure_scenario:
'register(" ", "value") succeeds and the entry is orphaned.',
}),
makeFinding({
line: 10,
failure_scenario:
"A user configures the threshold to 'extreme' and the action crashes.",
}),
]

const result = filterNonFindings(findings)

expect(result).toEqual({ findings, droppedAsNonFinding: 0 })
})

// --- Mixed and edge cases ---

it("returns empty findings when all findings are non-findings", () => {
const findings = [
makeFinding({ line: 1, failure_scenario: "N/A — tests are valid." }),
makeFinding({
line: 2,
failure_scenario: "Not applicable — already handled.",
}),
]

const result = filterNonFindings(findings)

expect(result).toEqual({ findings: [], droppedAsNonFinding: 2 })
})

it("filters selectively in a mixed set of findings", () => {
const realFinding = makeFinding({
line: 1,
failure_scenario:
'register(" ", "value") succeeds and the entry is orphaned.',
})
const nonFinding = makeFinding({
line: 2,
failure_scenario: "N/A — this is correct behavior.",
})

const result = filterNonFindings([realFinding, nonFinding])

expect(result).toEqual({
findings: [realFinding],
droppedAsNonFinding: 1,
})
})

it("does not drop a finding that contains 'None' mid-sentence", () => {
const finding = makeFinding({
failure_scenario:
"Returns None instead of an empty list when the input is empty.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [finding], droppedAsNonFinding: 0 })
})

it("drops a finding with leading whitespace before the prefix", () => {
const finding = makeFinding({
failure_scenario: " N/A — tests are valid.",
})

const result = filterNonFindings([finding])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 1 })
})

it("handles an empty findings array", () => {
const result = filterNonFindings([])

expect(result).toEqual({ findings: [], droppedAsNonFinding: 0 })
})
})
Loading
Loading