Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
97 changes: 3 additions & 94 deletions src/agents/chat_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,10 +24,10 @@
from dataclasses import dataclass

from agents.tools.base import Invocation, ToolRegistry, ToolResult
from agents.tools.evidence import format_evidence, limit_evidence
from agents.tools.grep import GrepTool
from agents.tools.retrieve import RetrieveTool
from llm.base import ChatResult, LLMClient, Message, ReasoningDelta, TextDelta, ToolCall
from rag.prompt import chunk_header
from rag.source_filter import SourceExclusions
from rag.types import RetrievalFilters, ScoredChunk
from store.base import VectorStore
Expand All @@ -37,27 +37,6 @@
# ends the loop (see `run`).
_MAX_STEPS = 3

# Per-chunk cap on what goes back to the model, so a handful of large chunks
# can't crowd out the conversation.
_SOURCE_CHARS = 800

# Caps on a *whole* tool result. The per-chunk limit alone bounds nothing:
# `grep` returns every match in the scoped corpus, so a broad pattern can
# return hundreds of chunks that are individually small and collectively
# larger than the context window.
#
# Both are needed, and each has to bite where the other doesn't: the char
# budget bounds full-size chunks (it stops at 10 of them), the chunk count
# bounds a long tail of small ones. Keep `_MAX_EVIDENCE_CHUNKS * _SOURCE_CHARS`
# above `_MAX_EVIDENCE_CHARS` or the char budget becomes unreachable.
#
# Applied per call rather than per turn, so what one search returns never
# depends on which others shared its turn; a step is therefore bounded by
# `_MAX_PARALLEL_TOOLS` times this — ~32k chars, which the smallest configured
# Ollama context can still be too tight for (see `OLLAMA_NUM_CTX`).
_MAX_EVIDENCE_CHUNKS = 12
_MAX_EVIDENCE_CHARS = 8_000

# Searches asked for in the same turn run together. The cap is low because each
# `retrieve` already forks a pair of threads internally (`rag.hybrid`), so the
# real thread count is about double this.
Expand Down Expand Up @@ -132,76 +111,6 @@ class Evidence:
ChatEvent = Invocation | Evidence | Reasoning | Token


def _limit_evidence(chunks: list[ScoredChunk]) -> list[ScoredChunk]:
"""The prefix of ``chunks`` that fits the budget, in the order given.

Deterministic and order-preserving: the tools already return their best
matches first, so taking a prefix keeps the most relevant sources. The
first chunk is always kept, so an oversized one is truncated by
``_SOURCE_CHARS`` rather than dropped entirely.
"""
selected: list[ScoredChunk] = []
used = 0
for chunk in chunks:
if len(selected) >= _MAX_EVIDENCE_CHUNKS:
break
size = min(len(chunk.text), _SOURCE_CHARS)
if selected and used + size > _MAX_EVIDENCE_CHARS:
break
selected.append(chunk)
used += size
return selected


def _format_evidence(result: ToolResult, chunks: list[ScoredChunk]) -> str:
"""What the model sees for one tool call — the sources themselves.

``chunks`` is what survived the budget, which is what the caller cites; a
dropped chunk is counted but never quoted, so the model is not asked to
answer from text it cannot see.
"""
if not chunks:
return result.summary or "No matches."
ordered_chunks = _order_evidence_for_display(chunks)
body = "\n\n---\n\n".join(
f"{chunk_header(chunk)}\n{chunk.text[:_SOURCE_CHARS]}"
for chunk in ordered_chunks
)
omitted = result.match_count - len(result.matched_chunks(chunks))
if omitted:
body += f"\n\n---\n\n({omitted} further match(es) omitted.)"
return body


def _order_evidence_for_display(chunks: list[ScoredChunk]) -> list[ScoredChunk]:
"""Keep artifact relevance order while restoring source order within each file.

Budget selection receives direct hits before neighbours so context cannot
displace a match. The model should nevertheless read each selected file in
its natural order; grouping after selection gives us both properties.
"""
by_artifact: dict[str, list[ScoredChunk]] = {}
for chunk in chunks:
by_artifact.setdefault(chunk.artifact_id, []).append(chunk)

ordered: list[ScoredChunk] = []
for artifact_chunks in by_artifact.values():
if all(chunk.position is None for chunk in artifact_chunks):
ordered.extend(artifact_chunks)
continue
ordered.extend(
sorted(
artifact_chunks,
key=lambda chunk: (
chunk.start_page if chunk.start_page is not None else 0,
chunk.position is None,
chunk.position if chunk.position is not None else 0,
),
)
)
return ordered


class ChatAgent:
def __init__(
self,
Expand Down Expand Up @@ -307,7 +216,7 @@ def run(self, question: str, history: list[Message]) -> Iterator[ChatEvent]:
for call, tool_result in zip(
result.tool_calls, self._run_tools(result.tool_calls), strict=True
):
chunks = _limit_evidence(tool_result.chunks)
chunks = limit_evidence(tool_result.chunks)
matched_chunks = tool_result.matched_chunks(chunks)
if matched_chunks:
yield Evidence(matched_chunks)
Expand All @@ -317,7 +226,7 @@ def run(self, question: str, history: list[Message]) -> Iterator[ChatEvent]:
messages.append(
Message(
role="tool",
content=_format_evidence(tool_result, chunks),
content=format_evidence(tool_result, chunks),
tool_call_id=call.id,
name=call.name,
)
Expand Down
2 changes: 1 addition & 1 deletion src/agents/tools/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ class ToolResult:
``summary`` is the one-line fallback shown to the model when there is
nothing to show — an error, or a search that matched nothing. When
``chunks`` is non-empty the model is given the chunks themselves instead;
see ``agents.chat_agent._format_evidence``.
see ``agents.tools.evidence.format_evidence``.
"""

summary: str
Expand Down
111 changes: 111 additions & 0 deletions src/agents/tools/evidence.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
"""How a search tool's result is budgeted, ordered and shown to the model.

Shared by every agent loop that runs ``retrieve``/``grep`` (the chat agent and
the onboarding buddy), so the two cannot drift on how much evidence one call
may return or how it reads. It lives beside the tools rather than in ``rag``
because it formats a ``ToolResult``, and ``rag`` must not import ``agents``.
"""

from collections.abc import Callable

from agents.tools.base import ToolResult
from rag.prompt import chunk_header
from rag.types import ScoredChunk

# Per-chunk cap on what goes back to the model, so a handful of large chunks
# can't crowd out the conversation.
SOURCE_CHARS = 800

# Caps on a *whole* tool result. The per-chunk limit alone bounds nothing:
# `grep` returns every match in the scoped corpus, so a broad pattern can
# return hundreds of chunks that are individually small and collectively
# larger than the context window.
#
# Both are needed, and each has to bite where the other doesn't: the char
# budget bounds full-size chunks (it stops at 10 of them), the chunk count
# bounds a long tail of small ones. Keep `MAX_EVIDENCE_CHUNKS * SOURCE_CHARS`
# above `MAX_EVIDENCE_CHARS` or the char budget becomes unreachable.
#
# Applied per call rather than per turn, so what one search returns never
# depends on which others shared its turn.
MAX_EVIDENCE_CHUNKS = 12
MAX_EVIDENCE_CHARS = 8_000


def limit_evidence(chunks: list[ScoredChunk]) -> list[ScoredChunk]:
"""The prefix of ``chunks`` that fits the budget, in the order given.

Deterministic and order-preserving: the tools already return their best
matches first, so taking a prefix keeps the most relevant sources. The
first chunk is always kept, so an oversized one is truncated by
``SOURCE_CHARS`` rather than dropped entirely.
"""
selected: list[ScoredChunk] = []
used = 0
for chunk in chunks:
if len(selected) >= MAX_EVIDENCE_CHUNKS:
break
size = min(len(chunk.text), SOURCE_CHARS)
if selected and used + size > MAX_EVIDENCE_CHARS:
break
selected.append(chunk)
used += size
return selected


def format_evidence(
result: ToolResult,
chunks: list[ScoredChunk],
*,
header: Callable[[ScoredChunk], str] = chunk_header,
empty: str | None = None,
) -> str:
"""What the model sees for one tool call — the sources themselves.

``chunks`` is what survived the budget, which is what the caller cites; a
dropped chunk is counted but never quoted, so the model is not asked to
answer from text it cannot see.

``header`` labels each chunk (the buddy adds a test-file warning to the
default). ``empty`` replaces the tool's own one-line summary when nothing
survived, for a caller whose persona reacts to specific wording.
"""
if not chunks:
return empty or result.summary or "No matches."
body = "\n\n---\n\n".join(
f"{header(chunk)}\n{chunk.text[:SOURCE_CHARS]}"
for chunk in order_evidence_for_display(chunks)
)
omitted = result.match_count - len(result.matched_chunks(chunks))
if omitted:
body += f"\n\n---\n\n({omitted} further match(es) omitted.)"
return body


def order_evidence_for_display(chunks: list[ScoredChunk]) -> list[ScoredChunk]:
"""Keep artifact relevance order while restoring source order within each file.

Budget selection receives direct hits before neighbours so context cannot
displace a match. The model should nevertheless read each selected file in
its natural order; grouping after selection gives us both properties.
"""
by_artifact: dict[str, list[ScoredChunk]] = {}
for chunk in chunks:
by_artifact.setdefault(chunk.artifact_id, []).append(chunk)

ordered: list[ScoredChunk] = []
for artifact_chunks in by_artifact.values():
if all(chunk.position is None for chunk in artifact_chunks):
ordered.extend(artifact_chunks)
continue
ordered.extend(
sorted(
artifact_chunks,
key=lambda chunk: (
chunk.start_page if chunk.start_page is not None else 0,
chunk.position is None,
chunk.position if chunk.position is not None else 0,
),
)
)
return ordered
20 changes: 19 additions & 1 deletion src/api/routes/buddy.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@
from onboarding.buddy_compact import compact_memory
from onboarding.buddy_open import stream_session
from onboarding.vocabulary import Vocabulary
from rag.types import RetrievalFilters
from store.base import VectorStore

logger = logging.getLogger(__name__)
Expand Down Expand Up @@ -66,6 +67,21 @@ def _to_toolspec(schema: BuddyToolSpecSchema) -> ToolSpec:
)


def _narrowing(body: BuddyAgentRequest) -> RetrievalFilters | None:
"""The reader's source-system and time narrowing, if they chose any.

The project scope is not in here: it is ``project_ids`` and always applies.
An empty source-system list means "all", as it did for chat.
"""
if body.filters is None:
return None
return RetrievalFilters(
source_systems=body.filters.source_systems or None,
time_from=body.filters.time_from,
time_to=body.filters.time_to,
)


@router.post(
"/onboarding/buddy/agent",
response_model=BuddyAgentResponse,
Expand All @@ -81,7 +97,7 @@ def buddy_agent(
) -> BuddyAgentResponse:
"""One turn of the tool-using buddy.

Executes ``search_docs`` locally (retrieval + citations) and returns as soon as it
Executes the searches locally (retrieval + citations) and returns as soon as it
either has a final answer or needs a backend-only tool run. The backend carries the
``messages`` list back verbatim, each pending tool's result appended as a ``tool``.

Expand All @@ -107,6 +123,7 @@ def buddy_agent(
project_ids=frozenset(body.project_ids),
capabilities_enabled=body.capabilities_enabled,
team_mode=body.team_mode,
filters=_narrowing(body),
)
except LLMUnavailableError as exc:
raise HTTPException(
Expand All @@ -131,6 +148,7 @@ def buddy_agent(
)
for cit in result.citations
],
reasoning=result.reasoning,
)


Expand Down
22 changes: 22 additions & 0 deletions src/api/schemas.py
Original file line number Diff line number Diff line change
Expand Up @@ -1451,6 +1451,17 @@ class BuddyAgentRequest(BaseModel):
"`capabilities_enabled`."
),
)
filters: ChatFilters | None = Field(
default=None,
description=(
"Narrows every search this turn runs (`search_docs` and `grep`) to "
"some source systems and/or a time window, on top of the project "
"scope in `project_ids`. When at least one search ran and every one "
"came back empty under these filters, the turn answers with a fixed "
"notice instead of letting the model answer from nothing. Send it "
"on every hop -- searches run on resume hops too."
),
)


class BuddyAgentResponse(BaseModel):
Expand All @@ -1475,6 +1486,17 @@ class BuddyAgentResponse(BaseModel):
default_factory=list[BuddyCitationSchema],
description="Sources the grounded searches drew on.",
)
reasoning: list[str] = Field(
default_factory=list[str],
description=(
"The model's reasoning on this hop, one entry per model call that "
"returned any, oldest first. Display-only: never part of "
"`messages`, so it is not carried back and never re-enters the "
"model's context. Each hop returns only its own; a caller showing "
"a whole turn concatenates the hops. Empty when the provider "
"exposes none."
),
)


class BuddyOpenRequest(BaseModel):
Expand Down
Loading
Loading