Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
20 commits
Select commit Hold shift + click to select a range
5c9f579
Tell the mentor it is walking a path, and how far it may go
DavidLeuter Sep 13, 2026
d74a358
Say which half of the product answers which question
DavidLeuter Sep 13, 2026
9c9c0ab
Stop promising buttons that no tool call made
DavidLeuter Sep 13, 2026
2ee0d51
Tell the mentor a skip is the PM's call, and pin the no-tools notice
DavidLeuter Sep 14, 2026
887fa40
Raise steps that are ready to close, and offer refreshers after a miss
DavidLeuter Sep 14, 2026
3f1dbec
Teach the mentor to file skip requests with the hire's reason
DavidLeuter Sep 14, 2026
000519e
Tell the mentor a phase is a graph, and how to answer a ready step
DavidLeuter Sep 14, 2026
4d80f1c
Tell the mentor to pass both halves of a step's placement
DavidLeuter Sep 14, 2026
318c51b
Make the onboarding path the buddy's only onboarding
DavidLeuter Sep 15, 2026
dfcf58e
Merge remote-tracking branch 'origin/dev' into feature/311-buddy-onbo…
DavidLeuter Sep 21, 2026
cc0cb80
Stop the mentor gating real work behind the path
DavidLeuter Sep 21, 2026
19e2a95
Merge origin/dev into feature/311-buddy-onboarding-tutor
DavidLeuter Sep 24, 2026
65f6f88
Expect the tutor identity in dev's buddy mode tests
DavidLeuter Sep 24, 2026
1660bc5
Tell the mentor phases are a choice, not a queue
DavidLeuter Sep 24, 2026
945cb5e
Correct broken orientation JSON once instead of dropping the packet
DavidLeuter Sep 24, 2026
b826174
Correct broken JSON once when grading answers and drawing diagrams
DavidLeuter Sep 24, 2026
bf99a17
Tidy the JSON retries and point the mentor at a hire's last wrong answer
DavidLeuter Sep 24, 2026
f2421d7
Merge origin/dev into feature/311-buddy-onboarding-tutor
DavidLeuter Sep 25, 2026
8f6adc3
Offer flag_to_pm whenever the hire asks for it
DavidLeuter Sep 26, 2026
bd740fa
Retry grading when a reply leaves an answer ungraded
DavidLeuter Sep 29, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
72 changes: 60 additions & 12 deletions src/api/routes/grading.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,8 @@

router = APIRouter()

_MAX_GRADING_ATTEMPTS = 2

SYSTEM_PROMPT = """
You grade short-text answers to onboarding knowledge-check questions.

Expand Down Expand Up @@ -47,7 +49,7 @@ class _GradedItem(BaseModel):


class _Payload(BaseModel):
results: list[_GradedItem] = []
results: list[_GradedItem]


def _build_prompt(answers: list[GradeAnswerItem]) -> list[Message]:
Expand All @@ -64,6 +66,62 @@ def _build_prompt(answers: list[GradeAnswerItem]) -> list[Message]:
]


def _grade(llm: LLMClient, to_grade: list[GradeAnswerItem]) -> dict[str, _GradedItem]:
"""Grade in one call, asking once more when the reply is unusable.

An unreadable reply marks every answer in it "could not be graded" -- which the
caller records as wrong. A small local model breaks its JSON now and then, and
a hire whose right answer came back wrong for that reason has been told
something false about their own knowledge. One correction round is cheap
against that.

Valid JSON is not enough: a reply that leaves an answer out (``{}``, or one id
missing) would record that answer as wrong just the same, so it gets the same
correction round. It must hold exactly one grade per requested id.
"""
expected_ids = {item.id for item in to_grade}
messages = _build_prompt(to_grade)
for attempt in range(_MAX_GRADING_ATTEMPTS):
try:
raw = llm.generate(messages)
except LLMUnavailableError as exc:
Comment thread
DavidLeuter marked this conversation as resolved.
raise HTTPException(status_code=503, detail=str(exc)) from exc
try:
payload = _Payload.model_validate_json(extract_json_object(raw))
graded_by_id = {item.id: item for item in payload.results}
if (
len(payload.results) != len(to_grade)
or set(graded_by_id) != expected_ids
):
raise ValueError(
"grading response does not match the requested answer ids: "
f"expected {sorted(expected_ids)}, got "
f"{[item.id for item in payload.results]}"
)
return graded_by_id
except (ValidationError, json.JSONDecodeError, ValueError) as exc:
logger.warning(
"Could not parse grade-answers output (attempt %d): %s",
attempt + 1,
exc,
)
if attempt + 1 == _MAX_GRADING_ATTEMPTS:
break
messages = [
*messages,
Message(role="assistant", content=raw),
Message(
role="user",
content=(
f"That response could not be validated: {exc}. Return the same "
"grades again as one valid JSON object matching the schema "
"exactly. Return JSON only."
),
),
]
return {}


@router.post(
"/grade-answers",
summary="Semantically grade short-text knowledge-check answers",
Expand Down Expand Up @@ -121,17 +179,7 @@ def grade_answers(
)

if to_grade:
try:
raw = llm.generate(_build_prompt(to_grade))
except LLMUnavailableError as exc:
raise HTTPException(status_code=503, detail=str(exc)) from exc

try:
payload = _Payload.model_validate_json(extract_json_object(raw))
graded_by_id = {item.id: item for item in payload.results}
except (ValidationError, json.JSONDecodeError, ValueError) as exc:
logger.warning("Could not parse grade-answers output: %s", exc)
graded_by_id = {}
graded_by_id = _grade(llm, to_grade)

for item in to_grade:
graded = graded_by_id.get(item.id)
Expand Down
29 changes: 27 additions & 2 deletions src/onboarding/buddy_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,7 +68,28 @@
_SOURCE_CHARS = 800
# How many internal search hops before we force a final answer, so a confused model
# can't loop forever gathering evidence it never uses.
_MAX_STEPS = 4
#
# Raised from 4 after a testing session: only *search-only* hops consume this budget (a
# hop that asks for a backend tool returns immediately), and a model that searched four
# times before deciding to act never got to act at all -- it hit the forced answer
# below, which has no tools, and promised the hire a button that could not exist. Six
# leaves room for a thorough answer and still bounds the loop.
_MAX_STEPS = 6

# What the model is told when the search budget is spent.
#
# It answers the hire's question with the persona still in front of it -- "offer
# `add_path_step`", "offer to claim it" -- and without this it took those at face value
# and told the hire to confirm something no tool call had produced. A model that knows
# it has no tools this turn can only do the honest thing: answer with what it has and
# leave the offer for the next turn.
_NO_TOOLS_THIS_TURN = (
"You have no tools available for this reply and cannot do anything or offer "
"anything: no confirm button can appear. Answer with what you already have. Do not "
"say you have done something, do not tell the hire to confirm, click or check "
"anything, and do not claim anything is on their screen. If something still needs "
"doing, say what it is and that you can set it up when they reply."
)


@dataclass
Expand Down Expand Up @@ -316,7 +337,11 @@ def run_agent_turn(
# Only local searches this turn -- loop and let the model reason over them.

# Step budget spent: force a final answer with no tools rather than loop forever.
forced = llm.generate(work)
#
# The notice goes to the model, not into the returned conversation: a final turn's
# messages are discarded by the caller, and a transcript carrying instructions about
# a budget nobody can see would be a strange thing to keep.
forced = llm.generate([*work, Message(role="system", content=_NO_TOOLS_THIS_TURN)])
return AgentTurnResult(
final=True,
text=forced,
Expand Down
Loading
Loading