From 5b307c56e2b6133fba1bb6f43c64d268adf4bd9e Mon Sep 17 00:00:00 2001 From: Linus Date: Fri, 18 Sep 2026 16:31:57 +0200 Subject: [PATCH 1/2] Tell the buddy persona which mode it is in The backend sends `capabilities_enabled` and `team_mode` on every agent hop and the service read neither, so a hire who turned capabilities off met a mentor still offering to act, and a manager met the new-hire mentor. Both flags now reach `build_persona`. Capabilities off states that it is answering from the project's material and offers nothing it cannot do. Team mode is its own persona: the reader is a project's manager, situations are facts rather than judgments of the person, areas open before their tools exist, and a change is only ever offered for the manager to confirm. The hire's arrival, claim and assessment clauses are dropped by mode as well as by mounting, so a future backend cannot put them in front of a manager. Read on every hop rather than only the first: the persona is rebuilt each time, so a resume that lost a flag would finish the turn in the other mode. Closes #193 Co-Authored-By: Claude Opus 5 --- src/api/routes/buddy.py | 6 ++ src/api/schemas.py | 21 ++++ src/onboarding/buddy_agent.py | 37 ++++++- src/onboarding/buddy_persona.py | 137 +++++++++++++++++++++++- tests/api/test_buddy.py | 42 ++++++++ tests/onboarding/test_buddy_agent.py | 86 +++++++++++++++ tests/onboarding/test_buddy_persona.py | 138 +++++++++++++++++++++++++ 7 files changed, 460 insertions(+), 7 deletions(-) diff --git a/src/api/routes/buddy.py b/src/api/routes/buddy.py index 916b175..498a0ce 100644 --- a/src/api/routes/buddy.py +++ b/src/api/routes/buddy.py @@ -84,6 +84,10 @@ def buddy_agent( Executes ``search_docs`` locally (retrieval + citations) and returns as soon as it either has a final answer or needs a backend-only tool run. The backend carries the ``messages`` list back verbatim, each pending tool's result appended as a ``tool``. + + ``capabilities_enabled`` and ``team_mode`` pick the persona. Both are read on every + hop rather than only the first: the persona is rebuilt each time, so a resume hop + that omitted one would finish the turn in the other mode. """ messages = [_to_message(m) for m in body.messages] backend_tools = [_to_toolspec(t) for t in body.backend_tools] @@ -101,6 +105,8 @@ def buddy_agent( contribution_verb_past=body.vocabulary.contribution_verb_past, ), project_ids=frozenset(body.project_ids), + capabilities_enabled=body.capabilities_enabled, + team_mode=body.team_mode, ) except LLMUnavailableError as exc: raise HTTPException( diff --git a/src/api/schemas.py b/src/api/schemas.py index a80bd40..56782de 100644 --- a/src/api/schemas.py +++ b/src/api/schemas.py @@ -1391,6 +1391,27 @@ class BuddyAgentRequest(BaseModel): "to no projects (admitting no material)." ), ) + capabilities_enabled: bool = Field( + default=True, + description=( + "False when the reader asked the corpus rather than the mentor: the " + "persona then says it is answering from the project's material only " + "and offers to do nothing. Send it on every hop — the persona is " + "rebuilt on each one, so a hop that drops it changes mode mid-turn." + ), + ) + team_mode: bool = Field( + default=False, + description=( + "True when the reader is a project's manager asking about that " + "project's team, rather than a hire asking about their own " + "onboarding. The persona then addresses a manager, states team " + "members' situations as facts rather than judgments, and offers " + "changes for the manager to confirm instead of claiming to have made " + "them. Send it on every hop, for the same reason as " + "`capabilities_enabled`." + ), + ) class BuddyAgentResponse(BaseModel): diff --git a/src/onboarding/buddy_agent.py b/src/onboarding/buddy_agent.py index e5a4adf..35e4b02 100644 --- a/src/onboarding/buddy_agent.py +++ b/src/onboarding/buddy_agent.py @@ -93,8 +93,15 @@ def _persona_prompt( summary: str | None, tool_names: Collection[str], vocabulary: Vocabulary, + capabilities_enabled: bool, + team_mode: bool, ) -> str: - persona = build_persona(tool_names, vocabulary) + persona = build_persona( + tool_names, + vocabulary, + capabilities_enabled=capabilities_enabled, + team_mode=team_mode, + ) if not summary: return persona return persona + _SUMMARY_HEADER + summary @@ -111,7 +118,12 @@ def _ensure_persona( summary: str | None, tool_names: Collection[str], vocabulary: Vocabulary, + capabilities_enabled: bool, + team_mode: bool, ) -> list[Message]: + # The system message a resume carries is replaced, not kept: only its summary + # survives. So both modes have to arrive on every hop -- a hop that dropped one + # would rebuild the default persona mid-turn and answer a manager as a hire. effective_summary = summary rest = messages if messages and messages[0]["role"] == "system": @@ -120,7 +132,13 @@ def _ensure_persona( rest = messages[1:] persona_msg = Message( role="system", - content=_persona_prompt(effective_summary, tool_names, vocabulary), + content=_persona_prompt( + effective_summary, + tool_names, + vocabulary, + capabilities_enabled, + team_mode, + ), ) return [persona_msg, *rest] @@ -191,6 +209,8 @@ def run_agent_turn( prior_summary: str | None = None, vocabulary: Vocabulary = DEFAULT_VOCABULARY, project_ids: frozenset[str] | None = None, + capabilities_enabled: bool = True, + team_mode: bool = False, ) -> AgentTurnResult: """Runs one agent turn: executes ``search_docs`` locally, pauses on backend tools. @@ -201,6 +221,10 @@ def run_agent_turn( ``prior_summary`` stands in for everything older than ``messages``. + ``capabilities_enabled`` and ``team_mode`` choose the persona and must be passed + on every hop, because the persona is rebuilt on every hop: a resume that lost + either would answer the rest of the turn as the default mentor. + This turn does not fold anything, and used not to be able to say that. A ``summarize_upto`` argument asked it to compact the oldest window messages *before* the model began composing a reply, and because the caller's cursor @@ -219,7 +243,14 @@ def run_agent_turn( # The persona describes exactly the tools this hire was mounted, never a fixed # catalogue: the backend decides what a given role can even have, and a mentor # told about a tool it does not have will offer the hire something impossible. - work = _ensure_persona(window, summary, {SEARCH_DOCS, *backend_names}, vocabulary) + work = _ensure_persona( + window, + summary, + {SEARCH_DOCS, *backend_names}, + vocabulary, + capabilities_enabled, + team_mode, + ) resolved_exclusions = exclusions if exclusions is not None else SourceExclusions() citations: list[Citation] = [] seen_chunk_ids: set[str] = set() diff --git a/src/onboarding/buddy_persona.py b/src/onboarding/buddy_persona.py index 4c96ce7..8894650 100644 --- a/src/onboarding/buddy_persona.py +++ b/src/onboarding/buddy_persona.py @@ -11,6 +11,18 @@ Any fixed clause must be checked against that vocabulary too -- wording like "clone the repository" puts engineering work in front of a role that has none. + +Two flags from the backend choose *which* persona is assembled, and both arrive on +every hop because the persona is rebuilt on every hop: + +- ``capabilities_enabled=False`` is the hire asking the corpus rather than the + mentor. No backend tool is mounted, so the tool-gated clauses fall away on their + own -- but an identity that still offers to act makes the resulting refusal read + as a bug, so this mode says plainly what it can and cannot do. +- ``team_mode=True`` is a different reader altogether: the manager of one project, + asking about that project's team. The hire-directed clauses are dropped by mode + rather than only by mounting, because a manager must never be told what *they* + should work on next. """ from collections.abc import Collection @@ -81,6 +93,63 @@ "later, so never call it a score, a result, or final.\n" ) +_SEARCH_ONLY_CLAUSE = ( + "- This turn you have `search_docs` and nothing else: you are answering from " + "the project's own material, not acting on anybody's behalf. Answer what the " + "material covers, say plainly where it does not, and do not offer to record, " + "claim, flag or change anything -- nothing is mounted to do it with, so an " + "offer you make here cannot be kept.\n" +) + +_TEAM_IDENTITY = ( + "You are the onboarding buddy in team mode: you are talking to the manager of " + "one project about that project's team. The person reading you is responsible " + "for these people; they are not being onboarded themselves. Never greet them " + "as a new hire, never describe their own onboarding, and never suggest what " + "they should work on next.\n" + "How you work:\n" +) + +_TEAM_READ_TOOLS = ( + "get_team_attention", + "find_member", + "get_member_progress", +) + +# The rule a manager's trust rests on, and the one a model breaks most readily: a +# person is not their situation. Work nobody has reviewed says something about the +# review queue, and a model that reports it as "Sam is behind" has invented a fact +# about Sam -- to the one reader who can act on it. +_TEAM_FACTS_CLAUSE = ( + "- Describe somebody's situation as facts about that situation, never as a " + "judgment of the person. Work waiting on a review is the reviewer's move, not " + "a failing of whoever is waiting on it. Never rank the team against each " + "other, never call anybody slow or behind, and never supply a reason nobody " + "gave you.\n" + "- When the tools do not cover what the manager asked, say so and say who " + "would know, rather than filling the gap yourself.\n" +) + +_TEAM_AREA_CLAUSE = ( + "- The manager's work is grouped into areas that stay closed until you open " + "one. Call `open_area` for the area they are actually asking about, and its " + "tools become available on your *next* step, not this one. Do not open an area " + "on the chance it might be useful.\n" +) + +# Deliberately a rule about you rather than about a named tool. Which actions are +# mounted changes from hop to hop as areas open, and this is the one sentence that +# must never be missing -- it is what stops you telling a manager that a change +# happened. It stays true when nothing is mounted to propose, and it names nothing +# that is not there. +_TEAM_PROPOSE_CLAUSE = ( + "- You never make a change yourself. A tool that would change something only " + "offers it: the manager sees what it would do and confirms it outside this " + "conversation, or does not. So say what you have put in front of them and " + "stop there -- never that a change is done, queued, applied or taken care of, " + "and never carry on as though they had already confirmed it.\n" +) + _STATE_TOOLS = ( "get_arrival_steps", "get_my_metrics", @@ -92,21 +161,40 @@ def build_persona( tool_names: Collection[str], vocabulary: Vocabulary = DEFAULT_VOCABULARY, + *, + capabilities_enabled: bool = True, + team_mode: bool = False, ) -> str: - """Assembles the mentor's persona for one hire. + """Assembles the persona for one reader of the buddy. Args: - tool_names: Every tool mounted for this hire, backend and local. A clause + tool_names: Every tool mounted for this reader, backend and local. A clause whose tools are absent is omitted rather than softened, so the persona - never mentions a capability this hire does not have. + never mentions a capability this reader does not have. vocabulary: What one unit of this hire's accepted work is called. Defaults - to the engineering wording when a caller supplies nothing. + to the engineering wording when a caller supplies nothing. Unused in + team mode, whose prose is about people rather than about their work. + capabilities_enabled: False when the reader asked the corpus rather than the + mentor. Nothing but ``search_docs`` is mounted, and the persona says so + instead of offering what it cannot do. + team_mode: True when the reader is a project's manager asking about that + project's team, rather than a hire asking about their own onboarding. Returns: The system prompt, without any conversation summary appended. """ available = set(tool_names) + if team_mode: + return _team_persona(available, capabilities_enabled) parts = [_IDENTITY] + if not capabilities_enabled: + # Nothing below is mounted, so every clause that follows would drop anyway. + # Said outright, because a mentor who still sounds able to act turns its own + # refusal into something that reads like a fault. + parts.append(_SEARCH_ONLY_CLAUSE) + parts.append(_GROUNDING_CLAUSE + ".\n") + parts.append(_FIXTURE_CLAUSE) + return "".join(parts) # First clause, because it is first in the conversation: what has to be true # before somebody can work comes before what they should work on. The backend @@ -147,3 +235,44 @@ def build_persona( "path to it." ) return "".join(parts) + + +def _team_persona(available: set[str], capabilities_enabled: bool) -> str: + """Assembles the persona a project's manager meets. + + The hire's clauses are not reused and not softened. Arrival, claiming a task and + the competency assessment are all about the reader's own onboarding, and their + tools are never mounted here anyway -- but dropping them by mode as well means a + backend that one day mounts one of them cannot put "let's settle where you're + starting from" in front of somebody's manager. + """ + parts = [_TEAM_IDENTITY] + if not capabilities_enabled: + parts.append(_SEARCH_ONLY_CLAUSE) + parts.append(_TEAM_FACTS_CLAUSE) + parts.append(_GROUNDING_CLAUSE + ".\n") + parts.append(_FIXTURE_CLAUSE) + return "".join(parts) + + team_reads = [name for name in _TEAM_READ_TOOLS if name in available] + if team_reads: + rendered = ", ".join(f"`{name}`" for name in team_reads) + parts.append( + f"- Use the team tools ({rendered}) for who on this project needs the " + "manager's attention and how one person is getting on, and " + "`search_docs` for how the project itself works. A member id the tools " + "give you is how you name somebody; never guess one.\n" + ) + parts.append(_TEAM_FACTS_CLAUSE) + if "open_area" in available: + parts.append(_TEAM_AREA_CLAUSE) + parts.append(_TEAM_PROPOSE_CLAUSE) + # No escalation offer: `flag_to_pm` raises a question *to* a manager, and the + # reader here is the manager. + parts.append(_GROUNDING_CLAUSE + ".\n") + parts.append(_FIXTURE_CLAUSE) + parts.append( + "- Lead with what needs the manager now and keep the rest short. Their " + "attention is the scarce thing, and a list of everything spends it." + ) + return "".join(parts) diff --git a/tests/api/test_buddy.py b/tests/api/test_buddy.py index 170e6c7..661f636 100644 --- a/tests/api/test_buddy.py +++ b/tests/api/test_buddy.py @@ -138,3 +138,45 @@ def test_open_stream_greets_and_carries_no_memory_note() -> None: assert '"type": "done"' in body # The caller cannot persist a note it is never handed. assert '"memory"' not in body + + +def test_team_mode_reaches_the_persona_the_backend_carries_back( + client: TestClient, +) -> None: + response = client.post( + _URL, + json={ + "messages": [{"role": "user", "content": "who needs me?"}], + "team_mode": True, + }, + ) + + assert response.status_code == 200 + persona = response.json()["messages"][0]["content"] + assert "manager of one project" in persona + + +def test_capabilities_off_reaches_the_persona(client: TestClient) -> None: + response = client.post( + _URL, + json={ + "messages": [{"role": "user", "content": "how does deployment work?"}], + "capabilities_enabled": False, + }, + ) + + assert response.status_code == 200 + assert "`search_docs` and nothing else" in response.json()["messages"][0]["content"] + + +def test_omitting_both_modes_is_the_hire_mentor(client: TestClient) -> None: + """A caller that has never heard of either field gets exactly today's buddy.""" + response = client.post( + _URL, + json={"messages": [{"role": "user", "content": "hello"}]}, + ) + + assert response.status_code == 200 + persona = response.json()["messages"][0]["content"] + assert "the mentor who guides a new hire" in persona + assert "manager of one project" not in persona diff --git a/tests/onboarding/test_buddy_agent.py b/tests/onboarding/test_buddy_agent.py index ef81877..88c7df1 100644 --- a/tests/onboarding/test_buddy_agent.py +++ b/tests/onboarding/test_buddy_agent.py @@ -326,3 +326,89 @@ def test_the_turn_never_folds_however_long_the_window_is() -> None: contents = [msg.get("content") for msg in result.messages] assert all(f"m{i}" in contents for i in range(1, 40)) assert result.final is True + + +def test_team_mode_builds_the_manager_persona() -> None: + llm = ScriptedLLMClient(turns=[], answer="answer") + + run_agent_turn( + [_user("who needs me?")], + [_tool("get_team_attention")], + llm, + StubVectorStore(), + team_mode=True, + ) + + persona = _system_of(llm.chat_calls[0]) + assert "manager of one project" in persona + assert "`get_team_attention`" in persona + + +def test_capabilities_off_builds_the_search_only_persona() -> None: + llm = ScriptedLLMClient(turns=[], answer="answer") + + run_agent_turn( + [_user("how does deployment work?")], + [], + llm, + StubVectorStore(), + capabilities_enabled=False, + ) + + assert "`search_docs` and nothing else" in _system_of(llm.chat_calls[0]) + + +def test_both_modes_survive_a_resume_hop_that_already_has_a_system_message() -> None: + """The resume replaces the system message and keeps only its summary, so a mode + that did not arrive again would finish the turn as the default mentor.""" + llm = ScriptedLLMClient(turns=[[("get_team_attention", {})]], answer="answer") + first = run_agent_turn( + [_user("who needs me?")], + [_tool("get_team_attention")], + llm, + StubVectorStore(), + prior_summary="Earlier turns about the team.", + team_mode=True, + ) + + llm2 = ScriptedLLMClient(turns=[], answer="answer") + run_agent_turn( + [*first.messages, Message(role="tool", content="nobody is blocked")], + [_tool("get_team_attention")], + llm2, + StubVectorStore(), + team_mode=True, + ) + + persona = _system_of(llm2.chat_calls[0]) + assert "manager of one project" in persona + assert "Earlier turns about the team." in persona + + +def test_a_resume_that_lost_team_mode_is_the_default_mentor_again() -> None: + """Pinning the failure the flag exists to prevent: it is per hop, not per turn.""" + llm = ScriptedLLMClient(turns=[], answer="answer") + first = run_agent_turn( + [_user("who needs me?")], + [_tool("get_team_attention")], + llm, + StubVectorStore(), + team_mode=True, + ) + + llm2 = ScriptedLLMClient(turns=[], answer="answer") + run_agent_turn( + first.messages, [_tool("get_team_attention")], llm2, StubVectorStore() + ) + + assert "manager of one project" not in _system_of(llm2.chat_calls[0]) + + +def test_the_modes_default_to_todays_behaviour() -> None: + llm = ScriptedLLMClient(turns=[], answer="answer") + + run_agent_turn([_user("hello")], [_GET_MY_METRICS], llm, StubVectorStore()) + + persona = _system_of(llm.chat_calls[0]) + assert "the mentor who guides a new hire" in persona + assert "search_docs` and nothing else" not in persona diff --git a/tests/onboarding/test_buddy_persona.py b/tests/onboarding/test_buddy_persona.py index ca871ef..457b679 100644 --- a/tests/onboarding/test_buddy_persona.py +++ b/tests/onboarding/test_buddy_persona.py @@ -161,3 +161,141 @@ def test_the_grounding_rule_survives_every_toolset() -> None: # test/fixture caveat are safety rules, not capabilities. assert "Ground every claim" in persona assert "test, fixture, or sample-data files" in persona + + +# --- Modes ----------------------------------------------------------------- +# +# Two flags pick which persona is assembled. The through-line above still holds +# inside each one: what is not mounted is never mentioned. What these add is that a +# mode is not a softening of the default -- capabilities off must not sound able to +# act, and team mode must not sound like it is talking to the person it describes. + +_TEAM_TOOLS = ( + "search_docs", + "get_team_attention", + "find_member", + "get_member_progress", + "open_area", +) + + +def test_capabilities_off_says_it_is_answering_from_the_material() -> None: + persona = build_persona(["search_docs"], capabilities_enabled=False) + + assert "`search_docs` and nothing else" in persona + assert "do not offer to record, claim, flag or change anything" in persona + + +def test_capabilities_off_keeps_grounding_and_the_fixture_caveat() -> None: + """Safety rules are not capabilities, so no mode drops them.""" + persona = build_persona(["search_docs"], capabilities_enabled=False) + + assert "Ground every claim" in persona + assert "test, fixture, or sample-data files" in persona + + +def test_capabilities_off_never_offers_an_escalation_or_a_claim() -> None: + # Nothing is mounted with capabilities off, so an offer here is one the hire + # cannot take up -- the refusal that follows reads as a fault in the product. + persona = build_persona(_ALL_TOOLS, capabilities_enabled=False) + + assert "flag_to_pm" not in persona + assert "claim_goal" not in persona + assert "get_arrival_steps" not in persona + + +def test_team_mode_addresses_the_manager_not_a_hire() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True) + + assert "manager of one project" in persona + assert "Never greet them as a new hire" in persona + assert "the mentor who guides a new hire" not in persona + + +def test_team_mode_drops_every_hire_directed_clause() -> None: + """Dropped by mode, not only by mounting: a manager must never be asked to + settle where *they* are starting from.""" + persona = build_persona([*_TEAM_TOOLS, *_ALL_TOOLS], team_mode=True) + + assert "get_arrival_steps" not in persona + assert "claim_goal" not in persona + assert "get_competencies_to_assess" not in persona + assert "record_assessment" not in persona + assert "hire-state tools" not in persona + + +def test_team_mode_states_situations_as_facts_about_the_situation() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True) + + assert "never as a judgment of the person" in persona + assert "the reviewer's move" in persona + assert "never call anybody slow or behind" in persona + + +def test_team_mode_never_claims_to_have_made_a_change() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True) + + assert "You never make a change yourself" in persona + assert "confirms it outside this conversation" in persona + + +def test_the_proposal_rule_holds_before_any_action_is_mounted() -> None: + """Actions arrive only once an area is opened, and the rule that stops the model + announcing a change must be there on the hop before that.""" + persona = build_persona(["search_docs", "get_team_attention"], team_mode=True) + + assert "You never make a change yourself" in persona + + +def test_team_mode_explains_areas_only_when_open_area_is_mounted() -> None: + with_areas = build_persona(_TEAM_TOOLS, team_mode=True) + without = build_persona( + [t for t in _TEAM_TOOLS if t != "open_area"], team_mode=True + ) + + assert "`open_area`" in with_areas + assert "available on your *next* step" in with_areas + assert "open_area" not in without + + +def test_team_mode_lists_only_the_team_tools_that_are_mounted() -> None: + persona = build_persona( + ["search_docs", "get_team_attention"], + team_mode=True, + ) + + assert "`get_team_attention`" in persona + assert "`find_member`" not in persona + assert "`get_member_progress`" not in persona + + +def test_team_mode_without_team_tools_describes_none_of_them() -> None: + persona = build_persona(["search_docs"], team_mode=True) + + assert "the team tools" not in persona + # The identity and the safety rules are all that is left, and they are enough. + assert "manager of one project" in persona + assert "Ground every claim" in persona + + +def test_team_mode_offers_no_escalation_to_a_pm() -> None: + """`flag_to_pm` raises a question *to* a manager; the reader here is one.""" + persona = build_persona([*_TEAM_TOOLS, "flag_to_pm"], team_mode=True) + + assert "flag_to_pm" not in persona + + +def test_team_mode_with_capabilities_off_is_search_only_and_still_a_manager() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True, capabilities_enabled=False) + + assert "manager of one project" in persona + assert "`search_docs` and nothing else" in persona + assert "never as a judgment of the person" in persona + assert "get_team_attention" not in persona + + +def test_the_defaults_are_todays_behaviour() -> None: + assert build_persona(_ALL_TOOLS) == build_persona( + _ALL_TOOLS, capabilities_enabled=True, team_mode=False + ) + assert build_persona(_ALL_TOOLS, DEFAULT_VOCABULARY) == build_persona(_ALL_TOOLS) From 3c6b86dd2ec6cc33dc01350c3fd71518ae4070dd Mon Sep 17 00:00:00 2001 From: Linus Date: Fri, 18 Sep 2026 16:34:19 +0200 Subject: [PATCH 2/2] Greet a manager in team mode instead of a new hire The open endpoint knew one reader. A manager opening team mode was welcomed back to their own onboarding and told their team's stalls as if they were theirs, because STATE is the team's attention list and the prompt describes it as the reader's own work in flight. `team_mode` on the open request picks a team system prompt and a team fallback greeting. It says who is reading, that STATE is about other people, that a situation is a fact rather than a judgment of the person, and that the suggested step is a question a manager would ask about their team. The two-part marker format is deliberately identical, so the marker, the held-back suffix and the done payload stay one code path: a second format here would be a second parser to keep in step with this one. The hire prompt and its fallback are unchanged. Closes #194 Co-Authored-By: Claude Opus 5 --- src/api/routes/buddy.py | 4 ++ src/api/schemas.py | 12 +++- src/onboarding/buddy_open.py | 70 ++++++++++++++++++++-- tests/api/test_buddy.py | 57 ++++++++++++++++++ tests/onboarding/test_buddy_open.py | 92 +++++++++++++++++++++++++++++ 5 files changed, 230 insertions(+), 5 deletions(-) diff --git a/src/api/routes/buddy.py b/src/api/routes/buddy.py index 498a0ce..d8512df 100644 --- a/src/api/routes/buddy.py +++ b/src/api/routes/buddy.py @@ -200,6 +200,9 @@ def buddy_open_stream( Emits ``token`` events carrying the greeting as it arrives and one terminal ``done`` carrying the whole greeting and any suggested action. Degrades to a plain welcome rather than erroring: opening the buddy must never fail the page. + + With ``team_mode`` the reader is a project's manager and ``state`` is their team's + attention list, so both the prompt and that plain welcome address a manager. """ def event_stream() -> Iterator[str]: @@ -209,6 +212,7 @@ def event_stream() -> Iterator[str]: recent=[_to_message(m) for m in body.recent], state=body.state, llm=llm, + team_mode=body.team_mode, ): yield sse_event(event) except LLMUnavailableError as exc: diff --git a/src/api/schemas.py b/src/api/schemas.py index 56782de..7b95a11 100644 --- a/src/api/schemas.py +++ b/src/api/schemas.py @@ -1458,7 +1458,17 @@ class BuddyOpenRequest(BaseModel): default="", description=( "A plain-text snapshot of the hire's current state (pull requests, tasks, " - "competencies) for the greeting to ground itself in." + "competencies) for the greeting to ground itself in. In team mode this is " + "the team's attention list instead." + ), + ) + team_mode: bool = Field( + default=False, + description=( + "True when a project's manager is opening a team conversation. The " + "greeting then addresses a manager about their team — `state` is the " + "team's attention list, not the reader's own onboarding — and the " + "suggested next step is a question about the team." ), ) diff --git a/src/onboarding/buddy_open.py b/src/onboarding/buddy_open.py index 64ea4bc..81a0eb8 100644 --- a/src/onboarding/buddy_open.py +++ b/src/onboarding/buddy_open.py @@ -17,6 +17,11 @@ generated ahead of it — strict JSON cannot be streamed as prose, which is why this uses markers. A marker arrives one chunk at a time, so a partial one must be held back as a candidate, never matched early or emitted. + +A visit can also be opened by a project's manager in team mode, where the reader is +responsible for the people the state describes rather than being one of them. That +swaps the system prompt and the fallback greeting; the two-part format and +everything that parses it are shared. """ import json @@ -28,6 +33,14 @@ _FALLBACK_GREETING = "Welcome back! How can I help with your onboarding today?" +# A manager opening team mode is not coming back to their own onboarding, so the +# hire's welcome is wrong in the one place it is guaranteed to be read: the model +# is unavailable and this is the whole greeting. +_TEAM_FALLBACK_GREETING = ( + "You're in team mode. Ask me who needs your attention, or how somebody on the " + "team is getting on." +) + def _format_recent(recent: list[Message]) -> str: lines = [ @@ -66,12 +79,48 @@ def _format_recent(recent: list[Message]) -> str: "you only read from it." ) +# The same two-part shape, deliberately: the marker, the held-back suffix and the +# done payload are parsed by one code path, so a team greeting that formatted itself +# differently would be a second parser to keep in step. Only the reader changes. +_TEAM_STREAM_SYSTEM = ( + "You are a perceptive onboarding buddy greeting a project's manager as they " + "open the chat in team mode. They manage one project and are asking about that " + "project's team -- they are not being onboarded themselves. You keep a private, " + "durable memory note about your conversations with this manager, and you speak " + "to them directly.\n" + "You are given: your MEMORY of these conversations (may be empty on the first " + "visit), the RECENT conversation since you last updated that memory (may be " + "empty), and STATE: the team's current attention list -- who is waiting on " + "somebody else and who has stalled. STATE is about the team, never about the " + "manager reading you.\n" + "Write your reply in exactly two parts, in this order, with nothing before the " + "first part:\n" + "PART 1 -- the greeting, as plain prose with no label and no quotes: a short, " + "first-person opener (2-4 sentences) that greets the manager and says the one " + "thing most worth their attention right now, grounded only in the state and " + "the memory. Be specific, not generic. Never invent facts that are not in the " + "memory or the state. Never welcome them as a new hire, never describe their " + "own onboarding, and never suggest what they should work on. Describe a " + "person's situation as a fact about that situation, never as a judgment of the " + "person -- work waiting on a review is the reviewer's move, not a failing of " + "whoever is waiting on it. The manager reads this as you type it, so it must " + "come first.\n" + f"PART 2 -- the line {_ACTION_MARKER} on its own, then ONE suggested next step " + 'as JSON {"label": short button text, "question": the message to send when the ' + "manager clicks it}, or the word none when nothing fits. The question is one a " + 'manager would ask about their team (for example "Who is blocked right now?" ' + 'or "How is the newest joiner getting on?"), never one about their own work.\n' + "Do not rewrite or restate your memory note. It is maintained separately; here " + "you only read from it." +) + def stream_session( memory: str | None, recent: list[Message], state: str, llm: LLMClient, + team_mode: bool = False, ) -> Iterator[dict[str, object]]: """Stream the greeting as the model writes it. @@ -93,15 +142,28 @@ def stream_session( An unavailable model yields the plain welcome -- opening a visit must never fail the page. + ### Team mode + + ``team_mode`` swaps the system prompt and the fallback, and nothing else. The + reader is a project's manager and ``state`` is their team's attention list, so a + hire's greeting would welcome them back to an onboarding that is not theirs and + describe their team's stalls as their own. The marker, the held-back suffix and + the ``done`` payload are the same code either way -- a second format here would + be a second parser to keep in step with this one. + @param memory: The mentor's durable note, or None on a first visit. Read from, never rewritten here -- see the module docstring. @param recent: The window since the memory was last updated. - @param state: A snapshot of the hire's current state. + @param state: A snapshot of the hire's current state, or in team mode the team's + attention list. + @param team_mode: True when a project's manager is opening a team conversation. @return: ``token`` events carrying the greeting as it arrives, then one terminal ``done`` carrying the whole greeting and any action. """ + system = _TEAM_STREAM_SYSTEM if team_mode else _STREAM_SYSTEM + fallback = _TEAM_FALLBACK_GREETING if team_mode else _FALLBACK_GREETING prompt = [ - Message(role="system", content=_STREAM_SYSTEM), + Message(role="system", content=system), Message( role="user", content=( @@ -155,7 +217,7 @@ def stream_session( greeting += emit yield {"type": "token", "content": emit} except LLMUnavailableError: - yield {"type": "done", "greeting": _FALLBACK_GREETING, "action": None} + yield {"type": "done", "greeting": fallback, "action": None} return # Whatever is still held back was never a marker after all, so it is prose. @@ -169,7 +231,7 @@ def stream_session( # Byte-identical to the concatenated tokens, deliberately. The client renders # the tokens and the caller persists this; if they differed, the message a hire # watched arrive would not be the one they see after a reload. - "greeting": greeting if greeting.strip() else _FALLBACK_GREETING, + "greeting": greeting if greeting.strip() else fallback, "action": ( {"label": label, "question": question} if label and question else None ), diff --git a/tests/api/test_buddy.py b/tests/api/test_buddy.py index 661f636..024aea4 100644 --- a/tests/api/test_buddy.py +++ b/tests/api/test_buddy.py @@ -180,3 +180,60 @@ def test_omitting_both_modes_is_the_hire_mentor(client: TestClient) -> None: persona = response.json()["messages"][0]["content"] assert "the mentor who guides a new hire" in persona assert "manager of one project" not in persona + + +def test_open_stream_takes_team_mode_and_greets_a_manager() -> None: + """The stub echoes one fixed answer, so what the flag changes here is the prompt + it was asked with -- which is the part the route is responsible for threading.""" + recorded: list[str] = [] + + class _RecordingLLM(StubLLMClient): + def stream(self, messages: list[Any]) -> Any: + recorded.append(str(messages[0]["content"])) + return super().stream(messages) + + app.dependency_overrides[get_llm] = lambda: _RecordingLLM( + generate_response="Two people are waiting on a review." + ) + try: + client = TestClient(app) + response = client.post( + "/api/v1/onboarding/buddy/open/stream", + json={ + "memory": None, + "recent": [], + "state": "Who needs attention: ...", + "team_mode": True, + }, + ) + finally: + app.dependency_overrides.clear() + + assert response.status_code == 200 + assert "Two people are waiting on a review." in response.text + assert "project's manager" in recorded[0] + + +def test_open_stream_without_team_mode_is_the_hire_greeting() -> None: + recorded: list[str] = [] + + class _RecordingLLM(StubLLMClient): + def stream(self, messages: list[Any]) -> Any: + recorded.append(str(messages[0]["content"])) + return super().stream(messages) + + app.dependency_overrides[get_llm] = lambda: _RecordingLLM( + generate_response="Welcome back, Sam!" + ) + try: + client = TestClient(app) + response = client.post( + "/api/v1/onboarding/buddy/open/stream", + json={"memory": None, "recent": [], "state": "1 open PR"}, + ) + finally: + app.dependency_overrides.clear() + + assert response.status_code == 200 + assert "greeting a new hire" in recorded[0] + assert "project's manager" not in recorded[0] diff --git a/tests/onboarding/test_buddy_open.py b/tests/onboarding/test_buddy_open.py index 58067ce..2b79d36 100644 --- a/tests/onboarding/test_buddy_open.py +++ b/tests/onboarding/test_buddy_open.py @@ -228,3 +228,95 @@ def test_the_done_greeting_is_exactly_what_was_streamed() -> None: events = list(stream_session(None, [], "", llm)) assert _done(events)["greeting"] == _tokens(events) + + +# --- Team mode ------------------------------------------------------------- +# +# The reader is a project's manager and STATE is their team's attention list. What +# must not change is everything that parses the model's reply: one format, one +# parser, whoever is reading. + + +def _system_of(prompt: list[Message] | None) -> str: + assert prompt is not None + return str(prompt[0]["content"]) + + +def test_team_mode_greets_a_manager_about_their_team() -> None: + llm = _StreamingStubLLM(["hi"]) + + list(stream_session(memory=None, recent=[], state="", llm=llm, team_mode=True)) + + system = _system_of(llm.last_prompt) + assert "project's manager" in system + assert "the team's current attention list" in system + assert "Never welcome them as a new hire" in system + + +def test_the_hire_greeting_is_untouched_by_default() -> None: + llm = _StreamingStubLLM(["hi"]) + + list(stream_session(memory=None, recent=[], state="", llm=llm)) + + system = _system_of(llm.last_prompt) + assert "greeting a new hire" in system + assert "project's manager" not in system + + +def test_the_team_suggestion_is_a_question_about_the_team() -> None: + llm = _StreamingStubLLM(["hi"]) + + list(stream_session(memory=None, recent=[], state="", llm=llm, team_mode=True)) + + assert "never one about their own work" in _system_of(llm.last_prompt) + + +def test_an_unavailable_model_falls_back_per_mode() -> None: + """The fallback is the one greeting guaranteed to be read in full, so a hire's + welcome would be wrong exactly where it is least recoverable.""" + team = _StreamingStubLLM(LLMUnavailableError("down")) + hire = _StreamingStubLLM(LLMUnavailableError("down")) + + team_done = _done(list(stream_session(None, [], "", team, team_mode=True))) + hire_done = _done(list(stream_session(None, [], "", hire))) + + assert team_done["greeting"] == ( + "You're in team mode. Ask me who needs your attention, or how somebody on " + "the team is getting on." + ) + assert hire_done["greeting"] == ( + "Welcome back! How can I help with your onboarding today?" + ) + + +def test_a_blank_team_greeting_falls_back_to_the_team_welcome() -> None: + llm = _StreamingStubLLM(["\n<<>>\n" + _ACTION]) + + done = _done(list(stream_session(None, [], "", llm, team_mode=True))) + + assert "team mode" in str(done["greeting"]) + + +def test_the_marker_and_the_done_payload_are_identical_in_team_mode() -> None: + llm = _StreamingStubLLM( + ["Two people ", "are waiting on a review.", "\n<<>>\n", _ACTION] + ) + + events = list(stream_session(None, [], "", llm, team_mode=True)) + + assert _tokens(events) == "Two people are waiting on a review." + done = _done(events) + assert done["greeting"] == _tokens(events) + assert done["action"] == { + "label": "What next?", + "question": "What should I work on?", + } + + +def test_a_team_marker_split_across_chunks_is_still_found() -> None: + llm = _StreamingStubLLM(["Nobody is blocked.", "\n<<>>\n", _ACTION]) + + events = list(stream_session(None, [], "", llm, team_mode=True)) + + assert _tokens(events).strip() == "Nobody is blocked." + assert _done(events)["action"] is not None