diff --git a/src/api/routes/buddy.py b/src/api/routes/buddy.py index 916b175..d8512df 100644 --- a/src/api/routes/buddy.py +++ b/src/api/routes/buddy.py @@ -84,6 +84,10 @@ def buddy_agent( Executes ``search_docs`` locally (retrieval + citations) and returns as soon as it either has a final answer or needs a backend-only tool run. The backend carries the ``messages`` list back verbatim, each pending tool's result appended as a ``tool``. + + ``capabilities_enabled`` and ``team_mode`` pick the persona. Both are read on every + hop rather than only the first: the persona is rebuilt each time, so a resume hop + that omitted one would finish the turn in the other mode. """ messages = [_to_message(m) for m in body.messages] backend_tools = [_to_toolspec(t) for t in body.backend_tools] @@ -101,6 +105,8 @@ def buddy_agent( contribution_verb_past=body.vocabulary.contribution_verb_past, ), project_ids=frozenset(body.project_ids), + capabilities_enabled=body.capabilities_enabled, + team_mode=body.team_mode, ) except LLMUnavailableError as exc: raise HTTPException( @@ -194,6 +200,9 @@ def buddy_open_stream( Emits ``token`` events carrying the greeting as it arrives and one terminal ``done`` carrying the whole greeting and any suggested action. Degrades to a plain welcome rather than erroring: opening the buddy must never fail the page. + + With ``team_mode`` the reader is a project's manager and ``state`` is their team's + attention list, so both the prompt and that plain welcome address a manager. """ def event_stream() -> Iterator[str]: @@ -203,6 +212,7 @@ def event_stream() -> Iterator[str]: recent=[_to_message(m) for m in body.recent], state=body.state, llm=llm, + team_mode=body.team_mode, ): yield sse_event(event) except LLMUnavailableError as exc: diff --git a/src/api/schemas.py b/src/api/schemas.py index a80bd40..7b95a11 100644 --- a/src/api/schemas.py +++ b/src/api/schemas.py @@ -1391,6 +1391,27 @@ class BuddyAgentRequest(BaseModel): "to no projects (admitting no material)." ), ) + capabilities_enabled: bool = Field( + default=True, + description=( + "False when the reader asked the corpus rather than the mentor: the " + "persona then says it is answering from the project's material only " + "and offers to do nothing. Send it on every hop — the persona is " + "rebuilt on each one, so a hop that drops it changes mode mid-turn." + ), + ) + team_mode: bool = Field( + default=False, + description=( + "True when the reader is a project's manager asking about that " + "project's team, rather than a hire asking about their own " + "onboarding. The persona then addresses a manager, states team " + "members' situations as facts rather than judgments, and offers " + "changes for the manager to confirm instead of claiming to have made " + "them. Send it on every hop, for the same reason as " + "`capabilities_enabled`." + ), + ) class BuddyAgentResponse(BaseModel): @@ -1437,7 +1458,17 @@ class BuddyOpenRequest(BaseModel): default="", description=( "A plain-text snapshot of the hire's current state (pull requests, tasks, " - "competencies) for the greeting to ground itself in." + "competencies) for the greeting to ground itself in. In team mode this is " + "the team's attention list instead." + ), + ) + team_mode: bool = Field( + default=False, + description=( + "True when a project's manager is opening a team conversation. The " + "greeting then addresses a manager about their team — `state` is the " + "team's attention list, not the reader's own onboarding — and the " + "suggested next step is a question about the team." ), ) diff --git a/src/onboarding/buddy_agent.py b/src/onboarding/buddy_agent.py index e5a4adf..35e4b02 100644 --- a/src/onboarding/buddy_agent.py +++ b/src/onboarding/buddy_agent.py @@ -93,8 +93,15 @@ def _persona_prompt( summary: str | None, tool_names: Collection[str], vocabulary: Vocabulary, + capabilities_enabled: bool, + team_mode: bool, ) -> str: - persona = build_persona(tool_names, vocabulary) + persona = build_persona( + tool_names, + vocabulary, + capabilities_enabled=capabilities_enabled, + team_mode=team_mode, + ) if not summary: return persona return persona + _SUMMARY_HEADER + summary @@ -111,7 +118,12 @@ def _ensure_persona( summary: str | None, tool_names: Collection[str], vocabulary: Vocabulary, + capabilities_enabled: bool, + team_mode: bool, ) -> list[Message]: + # The system message a resume carries is replaced, not kept: only its summary + # survives. So both modes have to arrive on every hop -- a hop that dropped one + # would rebuild the default persona mid-turn and answer a manager as a hire. effective_summary = summary rest = messages if messages and messages[0]["role"] == "system": @@ -120,7 +132,13 @@ def _ensure_persona( rest = messages[1:] persona_msg = Message( role="system", - content=_persona_prompt(effective_summary, tool_names, vocabulary), + content=_persona_prompt( + effective_summary, + tool_names, + vocabulary, + capabilities_enabled, + team_mode, + ), ) return [persona_msg, *rest] @@ -191,6 +209,8 @@ def run_agent_turn( prior_summary: str | None = None, vocabulary: Vocabulary = DEFAULT_VOCABULARY, project_ids: frozenset[str] | None = None, + capabilities_enabled: bool = True, + team_mode: bool = False, ) -> AgentTurnResult: """Runs one agent turn: executes ``search_docs`` locally, pauses on backend tools. @@ -201,6 +221,10 @@ def run_agent_turn( ``prior_summary`` stands in for everything older than ``messages``. + ``capabilities_enabled`` and ``team_mode`` choose the persona and must be passed + on every hop, because the persona is rebuilt on every hop: a resume that lost + either would answer the rest of the turn as the default mentor. + This turn does not fold anything, and used not to be able to say that. A ``summarize_upto`` argument asked it to compact the oldest window messages *before* the model began composing a reply, and because the caller's cursor @@ -219,7 +243,14 @@ def run_agent_turn( # The persona describes exactly the tools this hire was mounted, never a fixed # catalogue: the backend decides what a given role can even have, and a mentor # told about a tool it does not have will offer the hire something impossible. - work = _ensure_persona(window, summary, {SEARCH_DOCS, *backend_names}, vocabulary) + work = _ensure_persona( + window, + summary, + {SEARCH_DOCS, *backend_names}, + vocabulary, + capabilities_enabled, + team_mode, + ) resolved_exclusions = exclusions if exclusions is not None else SourceExclusions() citations: list[Citation] = [] seen_chunk_ids: set[str] = set() diff --git a/src/onboarding/buddy_open.py b/src/onboarding/buddy_open.py index 64ea4bc..81a0eb8 100644 --- a/src/onboarding/buddy_open.py +++ b/src/onboarding/buddy_open.py @@ -17,6 +17,11 @@ generated ahead of it — strict JSON cannot be streamed as prose, which is why this uses markers. A marker arrives one chunk at a time, so a partial one must be held back as a candidate, never matched early or emitted. + +A visit can also be opened by a project's manager in team mode, where the reader is +responsible for the people the state describes rather than being one of them. That +swaps the system prompt and the fallback greeting; the two-part format and +everything that parses it are shared. """ import json @@ -28,6 +33,14 @@ _FALLBACK_GREETING = "Welcome back! How can I help with your onboarding today?" +# A manager opening team mode is not coming back to their own onboarding, so the +# hire's welcome is wrong in the one place it is guaranteed to be read: the model +# is unavailable and this is the whole greeting. +_TEAM_FALLBACK_GREETING = ( + "You're in team mode. Ask me who needs your attention, or how somebody on the " + "team is getting on." +) + def _format_recent(recent: list[Message]) -> str: lines = [ @@ -66,12 +79,48 @@ def _format_recent(recent: list[Message]) -> str: "you only read from it." ) +# The same two-part shape, deliberately: the marker, the held-back suffix and the +# done payload are parsed by one code path, so a team greeting that formatted itself +# differently would be a second parser to keep in step. Only the reader changes. +_TEAM_STREAM_SYSTEM = ( + "You are a perceptive onboarding buddy greeting a project's manager as they " + "open the chat in team mode. They manage one project and are asking about that " + "project's team -- they are not being onboarded themselves. You keep a private, " + "durable memory note about your conversations with this manager, and you speak " + "to them directly.\n" + "You are given: your MEMORY of these conversations (may be empty on the first " + "visit), the RECENT conversation since you last updated that memory (may be " + "empty), and STATE: the team's current attention list -- who is waiting on " + "somebody else and who has stalled. STATE is about the team, never about the " + "manager reading you.\n" + "Write your reply in exactly two parts, in this order, with nothing before the " + "first part:\n" + "PART 1 -- the greeting, as plain prose with no label and no quotes: a short, " + "first-person opener (2-4 sentences) that greets the manager and says the one " + "thing most worth their attention right now, grounded only in the state and " + "the memory. Be specific, not generic. Never invent facts that are not in the " + "memory or the state. Never welcome them as a new hire, never describe their " + "own onboarding, and never suggest what they should work on. Describe a " + "person's situation as a fact about that situation, never as a judgment of the " + "person -- work waiting on a review is the reviewer's move, not a failing of " + "whoever is waiting on it. The manager reads this as you type it, so it must " + "come first.\n" + f"PART 2 -- the line {_ACTION_MARKER} on its own, then ONE suggested next step " + 'as JSON {"label": short button text, "question": the message to send when the ' + "manager clicks it}, or the word none when nothing fits. The question is one a " + 'manager would ask about their team (for example "Who is blocked right now?" ' + 'or "How is the newest joiner getting on?"), never one about their own work.\n' + "Do not rewrite or restate your memory note. It is maintained separately; here " + "you only read from it." +) + def stream_session( memory: str | None, recent: list[Message], state: str, llm: LLMClient, + team_mode: bool = False, ) -> Iterator[dict[str, object]]: """Stream the greeting as the model writes it. @@ -93,15 +142,28 @@ def stream_session( An unavailable model yields the plain welcome -- opening a visit must never fail the page. + ### Team mode + + ``team_mode`` swaps the system prompt and the fallback, and nothing else. The + reader is a project's manager and ``state`` is their team's attention list, so a + hire's greeting would welcome them back to an onboarding that is not theirs and + describe their team's stalls as their own. The marker, the held-back suffix and + the ``done`` payload are the same code either way -- a second format here would + be a second parser to keep in step with this one. + @param memory: The mentor's durable note, or None on a first visit. Read from, never rewritten here -- see the module docstring. @param recent: The window since the memory was last updated. - @param state: A snapshot of the hire's current state. + @param state: A snapshot of the hire's current state, or in team mode the team's + attention list. + @param team_mode: True when a project's manager is opening a team conversation. @return: ``token`` events carrying the greeting as it arrives, then one terminal ``done`` carrying the whole greeting and any action. """ + system = _TEAM_STREAM_SYSTEM if team_mode else _STREAM_SYSTEM + fallback = _TEAM_FALLBACK_GREETING if team_mode else _FALLBACK_GREETING prompt = [ - Message(role="system", content=_STREAM_SYSTEM), + Message(role="system", content=system), Message( role="user", content=( @@ -155,7 +217,7 @@ def stream_session( greeting += emit yield {"type": "token", "content": emit} except LLMUnavailableError: - yield {"type": "done", "greeting": _FALLBACK_GREETING, "action": None} + yield {"type": "done", "greeting": fallback, "action": None} return # Whatever is still held back was never a marker after all, so it is prose. @@ -169,7 +231,7 @@ def stream_session( # Byte-identical to the concatenated tokens, deliberately. The client renders # the tokens and the caller persists this; if they differed, the message a hire # watched arrive would not be the one they see after a reload. - "greeting": greeting if greeting.strip() else _FALLBACK_GREETING, + "greeting": greeting if greeting.strip() else fallback, "action": ( {"label": label, "question": question} if label and question else None ), diff --git a/src/onboarding/buddy_persona.py b/src/onboarding/buddy_persona.py index 4c96ce7..8894650 100644 --- a/src/onboarding/buddy_persona.py +++ b/src/onboarding/buddy_persona.py @@ -11,6 +11,18 @@ Any fixed clause must be checked against that vocabulary too -- wording like "clone the repository" puts engineering work in front of a role that has none. + +Two flags from the backend choose *which* persona is assembled, and both arrive on +every hop because the persona is rebuilt on every hop: + +- ``capabilities_enabled=False`` is the hire asking the corpus rather than the + mentor. No backend tool is mounted, so the tool-gated clauses fall away on their + own -- but an identity that still offers to act makes the resulting refusal read + as a bug, so this mode says plainly what it can and cannot do. +- ``team_mode=True`` is a different reader altogether: the manager of one project, + asking about that project's team. The hire-directed clauses are dropped by mode + rather than only by mounting, because a manager must never be told what *they* + should work on next. """ from collections.abc import Collection @@ -81,6 +93,63 @@ "later, so never call it a score, a result, or final.\n" ) +_SEARCH_ONLY_CLAUSE = ( + "- This turn you have `search_docs` and nothing else: you are answering from " + "the project's own material, not acting on anybody's behalf. Answer what the " + "material covers, say plainly where it does not, and do not offer to record, " + "claim, flag or change anything -- nothing is mounted to do it with, so an " + "offer you make here cannot be kept.\n" +) + +_TEAM_IDENTITY = ( + "You are the onboarding buddy in team mode: you are talking to the manager of " + "one project about that project's team. The person reading you is responsible " + "for these people; they are not being onboarded themselves. Never greet them " + "as a new hire, never describe their own onboarding, and never suggest what " + "they should work on next.\n" + "How you work:\n" +) + +_TEAM_READ_TOOLS = ( + "get_team_attention", + "find_member", + "get_member_progress", +) + +# The rule a manager's trust rests on, and the one a model breaks most readily: a +# person is not their situation. Work nobody has reviewed says something about the +# review queue, and a model that reports it as "Sam is behind" has invented a fact +# about Sam -- to the one reader who can act on it. +_TEAM_FACTS_CLAUSE = ( + "- Describe somebody's situation as facts about that situation, never as a " + "judgment of the person. Work waiting on a review is the reviewer's move, not " + "a failing of whoever is waiting on it. Never rank the team against each " + "other, never call anybody slow or behind, and never supply a reason nobody " + "gave you.\n" + "- When the tools do not cover what the manager asked, say so and say who " + "would know, rather than filling the gap yourself.\n" +) + +_TEAM_AREA_CLAUSE = ( + "- The manager's work is grouped into areas that stay closed until you open " + "one. Call `open_area` for the area they are actually asking about, and its " + "tools become available on your *next* step, not this one. Do not open an area " + "on the chance it might be useful.\n" +) + +# Deliberately a rule about you rather than about a named tool. Which actions are +# mounted changes from hop to hop as areas open, and this is the one sentence that +# must never be missing -- it is what stops you telling a manager that a change +# happened. It stays true when nothing is mounted to propose, and it names nothing +# that is not there. +_TEAM_PROPOSE_CLAUSE = ( + "- You never make a change yourself. A tool that would change something only " + "offers it: the manager sees what it would do and confirms it outside this " + "conversation, or does not. So say what you have put in front of them and " + "stop there -- never that a change is done, queued, applied or taken care of, " + "and never carry on as though they had already confirmed it.\n" +) + _STATE_TOOLS = ( "get_arrival_steps", "get_my_metrics", @@ -92,21 +161,40 @@ def build_persona( tool_names: Collection[str], vocabulary: Vocabulary = DEFAULT_VOCABULARY, + *, + capabilities_enabled: bool = True, + team_mode: bool = False, ) -> str: - """Assembles the mentor's persona for one hire. + """Assembles the persona for one reader of the buddy. Args: - tool_names: Every tool mounted for this hire, backend and local. A clause + tool_names: Every tool mounted for this reader, backend and local. A clause whose tools are absent is omitted rather than softened, so the persona - never mentions a capability this hire does not have. + never mentions a capability this reader does not have. vocabulary: What one unit of this hire's accepted work is called. Defaults - to the engineering wording when a caller supplies nothing. + to the engineering wording when a caller supplies nothing. Unused in + team mode, whose prose is about people rather than about their work. + capabilities_enabled: False when the reader asked the corpus rather than the + mentor. Nothing but ``search_docs`` is mounted, and the persona says so + instead of offering what it cannot do. + team_mode: True when the reader is a project's manager asking about that + project's team, rather than a hire asking about their own onboarding. Returns: The system prompt, without any conversation summary appended. """ available = set(tool_names) + if team_mode: + return _team_persona(available, capabilities_enabled) parts = [_IDENTITY] + if not capabilities_enabled: + # Nothing below is mounted, so every clause that follows would drop anyway. + # Said outright, because a mentor who still sounds able to act turns its own + # refusal into something that reads like a fault. + parts.append(_SEARCH_ONLY_CLAUSE) + parts.append(_GROUNDING_CLAUSE + ".\n") + parts.append(_FIXTURE_CLAUSE) + return "".join(parts) # First clause, because it is first in the conversation: what has to be true # before somebody can work comes before what they should work on. The backend @@ -147,3 +235,44 @@ def build_persona( "path to it." ) return "".join(parts) + + +def _team_persona(available: set[str], capabilities_enabled: bool) -> str: + """Assembles the persona a project's manager meets. + + The hire's clauses are not reused and not softened. Arrival, claiming a task and + the competency assessment are all about the reader's own onboarding, and their + tools are never mounted here anyway -- but dropping them by mode as well means a + backend that one day mounts one of them cannot put "let's settle where you're + starting from" in front of somebody's manager. + """ + parts = [_TEAM_IDENTITY] + if not capabilities_enabled: + parts.append(_SEARCH_ONLY_CLAUSE) + parts.append(_TEAM_FACTS_CLAUSE) + parts.append(_GROUNDING_CLAUSE + ".\n") + parts.append(_FIXTURE_CLAUSE) + return "".join(parts) + + team_reads = [name for name in _TEAM_READ_TOOLS if name in available] + if team_reads: + rendered = ", ".join(f"`{name}`" for name in team_reads) + parts.append( + f"- Use the team tools ({rendered}) for who on this project needs the " + "manager's attention and how one person is getting on, and " + "`search_docs` for how the project itself works. A member id the tools " + "give you is how you name somebody; never guess one.\n" + ) + parts.append(_TEAM_FACTS_CLAUSE) + if "open_area" in available: + parts.append(_TEAM_AREA_CLAUSE) + parts.append(_TEAM_PROPOSE_CLAUSE) + # No escalation offer: `flag_to_pm` raises a question *to* a manager, and the + # reader here is the manager. + parts.append(_GROUNDING_CLAUSE + ".\n") + parts.append(_FIXTURE_CLAUSE) + parts.append( + "- Lead with what needs the manager now and keep the rest short. Their " + "attention is the scarce thing, and a list of everything spends it." + ) + return "".join(parts) diff --git a/tests/api/test_buddy.py b/tests/api/test_buddy.py index 170e6c7..024aea4 100644 --- a/tests/api/test_buddy.py +++ b/tests/api/test_buddy.py @@ -138,3 +138,102 @@ def test_open_stream_greets_and_carries_no_memory_note() -> None: assert '"type": "done"' in body # The caller cannot persist a note it is never handed. assert '"memory"' not in body + + +def test_team_mode_reaches_the_persona_the_backend_carries_back( + client: TestClient, +) -> None: + response = client.post( + _URL, + json={ + "messages": [{"role": "user", "content": "who needs me?"}], + "team_mode": True, + }, + ) + + assert response.status_code == 200 + persona = response.json()["messages"][0]["content"] + assert "manager of one project" in persona + + +def test_capabilities_off_reaches_the_persona(client: TestClient) -> None: + response = client.post( + _URL, + json={ + "messages": [{"role": "user", "content": "how does deployment work?"}], + "capabilities_enabled": False, + }, + ) + + assert response.status_code == 200 + assert "`search_docs` and nothing else" in response.json()["messages"][0]["content"] + + +def test_omitting_both_modes_is_the_hire_mentor(client: TestClient) -> None: + """A caller that has never heard of either field gets exactly today's buddy.""" + response = client.post( + _URL, + json={"messages": [{"role": "user", "content": "hello"}]}, + ) + + assert response.status_code == 200 + persona = response.json()["messages"][0]["content"] + assert "the mentor who guides a new hire" in persona + assert "manager of one project" not in persona + + +def test_open_stream_takes_team_mode_and_greets_a_manager() -> None: + """The stub echoes one fixed answer, so what the flag changes here is the prompt + it was asked with -- which is the part the route is responsible for threading.""" + recorded: list[str] = [] + + class _RecordingLLM(StubLLMClient): + def stream(self, messages: list[Any]) -> Any: + recorded.append(str(messages[0]["content"])) + return super().stream(messages) + + app.dependency_overrides[get_llm] = lambda: _RecordingLLM( + generate_response="Two people are waiting on a review." + ) + try: + client = TestClient(app) + response = client.post( + "/api/v1/onboarding/buddy/open/stream", + json={ + "memory": None, + "recent": [], + "state": "Who needs attention: ...", + "team_mode": True, + }, + ) + finally: + app.dependency_overrides.clear() + + assert response.status_code == 200 + assert "Two people are waiting on a review." in response.text + assert "project's manager" in recorded[0] + + +def test_open_stream_without_team_mode_is_the_hire_greeting() -> None: + recorded: list[str] = [] + + class _RecordingLLM(StubLLMClient): + def stream(self, messages: list[Any]) -> Any: + recorded.append(str(messages[0]["content"])) + return super().stream(messages) + + app.dependency_overrides[get_llm] = lambda: _RecordingLLM( + generate_response="Welcome back, Sam!" + ) + try: + client = TestClient(app) + response = client.post( + "/api/v1/onboarding/buddy/open/stream", + json={"memory": None, "recent": [], "state": "1 open PR"}, + ) + finally: + app.dependency_overrides.clear() + + assert response.status_code == 200 + assert "greeting a new hire" in recorded[0] + assert "project's manager" not in recorded[0] diff --git a/tests/onboarding/test_buddy_agent.py b/tests/onboarding/test_buddy_agent.py index ef81877..88c7df1 100644 --- a/tests/onboarding/test_buddy_agent.py +++ b/tests/onboarding/test_buddy_agent.py @@ -326,3 +326,89 @@ def test_the_turn_never_folds_however_long_the_window_is() -> None: contents = [msg.get("content") for msg in result.messages] assert all(f"m{i}" in contents for i in range(1, 40)) assert result.final is True + + +def test_team_mode_builds_the_manager_persona() -> None: + llm = ScriptedLLMClient(turns=[], answer="answer") + + run_agent_turn( + [_user("who needs me?")], + [_tool("get_team_attention")], + llm, + StubVectorStore(), + team_mode=True, + ) + + persona = _system_of(llm.chat_calls[0]) + assert "manager of one project" in persona + assert "`get_team_attention`" in persona + + +def test_capabilities_off_builds_the_search_only_persona() -> None: + llm = ScriptedLLMClient(turns=[], answer="answer") + + run_agent_turn( + [_user("how does deployment work?")], + [], + llm, + StubVectorStore(), + capabilities_enabled=False, + ) + + assert "`search_docs` and nothing else" in _system_of(llm.chat_calls[0]) + + +def test_both_modes_survive_a_resume_hop_that_already_has_a_system_message() -> None: + """The resume replaces the system message and keeps only its summary, so a mode + that did not arrive again would finish the turn as the default mentor.""" + llm = ScriptedLLMClient(turns=[[("get_team_attention", {})]], answer="answer") + first = run_agent_turn( + [_user("who needs me?")], + [_tool("get_team_attention")], + llm, + StubVectorStore(), + prior_summary="Earlier turns about the team.", + team_mode=True, + ) + + llm2 = ScriptedLLMClient(turns=[], answer="answer") + run_agent_turn( + [*first.messages, Message(role="tool", content="nobody is blocked")], + [_tool("get_team_attention")], + llm2, + StubVectorStore(), + team_mode=True, + ) + + persona = _system_of(llm2.chat_calls[0]) + assert "manager of one project" in persona + assert "Earlier turns about the team." in persona + + +def test_a_resume_that_lost_team_mode_is_the_default_mentor_again() -> None: + """Pinning the failure the flag exists to prevent: it is per hop, not per turn.""" + llm = ScriptedLLMClient(turns=[], answer="answer") + first = run_agent_turn( + [_user("who needs me?")], + [_tool("get_team_attention")], + llm, + StubVectorStore(), + team_mode=True, + ) + + llm2 = ScriptedLLMClient(turns=[], answer="answer") + run_agent_turn( + first.messages, [_tool("get_team_attention")], llm2, StubVectorStore() + ) + + assert "manager of one project" not in _system_of(llm2.chat_calls[0]) + + +def test_the_modes_default_to_todays_behaviour() -> None: + llm = ScriptedLLMClient(turns=[], answer="answer") + + run_agent_turn([_user("hello")], [_GET_MY_METRICS], llm, StubVectorStore()) + + persona = _system_of(llm.chat_calls[0]) + assert "the mentor who guides a new hire" in persona + assert "search_docs` and nothing else" not in persona diff --git a/tests/onboarding/test_buddy_open.py b/tests/onboarding/test_buddy_open.py index 58067ce..2b79d36 100644 --- a/tests/onboarding/test_buddy_open.py +++ b/tests/onboarding/test_buddy_open.py @@ -228,3 +228,95 @@ def test_the_done_greeting_is_exactly_what_was_streamed() -> None: events = list(stream_session(None, [], "", llm)) assert _done(events)["greeting"] == _tokens(events) + + +# --- Team mode ------------------------------------------------------------- +# +# The reader is a project's manager and STATE is their team's attention list. What +# must not change is everything that parses the model's reply: one format, one +# parser, whoever is reading. + + +def _system_of(prompt: list[Message] | None) -> str: + assert prompt is not None + return str(prompt[0]["content"]) + + +def test_team_mode_greets_a_manager_about_their_team() -> None: + llm = _StreamingStubLLM(["hi"]) + + list(stream_session(memory=None, recent=[], state="", llm=llm, team_mode=True)) + + system = _system_of(llm.last_prompt) + assert "project's manager" in system + assert "the team's current attention list" in system + assert "Never welcome them as a new hire" in system + + +def test_the_hire_greeting_is_untouched_by_default() -> None: + llm = _StreamingStubLLM(["hi"]) + + list(stream_session(memory=None, recent=[], state="", llm=llm)) + + system = _system_of(llm.last_prompt) + assert "greeting a new hire" in system + assert "project's manager" not in system + + +def test_the_team_suggestion_is_a_question_about_the_team() -> None: + llm = _StreamingStubLLM(["hi"]) + + list(stream_session(memory=None, recent=[], state="", llm=llm, team_mode=True)) + + assert "never one about their own work" in _system_of(llm.last_prompt) + + +def test_an_unavailable_model_falls_back_per_mode() -> None: + """The fallback is the one greeting guaranteed to be read in full, so a hire's + welcome would be wrong exactly where it is least recoverable.""" + team = _StreamingStubLLM(LLMUnavailableError("down")) + hire = _StreamingStubLLM(LLMUnavailableError("down")) + + team_done = _done(list(stream_session(None, [], "", team, team_mode=True))) + hire_done = _done(list(stream_session(None, [], "", hire))) + + assert team_done["greeting"] == ( + "You're in team mode. Ask me who needs your attention, or how somebody on " + "the team is getting on." + ) + assert hire_done["greeting"] == ( + "Welcome back! How can I help with your onboarding today?" + ) + + +def test_a_blank_team_greeting_falls_back_to_the_team_welcome() -> None: + llm = _StreamingStubLLM(["\n<<>>\n" + _ACTION]) + + done = _done(list(stream_session(None, [], "", llm, team_mode=True))) + + assert "team mode" in str(done["greeting"]) + + +def test_the_marker_and_the_done_payload_are_identical_in_team_mode() -> None: + llm = _StreamingStubLLM( + ["Two people ", "are waiting on a review.", "\n<<>>\n", _ACTION] + ) + + events = list(stream_session(None, [], "", llm, team_mode=True)) + + assert _tokens(events) == "Two people are waiting on a review." + done = _done(events) + assert done["greeting"] == _tokens(events) + assert done["action"] == { + "label": "What next?", + "question": "What should I work on?", + } + + +def test_a_team_marker_split_across_chunks_is_still_found() -> None: + llm = _StreamingStubLLM(["Nobody is blocked.", "\n<<>>\n", _ACTION]) + + events = list(stream_session(None, [], "", llm, team_mode=True)) + + assert _tokens(events).strip() == "Nobody is blocked." + assert _done(events)["action"] is not None diff --git a/tests/onboarding/test_buddy_persona.py b/tests/onboarding/test_buddy_persona.py index ca871ef..457b679 100644 --- a/tests/onboarding/test_buddy_persona.py +++ b/tests/onboarding/test_buddy_persona.py @@ -161,3 +161,141 @@ def test_the_grounding_rule_survives_every_toolset() -> None: # test/fixture caveat are safety rules, not capabilities. assert "Ground every claim" in persona assert "test, fixture, or sample-data files" in persona + + +# --- Modes ----------------------------------------------------------------- +# +# Two flags pick which persona is assembled. The through-line above still holds +# inside each one: what is not mounted is never mentioned. What these add is that a +# mode is not a softening of the default -- capabilities off must not sound able to +# act, and team mode must not sound like it is talking to the person it describes. + +_TEAM_TOOLS = ( + "search_docs", + "get_team_attention", + "find_member", + "get_member_progress", + "open_area", +) + + +def test_capabilities_off_says_it_is_answering_from_the_material() -> None: + persona = build_persona(["search_docs"], capabilities_enabled=False) + + assert "`search_docs` and nothing else" in persona + assert "do not offer to record, claim, flag or change anything" in persona + + +def test_capabilities_off_keeps_grounding_and_the_fixture_caveat() -> None: + """Safety rules are not capabilities, so no mode drops them.""" + persona = build_persona(["search_docs"], capabilities_enabled=False) + + assert "Ground every claim" in persona + assert "test, fixture, or sample-data files" in persona + + +def test_capabilities_off_never_offers_an_escalation_or_a_claim() -> None: + # Nothing is mounted with capabilities off, so an offer here is one the hire + # cannot take up -- the refusal that follows reads as a fault in the product. + persona = build_persona(_ALL_TOOLS, capabilities_enabled=False) + + assert "flag_to_pm" not in persona + assert "claim_goal" not in persona + assert "get_arrival_steps" not in persona + + +def test_team_mode_addresses_the_manager_not_a_hire() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True) + + assert "manager of one project" in persona + assert "Never greet them as a new hire" in persona + assert "the mentor who guides a new hire" not in persona + + +def test_team_mode_drops_every_hire_directed_clause() -> None: + """Dropped by mode, not only by mounting: a manager must never be asked to + settle where *they* are starting from.""" + persona = build_persona([*_TEAM_TOOLS, *_ALL_TOOLS], team_mode=True) + + assert "get_arrival_steps" not in persona + assert "claim_goal" not in persona + assert "get_competencies_to_assess" not in persona + assert "record_assessment" not in persona + assert "hire-state tools" not in persona + + +def test_team_mode_states_situations_as_facts_about_the_situation() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True) + + assert "never as a judgment of the person" in persona + assert "the reviewer's move" in persona + assert "never call anybody slow or behind" in persona + + +def test_team_mode_never_claims_to_have_made_a_change() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True) + + assert "You never make a change yourself" in persona + assert "confirms it outside this conversation" in persona + + +def test_the_proposal_rule_holds_before_any_action_is_mounted() -> None: + """Actions arrive only once an area is opened, and the rule that stops the model + announcing a change must be there on the hop before that.""" + persona = build_persona(["search_docs", "get_team_attention"], team_mode=True) + + assert "You never make a change yourself" in persona + + +def test_team_mode_explains_areas_only_when_open_area_is_mounted() -> None: + with_areas = build_persona(_TEAM_TOOLS, team_mode=True) + without = build_persona( + [t for t in _TEAM_TOOLS if t != "open_area"], team_mode=True + ) + + assert "`open_area`" in with_areas + assert "available on your *next* step" in with_areas + assert "open_area" not in without + + +def test_team_mode_lists_only_the_team_tools_that_are_mounted() -> None: + persona = build_persona( + ["search_docs", "get_team_attention"], + team_mode=True, + ) + + assert "`get_team_attention`" in persona + assert "`find_member`" not in persona + assert "`get_member_progress`" not in persona + + +def test_team_mode_without_team_tools_describes_none_of_them() -> None: + persona = build_persona(["search_docs"], team_mode=True) + + assert "the team tools" not in persona + # The identity and the safety rules are all that is left, and they are enough. + assert "manager of one project" in persona + assert "Ground every claim" in persona + + +def test_team_mode_offers_no_escalation_to_a_pm() -> None: + """`flag_to_pm` raises a question *to* a manager; the reader here is one.""" + persona = build_persona([*_TEAM_TOOLS, "flag_to_pm"], team_mode=True) + + assert "flag_to_pm" not in persona + + +def test_team_mode_with_capabilities_off_is_search_only_and_still_a_manager() -> None: + persona = build_persona(_TEAM_TOOLS, team_mode=True, capabilities_enabled=False) + + assert "manager of one project" in persona + assert "`search_docs` and nothing else" in persona + assert "never as a judgment of the person" in persona + assert "get_team_attention" not in persona + + +def test_the_defaults_are_todays_behaviour() -> None: + assert build_persona(_ALL_TOOLS) == build_persona( + _ALL_TOOLS, capabilities_enabled=True, team_mode=False + ) + assert build_persona(_ALL_TOOLS, DEFAULT_VOCABULARY) == build_persona(_ALL_TOOLS)