diff --git a/strix/agents/factory.py b/strix/agents/factory.py index 063a5dd4..61af684d 100644 --- a/strix/agents/factory.py +++ b/strix/agents/factory.py @@ -61,7 +61,6 @@ from strix.tools.reporting.tool import ( list_reports, update_vulnerability_report, ) -from strix.tools.respond.tool import respond_to_user from strix.tools.thinking.tool import think from strix.tools.threat_model.tools import ( amend_threat_model, @@ -76,6 +75,7 @@ from strix.tools.todo.tools import ( mark_todo_pending, update_todo, ) +from strix.tools.wait_for_user.tool import wait_for_user from strix.tools.web_search.tool import web_get_contents, web_search @@ -509,7 +509,7 @@ def _make_shell_configurator(*, chat_completions: bool, strict_schemas: bool) -> # Tools that hand control away by parking the agent rather than ending the scan. -_PARKING_TOOLS: frozenset[str] = frozenset({"respond_to_user", "wait_for_agents"}) +_PARKING_TOOLS: frozenset[str] = frozenset({"wait_for_user", "wait_for_agents"}) def _lifecycle_tool_completed(tool_name: str, output: Any) -> bool: @@ -697,7 +697,7 @@ def build_strix_agent( agent_tools = [*_EXTRA_TOOLS, *(extra_tools or [])] if interactive: # Yielding to the user is only meaningful when one is attached. - agent_tools.append(respond_to_user) + agent_tools.append(wait_for_user) if is_root: tools: list[Tool] = [*_BASE_TOOLS, *agent_tools, finish_scan] else: diff --git a/strix/agents/prompts/system_prompt.jinja b/strix/agents/prompts/system_prompt.jinja index 4af8ff4f..dde39f14 100644 --- a/strix/agents/prompts/system_prompt.jinja +++ b/strix/agents/prompts/system_prompt.jinja @@ -33,18 +33,19 @@ INTER-AGENT MESSAGES: {% if interactive %} INTERACTIVE BEHAVIOR: - You are in an interactive conversation with a user. -- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — text is shown to the user, and then execution continues. - - To answer the user and hand control back, call respond_to_user. It delivers your message AND parks you for their reply in one call, so there is no way to answer and then forget to stop. This is the ONLY way to yield to the user. +- Everything you write as plain text is shown to the user, as you write it. Plain text is the ONLY way to talk to the user: there is no message tool and no message argument anywhere. +- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — it is shown, and then execution continues. + - To hand control to the user, call wait_for_user. It takes no arguments and says nothing; it only stops you until they reply. This is the ONLY way to yield to the user. - To wait on another AGENT (a child's report, a peer's reply), call wait_for_agents. That is not a way to reach the user. - To end the whole engagement, call the lifecycle tool: finish_scan (root) or agent_finish (subagent). - A turn that ends with plain text and no tool call does NOT stop you: the system nudges you to continue and will re-run you. Do not rely on going silent to pause — it will not pause you. -- Answering a user question: put the answer in respond_to_user's message. Do not write the answer as plain text and then fall silent — that does not reach a stopping point, it just triggers a continuation nudge. -- If all you want to do is reply and stop, that whole turn is ONE respond_to_user call carrying the answer. Do not write the answer as text and then call respond_to_user as well: the user reads it twice. -- If you do end a turn on plain text and the nudge arrives, your words already reached the user. Do not restate them: call respond_to_user with NO message to simply wait, or with only whatever you still need to add. -- You may include brief explanatory text before a tool call, and you can narrate while you work — plain text is shown to the user as you go. Narrating is free; respond_to_user is specifically the act of WAITING for the user, so do not call it just to give a status update. +- Answering the user: write the answer as plain text, then call wait_for_user in the same turn. Never call wait_for_user before you have written your reply: the user would be handed a silent turn, and the system sends you back to write it. +- Never restate, summarize, or close out what you have already written, in text or in any tool argument. Once it is written, the user has read it; the only thing left to do is call wait_for_user. +- If you end a turn on plain text and the nudge arrives, your words already reached the user. Do not repeat them: call wait_for_user. +- You can narrate while you work — plain text is shown to the user as you go. Narrating is free; wait_for_user is specifically the act of WAITING for the user, so do not call it just to give a status update. - Respond naturally when the user asks questions or gives instructions. -- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and respond_to_user only when you genuinely need the user. -- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, send it with respond_to_user. +- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and wait_for_user only when you genuinely need the user. +- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, write it and call wait_for_user. {% else %} AUTONOMOUS BEHAVIOR: - Work autonomously by default diff --git a/strix/core/execution.py b/strix/core/execution.py index 69a5cf87..ac44c53f 100644 --- a/strix/core/execution.py +++ b/strix/core/execution.py @@ -10,7 +10,7 @@ from collections.abc import Callable from functools import cache from typing import TYPE_CHECKING, Any, cast -from agents import RunConfig, Runner +from agents import ItemHelpers, MessageOutputItem, RunConfig, Runner from agents.exceptions import AgentsException, MaxTurnsExceeded, UserError from agents.sandbox.errors import ExecTransportError from openai import ( @@ -505,13 +505,18 @@ async def _run_until_lifecycle( """Drive an agent until an explicit lifecycle tool settles its status. A turn that ends without ``finish_scan``, ``agent_finish``, - ``respond_to_user``, or ``wait_for_agents`` leaves the agent ``running``: + ``wait_for_user``, or ``wait_for_agents`` leaves the agent ``running``: plain text never terminates a run and never yields to the user. Such a turn - is nudged back into a tool call, bounded by a recovery limit. + is nudged back into a tool call, bounded by a recovery limit. The same + budget covers a ``wait_for_user`` call made before anything was said to the + user since their last message: plain text is the only channel to them, so + that park would hand them a silent turn, and the agent is sent back to + write its reply instead. """ result: RunResultBase | None = None input_data: Any = initial_input recovery_limit = _INTERACTIVE_TOOL_RECOVERY_LIMIT if interactive else max(1, max_turns) + said_to_user = False while True: if coordinator.budget_stopped: @@ -568,21 +573,26 @@ async def _run_until_lifecycle( input_data = [] continue + said_to_user = said_to_user or _said_to_user(result) status = await _agent_status(coordinator, agent_id) - if status != "running": + silent_yield = ( + interactive and not said_to_user and await _parked_for_user(coordinator, agent_id) + ) + if status != "running" and not silent_yield: await coordinator.reset_recovery(agent_id) return result recoveries = await coordinator.record_recovery(agent_id) - logger.warning( - "agent %s ended a turn without a lifecycle tool call (interactive=%s); " - "forcing tool continuation (%d/%d): %s", + _log_recovery( agent_id, - interactive, + result, recoveries, recovery_limit, - _final_output_preview(result), + interactive=interactive, + silent_yield=silent_yield, ) + if silent_yield: + await coordinator.mark_running(agent_id) if recoveries >= recovery_limit: return await _exhausted_recovery(coordinator, agent_id, result, interactive=interactive) @@ -593,6 +603,7 @@ async def _run_until_lifecycle( attempt=recoveries, limit=recovery_limit, interactive=interactive, + silent_yield=silent_yield, ) @@ -884,6 +895,51 @@ async def _agent_status(coordinator: AgentCoordinator, agent_id: str) -> Status return coordinator.statuses.get(agent_id) +async def _parked_for_user(coordinator: AgentCoordinator, agent_id: str) -> bool: + async with coordinator._lock: + return ( + coordinator.statuses.get(agent_id) == "waiting" + and coordinator.wait_kinds.get(agent_id) == "user" + ) + + +def _log_recovery( + agent_id: str, + result: RunResultBase | None, + attempt: int, + limit: int, + *, + interactive: bool, + silent_yield: bool, +) -> None: + if silent_yield: + logger.warning( + "agent %s called wait_for_user without saying anything to the user; " + "sending it back to reply (%d/%d)", + agent_id, + attempt, + limit, + ) + return + logger.warning( + "agent %s ended a turn without a lifecycle tool call (interactive=%s); " + "forcing tool continuation (%d/%d): %s", + agent_id, + interactive, + attempt, + limit, + _final_output_preview(result), + ) + + +def _said_to_user(result: RunResultBase | None) -> bool: + """Whether the run produced any assistant text, the only channel to the user.""" + for item in getattr(result, "new_items", ()) or (): + if isinstance(item, MessageOutputItem) and ItemHelpers.text_message_output(item).strip(): + return True + return False + + def _final_output_preview(result: RunResultBase | None) -> str: final_output = getattr(result, "final_output", None) if final_output is None: @@ -901,15 +957,23 @@ async def _append_tool_required_message( attempt: int, limit: int, interactive: bool, + silent_yield: bool = False, ) -> list[dict[str, str]]: finish_tool = "finish_scan" if context.get("parent_id") is None else "agent_finish" - if interactive: + if silent_yield: + message = ( + "You called wait_for_user without having written anything to the user since " + "their last message, so they would be handed a silent turn. Plain text is the " + "only channel to the user: write your reply as plain text now, then call " + f"wait_for_user. This is recovery attempt {attempt}/{limit}." + ) + elif interactive: message = ( "Your previous message ended a turn without a tool call. Plain text never ends " "execution and never hands control to the user: it is shown to the user, and the " "run continues. Continue immediately and call exactly one tool. " - "If you have something to tell the user and nothing to do until they reply, " - "call respond_to_user — with no message if you have already said it. " + "If you have nothing to do until the user replies, call wait_for_user; your " + "text already reached them, so do not repeat it. " "If you are blocked waiting for another agent, call wait_for_agents. " f"If the whole engagement is complete, call {finish_tool}. " "Otherwise use the appropriate execution or planning tool. " diff --git a/strix/interface/tui/internal/render/registry.go b/strix/interface/tui/internal/render/registry.go index 9b7f4f35..7813c991 100644 --- a/strix/interface/tui/internal/render/registry.go +++ b/strix/interface/tui/internal/render/registry.go @@ -94,6 +94,8 @@ func Tool(data map[string]any) string { return renderListReports(result) case "get_report": return renderGetReport(result) + case "wait_for_user": + return renderWaitForUser() case "respond_to_user": return renderRespondToUser(args) case "finish_scan": diff --git a/strix/interface/tui/internal/render/render_test.go b/strix/interface/tui/internal/render/render_test.go index 13474b84..383dac55 100644 --- a/strix/interface/tui/internal/render/render_test.go +++ b/strix/interface/tui/internal/render/render_test.go @@ -136,7 +136,12 @@ func TestToolDispatchCoversKnownTools(t *testing.T) { []string{"report read", "not found"}, }, { - "respond_to_user", + "wait_for_user", + tool("wait_for_user", map[string]any{}, nil, "completed"), + []string{"waiting for your reply"}, + }, + { + "respond_to_user (recorded runs)", tool("respond_to_user", map[string]any{"message": "Here is the answer"}, nil, "completed"), []string{"Here is the answer", "waiting for your reply"}, }, @@ -320,6 +325,9 @@ func TestCollapseToolOnlyOutputHeavyTools(t *testing.T) { if out, expandable := CollapseTool(full, "respond_to_user", false); expandable || out != full { t.Fatal("respond_to_user must never collapse") } + if out, expandable := CollapseTool(full, "wait_for_user", false); expandable || out != full { + t.Fatal("wait_for_user must never collapse") + } } func TestReportSectionsRenderMarkdown(t *testing.T) { diff --git a/strix/interface/tui/internal/render/respond.go b/strix/interface/tui/internal/render/respond.go deleted file mode 100644 index 7b8452a3..00000000 --- a/strix/interface/tui/internal/render/respond.go +++ /dev/null @@ -1,18 +0,0 @@ -package render - -import "strings" - -// --------------------------------------------------------------------------- -// Direct replies (respond_renderer.py) -// --------------------------------------------------------------------------- - -// renderRespondToUser shows the reply as the agent's own prose, since -// respond_to_user carries the message the user is meant to read. -func renderRespondToUser(args map[string]any) string { - var b strings.Builder - if message := StringValue(args["message"]); message != "" { - b.WriteString(renderAssistantMarkdown(message) + "\n\n") - } - b.WriteString(Col(Gray).Render("○ ") + Dim().Render("waiting for your reply")) - return b.String() -} diff --git a/strix/interface/tui/internal/render/wait_user.go b/strix/interface/tui/internal/render/wait_user.go new file mode 100644 index 00000000..2e5cbcc0 --- /dev/null +++ b/strix/interface/tui/internal/render/wait_user.go @@ -0,0 +1,25 @@ +package render + +import "strings" + +// --------------------------------------------------------------------------- +// Handing the turn to the user +// --------------------------------------------------------------------------- + +// renderWaitForUser marks the agent parked for the user's reply. The call +// carries no text: whatever the agent had to say was already shown as its +// own prose. +func renderWaitForUser() string { + return Col(Gray).Render("○ ") + Dim().Render("waiting for your reply") +} + +// renderRespondToUser renders the retired respond_to_user call from recorded +// runs, whose message argument was the reply the user meant to read. +func renderRespondToUser(args map[string]any) string { + var b strings.Builder + if message := StringValue(args["message"]); message != "" { + b.WriteString(renderAssistantMarkdown(message) + "\n\n") + } + b.WriteString(renderWaitForUser()) + return b.String() +} diff --git a/strix/interface/viewer/frontend/src/components/live/tool-renderers/RespondRenderer.tsx b/strix/interface/viewer/frontend/src/components/live/tool-renderers/RespondRenderer.tsx deleted file mode 100644 index e963a35e..00000000 --- a/strix/interface/viewer/frontend/src/components/live/tool-renderers/RespondRenderer.tsx +++ /dev/null @@ -1,20 +0,0 @@ -"use client"; - -import type { ToolRendererProps } from "@/types/events"; -import Markdown from "./Markdown"; - -/** - * `respond_to_user` carries the message the user is meant to read, so it renders - * as the agent's own prose rather than as a tool call. - */ -export default function RespondRenderer({ args }: ToolRendererProps) { - const message = (args.message as string) ?? ""; - if (!message) return null; - - return ( -
h?f-h:void 0;return hy(o,y,b,_)};if(t){const s=t+fy,o=a;a=c=>c.startsWith(s)?o(c.slice(s.length)):hy(sA,!1,c,void 0,!0)}if(r){const s=a;a=o=>r({className:o,parseClassName:s})}return a},oA=e=>{const t=new Map;return e.orderSensitiveModifiers.forEach((r,a)=>{t.set(r,1e6+a)}),r=>{const a=[];let s=[];for(let o=0;o h?f-h:void 0;return hy(o,y,b,_)};if(t){const s=t+fy,o=a;a=c=>c.startsWith(s)?o(c.slice(s.length)):hy(lA,!1,c,void 0,!0)}if(r){const s=a;a=o=>r({className:o,parseClassName:s})}return a},cA=e=>{const t=new Map;return e.orderSensitiveModifiers.forEach((r,a)=>{t.set(r,1e6+a)}),r=>{const a=[];let s=[];for(let o=0;oc))return;const H=t.events.length;let z=H,Y,j;for(;z--;)if(t.events[z][0]==="exit"&&t.events[z][1].type==="chunkFlow"){if(Y){j=t.events[z][1].end;break}Y=!0}for(w(a),R=H;R{if(typeof e=="string"){nA(e,t,r);return}if(typeof e=="function"){rA(e,t,r,a);return}iA(e,t,r,a)},nA=(e,t,r)=>{const a=e===""?t:aw(t,e);a.classGroupId=r},rA=(e,t,r,a)=>{if(aA(e)){Mp(e(a),t,r,a);return}t.validators===null&&(t.validators=[]),t.validators.push(KT(r,e))},iA=(e,t,r,a)=>{const s=Object.entries(e),o=s.length;for(let c=0;c"isThemeGetter"in e&&e.isThemeGetter===!0,sA=e=>{if(e<1)return{get:()=>{},set:()=>{}};let t=0,r=Object.create(null),a=Object.create(null);const s=(o,c)=>{r[o]=c,t++,t>e&&(t=0,a=r,r=Object.create(null))};return{get(o){let c=r[o];if(c!==void 0)return c;if((c=a[o])!==void 0)return s(o,c),c},set(o,c){o in r?r[o]=c:s(o,c)}}},Vm="!",fy=":",lA=[],hy=(e,t,r,a,s)=>({modifiers:e,hasImportantModifier:t,baseClassName:r,maybePostfixModifierPosition:a,isExternal:s}),oA=e=>{const{prefix:t,experimentalParseClassName:r}=e;let a=s=>{const o=[];let c=0,d=0,h=0,f;const p=s.length;for(let N=0;Nc))return;const H=t.events.length;let z=H,Y,j;for(;z--;)if(t.events[z][0]==="exit"&&t.events[z][1].type==="chunkFlow"){if(Y){j=t.events[z][1].end;break}Y=!0}for(w(a),R=H;R