refactor(agents): replace respond_to_user with a text-free wait_for_user

Plain text is now the only channel to the user. wait_for_user takes no
arguments and only parks the agent, so a reply can no longer be written
twice (once as text, once as a tool argument). The driver bounces a
wait_for_user call that follows no assistant text back into recovery.
TUI and viewer render wait_for_user as the waiting marker alone and keep
rendering recorded respond_to_user events with their message.
This commit is contained in:
Ahmed Allam 2026-10-05 00:39:24 +00:00 • committed by Ahmed Allam
parent 6b7e3b77e8
commit 16a316540c
20 changed files with 525 additions and 277 deletions

View file

@ -61,7 +61,6 @@ from strix.tools.reporting.tool import (
list_reports,
update_vulnerability_report,
)
from strix.tools.respond.tool import respond_to_user
from strix.tools.thinking.tool import think
from strix.tools.threat_model.tools import (
amend_threat_model,
@ -76,6 +75,7 @@ from strix.tools.todo.tools import (
mark_todo_pending,
update_todo,
)
from strix.tools.wait_for_user.tool import wait_for_user
from strix.tools.web_search.tool import web_get_contents, web_search
@ -509,7 +509,7 @@ def _make_shell_configurator(*, chat_completions: bool, strict_schemas: bool) ->
# Tools that hand control away by parking the agent rather than ending the scan.
_PARKING_TOOLS: frozenset[str] = frozenset({"respond_to_user", "wait_for_agents"})
_PARKING_TOOLS: frozenset[str] = frozenset({"wait_for_user", "wait_for_agents"})
def _lifecycle_tool_completed(tool_name: str, output: Any) -> bool:
@ -697,7 +697,7 @@ def build_strix_agent(
agent_tools = [*_EXTRA_TOOLS, *(extra_tools or [])]
if interactive:
# Yielding to the user is only meaningful when one is attached.
agent_tools.append(respond_to_user)
agent_tools.append(wait_for_user)
if is_root:
tools: list[Tool] = [*_BASE_TOOLS, *agent_tools, finish_scan]
else:

View file

@ -33,18 +33,19 @@ INTER-AGENT MESSAGES:
{% if interactive %}
INTERACTIVE BEHAVIOR:
- You are in an interactive conversation with a user.
- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — text is shown to the user, and then execution continues.
- To answer the user and hand control back, call respond_to_user. It delivers your message AND parks you for their reply in one call, so there is no way to answer and then forget to stop. This is the ONLY way to yield to the user.
- Everything you write as plain text is shown to the user, as you write it. Plain text is the ONLY way to talk to the user: there is no message tool and no message argument anywhere.
- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — it is shown, and then execution continues.
- To hand control to the user, call wait_for_user. It takes no arguments and says nothing; it only stops you until they reply. This is the ONLY way to yield to the user.
- To wait on another AGENT (a child's report, a peer's reply), call wait_for_agents. That is not a way to reach the user.
- To end the whole engagement, call the lifecycle tool: finish_scan (root) or agent_finish (subagent).
- A turn that ends with plain text and no tool call does NOT stop you: the system nudges you to continue and will re-run you. Do not rely on going silent to pause — it will not pause you.
- Answering a user question: put the answer in respond_to_user's message. Do not write the answer as plain text and then fall silent — that does not reach a stopping point, it just triggers a continuation nudge.
- If all you want to do is reply and stop, that whole turn is ONE respond_to_user call carrying the answer. Do not write the answer as text and then call respond_to_user as well: the user reads it twice.
- If you do end a turn on plain text and the nudge arrives, your words already reached the user. Do not restate them: call respond_to_user with NO message to simply wait, or with only whatever you still need to add.
- You may include brief explanatory text before a tool call, and you can narrate while you work — plain text is shown to the user as you go. Narrating is free; respond_to_user is specifically the act of WAITING for the user, so do not call it just to give a status update.
- Answering the user: write the answer as plain text, then call wait_for_user in the same turn. Never call wait_for_user before you have written your reply: the user would be handed a silent turn, and the system sends you back to write it.
- Never restate, summarize, or close out what you have already written, in text or in any tool argument. Once it is written, the user has read it; the only thing left to do is call wait_for_user.
- If you end a turn on plain text and the nudge arrives, your words already reached the user. Do not repeat them: call wait_for_user.
- You can narrate while you work — plain text is shown to the user as you go. Narrating is free; wait_for_user is specifically the act of WAITING for the user, so do not call it just to give a status update.
- Respond naturally when the user asks questions or gives instructions.
- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and respond_to_user only when you genuinely need the user.
- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, send it with respond_to_user.
- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and wait_for_user only when you genuinely need the user.
- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, write it and call wait_for_user.
{% else %}
AUTONOMOUS BEHAVIOR:
- Work autonomously by default

View file

@ -10,7 +10,7 @@ from collections.abc import Callable
from functools import cache
from typing import TYPE_CHECKING, Any, cast
from agents import RunConfig, Runner
from agents import ItemHelpers, MessageOutputItem, RunConfig, Runner
from agents.exceptions import AgentsException, MaxTurnsExceeded, UserError
from agents.sandbox.errors import ExecTransportError
from openai import (
@ -505,13 +505,18 @@ async def _run_until_lifecycle(
"""Drive an agent until an explicit lifecycle tool settles its status.
A turn that ends without ``finish_scan``, ``agent_finish``,
``respond_to_user``, or ``wait_for_agents`` leaves the agent ``running``:
``wait_for_user``, or ``wait_for_agents`` leaves the agent ``running``:
plain text never terminates a run and never yields to the user. Such a turn
is nudged back into a tool call, bounded by a recovery limit.
is nudged back into a tool call, bounded by a recovery limit. The same
budget covers a ``wait_for_user`` call made before anything was said to the
user since their last message: plain text is the only channel to them, so
that park would hand them a silent turn, and the agent is sent back to
write its reply instead.
"""
result: RunResultBase | None = None
input_data: Any = initial_input
recovery_limit = _INTERACTIVE_TOOL_RECOVERY_LIMIT if interactive else max(1, max_turns)
said_to_user = False
while True:
if coordinator.budget_stopped:
@ -568,21 +573,26 @@ async def _run_until_lifecycle(
input_data = []
continue
said_to_user = said_to_user or _said_to_user(result)
status = await _agent_status(coordinator, agent_id)
if status != "running":
silent_yield = (
interactive and not said_to_user and await _parked_for_user(coordinator, agent_id)
)
if status != "running" and not silent_yield:
await coordinator.reset_recovery(agent_id)
return result
recoveries = await coordinator.record_recovery(agent_id)
logger.warning(
"agent %s ended a turn without a lifecycle tool call (interactive=%s); "
"forcing tool continuation (%d/%d): %s",
_log_recovery(
agent_id,
interactive,
result,
recoveries,
recovery_limit,
_final_output_preview(result),
interactive=interactive,
silent_yield=silent_yield,
)
if silent_yield:
await coordinator.mark_running(agent_id)
if recoveries >= recovery_limit:
return await _exhausted_recovery(coordinator, agent_id, result, interactive=interactive)
@ -593,6 +603,7 @@ async def _run_until_lifecycle(
attempt=recoveries,
limit=recovery_limit,
interactive=interactive,
silent_yield=silent_yield,
)
@ -884,6 +895,51 @@ async def _agent_status(coordinator: AgentCoordinator, agent_id: str) -> Status
return coordinator.statuses.get(agent_id)
async def _parked_for_user(coordinator: AgentCoordinator, agent_id: str) -> bool:
async with coordinator._lock:
return (
coordinator.statuses.get(agent_id) == "waiting"
and coordinator.wait_kinds.get(agent_id) == "user"
)
def _log_recovery(
agent_id: str,
result: RunResultBase | None,
attempt: int,
limit: int,
*,
interactive: bool,
silent_yield: bool,
) -> None:
if silent_yield:
logger.warning(
"agent %s called wait_for_user without saying anything to the user; "
"sending it back to reply (%d/%d)",
agent_id,
attempt,
limit,
)
return
logger.warning(
"agent %s ended a turn without a lifecycle tool call (interactive=%s); "
"forcing tool continuation (%d/%d): %s",
agent_id,
interactive,
attempt,
limit,
_final_output_preview(result),
)
def _said_to_user(result: RunResultBase | None) -> bool:
"""Whether the run produced any assistant text, the only channel to the user."""
for item in getattr(result, "new_items", ()) or ():
if isinstance(item, MessageOutputItem) and ItemHelpers.text_message_output(item).strip():
return True
return False
def _final_output_preview(result: RunResultBase | None) -> str:
final_output = getattr(result, "final_output", None)
if final_output is None:
@ -901,15 +957,23 @@ async def _append_tool_required_message(
attempt: int,
limit: int,
interactive: bool,
silent_yield: bool = False,
) -> list[dict[str, str]]:
finish_tool = "finish_scan" if context.get("parent_id") is None else "agent_finish"
if interactive:
if silent_yield:
message = (
"You called wait_for_user without having written anything to the user since "
"their last message, so they would be handed a silent turn. Plain text is the "
"only channel to the user: write your reply as plain text now, then call "
f"wait_for_user. This is recovery attempt {attempt}/{limit}."
)
elif interactive:
message = (
"Your previous message ended a turn without a tool call. Plain text never ends "
"execution and never hands control to the user: it is shown to the user, and the "
"run continues. Continue immediately and call exactly one tool. "
"If you have something to tell the user and nothing to do until they reply, "
"call respond_to_user — with no message if you have already said it. "
"If you have nothing to do until the user replies, call wait_for_user; your "
"text already reached them, so do not repeat it. "
"If you are blocked waiting for another agent, call wait_for_agents. "
f"If the whole engagement is complete, call {finish_tool}. "
"Otherwise use the appropriate execution or planning tool. "

View file

@ -94,6 +94,8 @@ func Tool(data map[string]any) string {
return renderListReports(result)
case "get_report":
return renderGetReport(result)
case "wait_for_user":
return renderWaitForUser()
case "respond_to_user":
return renderRespondToUser(args)
case "finish_scan":

View file

@ -136,7 +136,12 @@ func TestToolDispatchCoversKnownTools(t *testing.T) {
[]string{"report read", "not found"},
},
{
"respond_to_user",
"wait_for_user",
tool("wait_for_user", map[string]any{}, nil, "completed"),
[]string{"waiting for your reply"},
},
{
"respond_to_user (recorded runs)",
tool("respond_to_user", map[string]any{"message": "Here is the answer"}, nil, "completed"),
[]string{"Here is the answer", "waiting for your reply"},
},
@ -320,6 +325,9 @@ func TestCollapseToolOnlyOutputHeavyTools(t *testing.T) {
if out, expandable := CollapseTool(full, "respond_to_user", false); expandable || out != full {
t.Fatal("respond_to_user must never collapse")
}
if out, expandable := CollapseTool(full, "wait_for_user", false); expandable || out != full {
t.Fatal("wait_for_user must never collapse")
}
}
func TestReportSectionsRenderMarkdown(t *testing.T) {

View file

@ -1,18 +0,0 @@
package render
import "strings"
// ---------------------------------------------------------------------------
// Direct replies (respond_renderer.py)
// ---------------------------------------------------------------------------
// renderRespondToUser shows the reply as the agent's own prose, since
// respond_to_user carries the message the user is meant to read.
func renderRespondToUser(args map[string]any) string {
var b strings.Builder
if message := StringValue(args["message"]); message != "" {
b.WriteString(renderAssistantMarkdown(message) + "\n\n")
}
b.WriteString(Col(Gray).Render("○ ") + Dim().Render("waiting for your reply"))
return b.String()
}

View file

@ -0,0 +1,25 @@
package render
import "strings"
// ---------------------------------------------------------------------------
// Handing the turn to the user
// ---------------------------------------------------------------------------
// renderWaitForUser marks the agent parked for the user's reply. The call
// carries no text: whatever the agent had to say was already shown as its
// own prose.
func renderWaitForUser() string {
return Col(Gray).Render("○ ") + Dim().Render("waiting for your reply")
}
// renderRespondToUser renders the retired respond_to_user call from recorded
// runs, whose message argument was the reply the user meant to read.
func renderRespondToUser(args map[string]any) string {
var b strings.Builder
if message := StringValue(args["message"]); message != "" {
b.WriteString(renderAssistantMarkdown(message) + "\n\n")
}
b.WriteString(renderWaitForUser())
return b.String()
}

View file

@ -1,20 +0,0 @@
"use client";
import type { ToolRendererProps } from "@/types/events";
import Markdown from "./Markdown";
/**
* `respond_to_user` carries the message the user is meant to read, so it renders
* as the agent's own prose rather than as a tool call.
*/
export default function RespondRenderer({ args }: ToolRendererProps) {
const message = (args.message as string) ?? "";
if (!message) return null;
return (
<div>
<Markdown text={message} />
<div className="mt-1.5 text-[#888] text-[13px]">waiting for your reply</div>
</div>
);
}

View file

@ -0,0 +1,22 @@
"use client";
import type { ToolRendererProps } from "@/types/events";
import Markdown from "./Markdown";
/**
* `wait_for_user` hands the turn to the user and carries no text: whatever the
* agent had to say was already shown as its own prose. Recorded runs still hold
* `respond_to_user` calls, whose `message` argument was the reply itself.
*/
export default function WaitForUserRenderer({ args }: ToolRendererProps) {
const message = typeof args.message === "string" ? args.message : "";
return (
<div>
{message && <Markdown text={message} />}
<div className={message ? "mt-1.5 text-[#888] text-[13px]" : "text-[#888] text-[13px]"}>
waiting for your reply
</div>
</div>
);
}

View file

@ -25,7 +25,7 @@ import NotesRenderer from "./NotesRenderer";
import TodoRenderer from "./TodoRenderer";
import FallbackRenderer from "./FallbackRenderer";
import LoadSkillRenderer from "./LoadSkillRenderer";
import RespondRenderer from "./RespondRenderer";
import WaitForUserRenderer from "./WaitForUserRenderer";
import CoverageRenderer from "./CoverageRenderer";
import ThreatModelRenderer from "./ThreatModelRenderer";
import McpRenderer from "./McpRenderer";
@ -122,7 +122,7 @@ const CATEGORY_TOOLS: Record<ToolCategory, readonly string[]> = {
agents: ["create_agent", "agent_finish", "send_message_to_agent", "wait_for_agents", "view_agent_graph", "stop_agent"],
search: ["web_search"],
// scan_start_info / subagent_start_info are strix-app synthetic events; finish_scan is the engine's
lifecycle: ["scan_start_info", "subagent_start_info", "finish_scan", "respond_to_user"],
lifecycle: ["scan_start_info", "subagent_start_info", "finish_scan", "wait_for_user", "respond_to_user"],
notes: ["create_note", "delete_note", "update_note", "list_notes", "get_note"],
skills: ["load_skill"],
todos: ["create_todo", "list_todos", "update_todo", "mark_todo_done", "mark_todo_pending", "delete_todo"],
@ -147,7 +147,8 @@ const TOOL_CATEGORY: Record<string, ToolCategory> = Object.fromEntries(
*/
const RENDERER_OVERRIDES: Partial<Record<string, ComponentType<ToolRendererProps>>> = {
finish_scan: FinishRenderer,
respond_to_user: RespondRenderer,
wait_for_user: WaitForUserRenderer,
respond_to_user: WaitForUserRenderer,
apply_patch: ApplyPatchRenderer,
view_image: ViewImageRenderer,
list_reports: ReportListRenderer,
@ -163,6 +164,7 @@ const ICON_OVERRIDES: Partial<Record<string, ToolIconMeta>> = {
agent_finish: { icon: Flag, color: "text-cyan-400" },
send_message_to_agent: { icon: MessageCircle, color: "text-cyan-400" },
wait_for_agents: { icon: MessageCircle, color: "text-cyan-400" },
wait_for_user: { icon: MessageCircle, color: "text-emerald-400" },
respond_to_user: { icon: MessageCircle, color: "text-emerald-400" },
view_agent_graph: { icon: Eye, color: "text-cyan-400" },
stop_agent: { icon: Ban, color: "text-red-400" },

View file

@ -6,7 +6,7 @@
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<meta name="color-scheme" content="dark" />
<title>Strix Results</title>
<script type="module" crossorigin src="./assets/index-DbPDzbcJ.js"></script>
<script type="module" crossorigin src="./assets/index-B1JzV4Vu.js"></script>
<link rel="stylesheet" crossorigin href="./assets/index-CrLE9rrV.css">
</head>
<body>

View file

@ -311,8 +311,8 @@ async def wait_for_agents( # noqa: PLR0911
**This tool is only for waiting on other agents.** Two things it is
NOT for:
- **Talking to the user.** Use ``respond_to_user``, which delivers
your message and hands control back in one call.
- **Waiting for the user.** Write your reply as plain text, then call
``wait_for_user`` to hand control back.
- **Waiting for a long-running command.** This tool does not watch
processes at all — it sleeps until a *message* arrives, so it
burns the full timeout even if your command finished a second

View file

@ -1,6 +0,0 @@
"""User-facing reply tool for interactive sessions."""
from strix.tools.respond.tool import respond_to_user
__all__ = ["respond_to_user"]

View file

@ -0,0 +1,6 @@
"""Hand-the-turn-to-the-user tool for interactive sessions."""
from strix.tools.wait_for_user.tool import wait_for_user
__all__ = ["wait_for_user"]

View file

@ -1,4 +1,4 @@
"""``respond_to_user`` — deliver a reply and hand control back to the user."""
"""``wait_for_user`` — hand control back to the user and wait for their reply."""
from __future__ import annotations
@ -15,23 +15,23 @@ def _ctx(ctx: RunContextWrapper) -> dict[str, Any]:
@function_tool
async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
"""Answer the user and hand control back to them.
async def wait_for_user(ctx: RunContextWrapper) -> str:
"""Hand control to the user and wait for their reply.
This is the ONLY way to yield to the user. Delivering the message and
yielding are the same call on purpose: there is no way to answer and
then forget to stop, and no way to stop without having answered.
This is the ONLY way to yield to the user. It carries no text: everything
you write as plain text is already shown to the user, so write your reply
first, as plain text, then call this to stop and wait. Never restate in
any form what you have already written.
Call it when you have something for the user and nothing to do until
they reply — you answered their question, you need a decision or a
credential only they can give, or you finished a chunk of work and
want direction. You resume exactly where you left off when they
reply, with everything you have done so far intact.
Call it when you have nothing to do until the user replies — you answered
their question, you need a decision or a credential only they can give,
or you finished a chunk of work and want direction. You resume exactly
where you left off when they reply, with everything you have done so far
intact.
Do NOT call it to narrate progress or to think out loud. Plain text
is still shown to the user as you work, so say whatever you like
mid-task without stopping; ``respond_to_user`` is specifically the
act of *waiting* for them. Every call costs the user their attention.
Do NOT call it to narrate progress or to think out loud: plain text is
shown to the user as you work, so say whatever you like mid-task without
stopping. Every call costs the user their attention.
Not for these:
@ -39,16 +39,6 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
use ``wait_for_agents``.
- **Ending the engagement** — use ``finish_scan`` (root) or
``agent_finish`` (subagent). Those are terminal; this is a pause.
Args:
message: What to say to the user. Self-contained: they may not
have followed the tool calls that led here. Lead with the
answer or the decision you need, and if you are blocked, say
exactly what you need from them.
Omit it when you have just said your piece as plain text and
only need to wait: that text has already reached them, and
repeating it makes them read the same answer twice.
"""
inner = _ctx(ctx)
coordinator = coordinator_from_context(inner)
@ -79,7 +69,7 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
stopped = coordinator.statuses.get(me) == "stopped"
if stopped:
return json.dumps(
{"success": True, "wait_outcome": "stopped", "message": message},
{"success": True, "wait_outcome": "stopped"},
ensure_ascii=False,
default=str,
)
@ -94,8 +84,7 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
"success": True,
"wait_outcome": "message_arrived",
"pending_messages": pending,
"message": message,
"note": "Your reply was delivered; the user had already sent a new message.",
"note": "The user had already sent a new message; keep going.",
},
ensure_ascii=False,
default=str,
@ -106,8 +95,7 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
{
"success": True,
"wait_outcome": "waiting",
"message": message,
"note": "Reply delivered; parked until the user responds.",
"note": "Parked until the user responds.",
},
ensure_ascii=False,
default=str,

View file

@ -99,13 +99,13 @@ def test_no_override_renders_builtin_prompt() -> None:
assert agent.instructions != ""
def test_respond_to_user_is_interactive_only() -> None:
def test_wait_for_user_is_interactive_only() -> None:
"""Yielding to the user is meaningless when no user is attached."""
interactive = factory.build_strix_agent(is_root=True, interactive=True)
autonomous = factory.build_strix_agent(is_root=True, interactive=False)
assert "respond_to_user" in [t.name for t in interactive.tools]
assert "respond_to_user" not in [t.name for t in autonomous.tools]
assert "wait_for_user" in [t.name for t in interactive.tools]
assert "wait_for_user" not in [t.name for t in autonomous.tools]
def test_wait_for_agents_is_available_in_both_modes() -> None:

View file

@ -13,10 +13,10 @@ from agents.exceptions import MaxTurnsExceeded
from agents.items import MessageOutputItem
from agents.memory import SQLiteSession
from agents.tool_context import ToolContext
from openai.types.responses import ResponseOutputMessage, ResponseOutputRefusal
from openai.types.responses import ResponseOutputMessage, ResponseOutputRefusal, ResponseOutputText
from strix.core import execution
from strix.core.agents import AgentCoordinator
from strix.core.agents import AgentCoordinator, WaitKind
from strix.core.execution import (
_notify_root_on_budget_reserve,
notify_parent_on_terminal,
@ -1030,7 +1030,7 @@ async def test_interactive_text_only_turn_is_nudged_instead_of_parking(
# The retry carries an explicit "call a tool" nudge rather than empty input.
nudge = calls[1][0]["content"]
assert "without a tool call" in nudge
assert "respond_to_user" in nudge
assert "wait_for_user" in nudge
assert coordinator.statuses["root"] == "completed"
@ -1038,7 +1038,7 @@ async def test_interactive_text_only_turn_is_nudged_instead_of_parking(
async def test_interactive_explicit_park_gets_no_nudge(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""``waiting`` is only reachable via respond_to_user / wait_for_agents."""
"""``waiting`` is only reachable via wait_for_user / wait_for_agents."""
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
@ -1161,7 +1161,7 @@ async def test_tool_required_message_is_persisted_to_the_session(tmp_path: Any)
stored = [cast("dict[str, Any]", i) for i in await session.get_items()]
assert "finish_scan" in stored[0]["content"]
assert "respond_to_user" in stored[0]["content"]
assert "wait_for_user" in stored[0]["content"]
session.close()
@ -1277,13 +1277,9 @@ async def test_wait_kind_survives_a_snapshot_round_trip() -> None:
async def test_interactive_nudge_offers_waiting_without_repeating() -> None:
"""The nudge is the instruction an agent reads when it is stranded here.
It is where the option to wait on what was already said has to be, not only
in the system prompt: an agent that ended a turn on plain text reasons off
this text, and without the clause it restates its answer to reach a tool
call, so the user reads it twice.
The clause holds whatever the turn did, because the agent is the one who
knows whether it spoke — this fires for a turn that produced no text at all.
An agent that ended a turn on plain text reasons off this text, and without
the clause it restates its answer to reach a tool call, so the user reads it
twice.
"""
items = await execution._append_tool_required_message(
session=None,
@ -1293,7 +1289,192 @@ async def test_interactive_nudge_offers_waiting_without_repeating() -> None:
interactive=True,
)
assert "with no message if you have already said it" in items[0]["content"]
assert "call wait_for_user" in items[0]["content"]
assert "do not repeat it" in items[0]["content"]
def _cycle_with_items(
coordinator: AgentCoordinator,
agent_id: str,
script: list[tuple[str, WaitKind | None, list[Any]]],
calls: list[Any],
) -> Any:
"""Fake run cycle scripted as (status, wait_kind, new_items) per call."""
async def _cycle(*_args: Any, **kwargs: Any) -> Any:
calls.append(kwargs.get("input_data"))
status, wait_kind, new_items = script[min(len(calls) - 1, len(script) - 1)]
if wait_kind is not None:
await coordinator.park_waiting(agent_id, wait_kind=wait_kind)
else:
await coordinator.set_status(agent_id, status)
return MagicMock(final_output="", new_items=new_items)
return _cycle
def _text_item(text: str) -> MessageOutputItem:
return MessageOutputItem(
agent=MagicMock(),
raw_item=ResponseOutputMessage(
id="msg-1",
content=[ResponseOutputText(type="output_text", text=text, annotations=[])],
role="assistant",
status="completed",
type="message",
),
)
@pytest.mark.asyncio
async def test_silent_wait_for_user_is_sent_back_to_reply(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""``wait_for_user`` carries no text, so a park with nothing said is a silent turn.
The driver undoes the park and nudges the agent to write its reply; the
second cycle, which does say something, parks for real.
"""
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
monkeypatch.setattr(
execution,
"_run_cycle_parked",
_cycle_with_items(
coordinator,
"root",
[("waiting", "user", []), ("waiting", "user", [_text_item("Here is the answer.")])],
calls,
),
)
await _drive(coordinator, "root", interactive=True)
assert len(calls) == 2
nudge = calls[1][0]["content"]
assert "without having written anything" in nudge
assert "wait_for_user" in nudge
assert coordinator.statuses["root"] == "waiting"
assert coordinator.wait_kinds["root"] == "user"
@pytest.mark.asyncio
async def test_wait_for_user_after_text_parks_without_a_nudge(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Text written in the same turn as the call is the reply; the park stands."""
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
monkeypatch.setattr(
execution,
"_run_cycle_parked",
_cycle_with_items(
coordinator, "root", [("waiting", "user", [_text_item("Done, see above.")])], calls
),
)
await _drive(coordinator, "root", interactive=True)
assert len(calls) == 1
assert coordinator.statuses["root"] == "waiting"
@pytest.mark.asyncio
async def test_text_from_an_earlier_nudged_turn_counts_as_the_reply(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A text-only turn is nudged; the bare ``wait_for_user`` that follows is right.
The words already reached the user in the first cycle, so the second must
not be bounced back for repeating them.
"""
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
monkeypatch.setattr(
execution,
"_run_cycle_parked",
_cycle_with_items(
coordinator,
"root",
[
("running", None, [_text_item("The scan found two issues.")]),
("waiting", "user", []),
],
calls,
),
)
await _drive(coordinator, "root", interactive=True)
assert len(calls) == 2
assert coordinator.statuses["root"] == "waiting"
@pytest.mark.asyncio
async def test_whitespace_only_text_does_not_count_as_a_reply(
monkeypatch: pytest.MonkeyPatch,
) -> None:
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
monkeypatch.setattr(
execution,
"_run_cycle_parked",
_cycle_with_items(
coordinator,
"root",
[("waiting", "user", [_text_item(" \n")]), ("waiting", "user", [_text_item("ok")])],
calls,
),
)
await _drive(coordinator, "root", interactive=True)
assert len(calls) == 2
@pytest.mark.asyncio
async def test_a_wait_on_agents_is_never_a_silent_yield(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Only the human wait needs words first; waiting on a child says nothing by design."""
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
monkeypatch.setattr(
execution,
"_run_cycle_parked",
_cycle_with_items(coordinator, "root", [("waiting", "agents", [])], calls),
)
await _drive(coordinator, "root", interactive=True)
assert len(calls) == 1
assert coordinator.statuses["root"] == "waiting"
@pytest.mark.asyncio
async def test_repeated_silent_yields_exhaust_the_recovery_budget(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""An agent that never writes anything is parked as stalled, not looped forever."""
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
calls: list[Any] = []
monkeypatch.setattr(
execution,
"_run_cycle_parked",
_cycle_with_items(coordinator, "root", [("waiting", "user", [])], calls),
)
await _drive(coordinator, "root", interactive=True)
assert len(calls) == execution._INTERACTIVE_TOOL_RECOVERY_LIMIT
assert coordinator.statuses["root"] == "waiting"
assert coordinator.wait_kinds["root"] == "stalled"
@pytest.mark.asyncio
@ -1307,4 +1488,4 @@ async def test_autonomous_nudge_does_not_offer_the_user() -> None:
interactive=False,
)
assert "respond_to_user" not in items[0]["content"]
assert "wait_for_user" not in items[0]["content"]

View file

@ -16,12 +16,12 @@ from strix.agents import factory
from strix.tools.agents_graph.tools import agent_finish
from strix.tools.finish.tool import finish_scan
from strix.tools.reporting.tool import create_vulnerability_report
from strix.tools.respond.tool import respond_to_user
from strix.tools.wait_for_user.tool import wait_for_user
_SCAN_AGENT_TOOLS = [
tool
for tool in (*factory._BASE_TOOLS, finish_scan, agent_finish, respond_to_user)
for tool in (*factory._BASE_TOOLS, finish_scan, agent_finish, wait_for_user)
if isinstance(tool, FunctionTool)
]

View file

@ -1,4 +1,4 @@
"""Tests for the ``respond_to_user`` yield tool."""
"""Tests for the ``wait_for_user`` yield tool."""
from __future__ import annotations
@ -9,17 +9,17 @@ import pytest
from agents.tool_context import ToolContext
from strix.core.agents import AgentCoordinator
from strix.tools.respond.tool import respond_to_user
from strix.tools.wait_for_user.tool import wait_for_user
async def _call(context: dict[str, Any], message: str = "here is what I found") -> dict[str, Any]:
async def _call(context: dict[str, Any]) -> dict[str, Any]:
ctx = ToolContext(
context=context,
tool_name="respond_to_user",
tool_name="wait_for_user",
tool_call_id="call-1",
tool_arguments="{}",
)
raw = await respond_to_user.on_invoke_tool(ctx, json.dumps({"message": message}))
raw = await wait_for_user.on_invoke_tool(ctx, "{}")
return json.loads(raw) # type: ignore[no-any-return]
@ -29,15 +29,24 @@ async def _context(*, interactive: bool, agent_id: str = "root") -> dict[str, An
return {"coordinator": coordinator, "agent_id": agent_id, "interactive": interactive}
def test_takes_no_arguments() -> None:
"""Plain text is the only channel to the user, so the tool carries none.
A message parameter here is a second channel the model fills with the same
words it already wrote, and the user reads its answer twice.
"""
assert wait_for_user.params_json_schema.get("properties", {}) == {}
@pytest.mark.asyncio
async def test_parks_the_agent_and_carries_the_message() -> None:
async def test_parks_the_agent_for_the_user() -> None:
context = await _context(interactive=True)
result = await _call(context)
coordinator = context["coordinator"]
assert result["success"] is True
assert result["wait_outcome"] == "waiting"
assert result["message"] == "here is what I found"
assert "message" not in result
assert coordinator.statuses["root"] == "waiting"
# Recorded as a human wait, so the driver never auto-resumes it.
assert coordinator.wait_kinds["root"] == "user"
@ -66,29 +75,13 @@ async def test_a_message_that_already_arrived_is_taken_instead_of_parking() -> N
assert coordinator.statuses["root"] == "running"
async def _call_without_message(context: dict[str, Any]) -> dict[str, Any]:
ctx = ToolContext(
context=context,
tool_name="respond_to_user",
tool_call_id="call-1",
tool_arguments="{}",
)
raw = await respond_to_user.on_invoke_tool(ctx, "{}")
return json.loads(raw) # type: ignore[no-any-return]
@pytest.mark.asyncio
async def test_parks_without_a_message() -> None:
"""An agent that has already said its piece as plain text can just wait.
The nudge is what leaves it here, and while a message was required the only
way to stop was to send the same answer a second time.
"""
async def test_a_stopped_agent_is_left_stopped() -> None:
context = await _context(interactive=True)
coordinator = context["coordinator"]
await coordinator.set_status("root", "stopped")
result = await _call_without_message(context)
result = await _call(context)
assert result["success"] is True
assert result["wait_outcome"] == "waiting"
assert result["message"] == ""
assert context["coordinator"].statuses["root"] == "waiting"
assert result["wait_outcome"] == "stopped"
assert coordinator.statuses["root"] == "stopped"