mirror of
https://github.com/usestrix/strix.git
synced 2026-10-05 02:41:38 +00:00
refactor(agents): replace respond_to_user with a text-free wait_for_user
Plain text is now the only channel to the user. wait_for_user takes no arguments and only parks the agent, so a reply can no longer be written twice (once as text, once as a tool argument). The driver bounces a wait_for_user call that follows no assistant text back into recovery. TUI and viewer render wait_for_user as the waiting marker alone and keep rendering recorded respond_to_user events with their message.
This commit is contained in:
parent
6b7e3b77e8
commit
16a316540c
20 changed files with 525 additions and 277 deletions
|
|
@ -61,7 +61,6 @@ from strix.tools.reporting.tool import (
|
|||
list_reports,
|
||||
update_vulnerability_report,
|
||||
)
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
from strix.tools.thinking.tool import think
|
||||
from strix.tools.threat_model.tools import (
|
||||
amend_threat_model,
|
||||
|
|
@ -76,6 +75,7 @@ from strix.tools.todo.tools import (
|
|||
mark_todo_pending,
|
||||
update_todo,
|
||||
)
|
||||
from strix.tools.wait_for_user.tool import wait_for_user
|
||||
from strix.tools.web_search.tool import web_get_contents, web_search
|
||||
|
||||
|
||||
|
|
@ -509,7 +509,7 @@ def _make_shell_configurator(*, chat_completions: bool, strict_schemas: bool) ->
|
|||
|
||||
|
||||
# Tools that hand control away by parking the agent rather than ending the scan.
|
||||
_PARKING_TOOLS: frozenset[str] = frozenset({"respond_to_user", "wait_for_agents"})
|
||||
_PARKING_TOOLS: frozenset[str] = frozenset({"wait_for_user", "wait_for_agents"})
|
||||
|
||||
|
||||
def _lifecycle_tool_completed(tool_name: str, output: Any) -> bool:
|
||||
|
|
@ -697,7 +697,7 @@ def build_strix_agent(
|
|||
agent_tools = [*_EXTRA_TOOLS, *(extra_tools or [])]
|
||||
if interactive:
|
||||
# Yielding to the user is only meaningful when one is attached.
|
||||
agent_tools.append(respond_to_user)
|
||||
agent_tools.append(wait_for_user)
|
||||
if is_root:
|
||||
tools: list[Tool] = [*_BASE_TOOLS, *agent_tools, finish_scan]
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -33,18 +33,19 @@ INTER-AGENT MESSAGES:
|
|||
{% if interactive %}
|
||||
INTERACTIVE BEHAVIOR:
|
||||
- You are in an interactive conversation with a user.
|
||||
- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — text is shown to the user, and then execution continues.
|
||||
- To answer the user and hand control back, call respond_to_user. It delivers your message AND parks you for their reply in one call, so there is no way to answer and then forget to stop. This is the ONLY way to yield to the user.
|
||||
- Everything you write as plain text is shown to the user, as you write it. Plain text is the ONLY way to talk to the user: there is no message tool and no message argument anywhere.
|
||||
- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — it is shown, and then execution continues.
|
||||
- To hand control to the user, call wait_for_user. It takes no arguments and says nothing; it only stops you until they reply. This is the ONLY way to yield to the user.
|
||||
- To wait on another AGENT (a child's report, a peer's reply), call wait_for_agents. That is not a way to reach the user.
|
||||
- To end the whole engagement, call the lifecycle tool: finish_scan (root) or agent_finish (subagent).
|
||||
- A turn that ends with plain text and no tool call does NOT stop you: the system nudges you to continue and will re-run you. Do not rely on going silent to pause — it will not pause you.
|
||||
- Answering a user question: put the answer in respond_to_user's message. Do not write the answer as plain text and then fall silent — that does not reach a stopping point, it just triggers a continuation nudge.
|
||||
- If all you want to do is reply and stop, that whole turn is ONE respond_to_user call carrying the answer. Do not write the answer as text and then call respond_to_user as well: the user reads it twice.
|
||||
- If you do end a turn on plain text and the nudge arrives, your words already reached the user. Do not restate them: call respond_to_user with NO message to simply wait, or with only whatever you still need to add.
|
||||
- You may include brief explanatory text before a tool call, and you can narrate while you work — plain text is shown to the user as you go. Narrating is free; respond_to_user is specifically the act of WAITING for the user, so do not call it just to give a status update.
|
||||
- Answering the user: write the answer as plain text, then call wait_for_user in the same turn. Never call wait_for_user before you have written your reply: the user would be handed a silent turn, and the system sends you back to write it.
|
||||
- Never restate, summarize, or close out what you have already written, in text or in any tool argument. Once it is written, the user has read it; the only thing left to do is call wait_for_user.
|
||||
- If you end a turn on plain text and the nudge arrives, your words already reached the user. Do not repeat them: call wait_for_user.
|
||||
- You can narrate while you work — plain text is shown to the user as you go. Narrating is free; wait_for_user is specifically the act of WAITING for the user, so do not call it just to give a status update.
|
||||
- Respond naturally when the user asks questions or gives instructions.
|
||||
- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and respond_to_user only when you genuinely need the user.
|
||||
- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, send it with respond_to_user.
|
||||
- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and wait_for_user only when you genuinely need the user.
|
||||
- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, write it and call wait_for_user.
|
||||
{% else %}
|
||||
AUTONOMOUS BEHAVIOR:
|
||||
- Work autonomously by default
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ from collections.abc import Callable
|
|||
from functools import cache
|
||||
from typing import TYPE_CHECKING, Any, cast
|
||||
|
||||
from agents import RunConfig, Runner
|
||||
from agents import ItemHelpers, MessageOutputItem, RunConfig, Runner
|
||||
from agents.exceptions import AgentsException, MaxTurnsExceeded, UserError
|
||||
from agents.sandbox.errors import ExecTransportError
|
||||
from openai import (
|
||||
|
|
@ -505,13 +505,18 @@ async def _run_until_lifecycle(
|
|||
"""Drive an agent until an explicit lifecycle tool settles its status.
|
||||
|
||||
A turn that ends without ``finish_scan``, ``agent_finish``,
|
||||
``respond_to_user``, or ``wait_for_agents`` leaves the agent ``running``:
|
||||
``wait_for_user``, or ``wait_for_agents`` leaves the agent ``running``:
|
||||
plain text never terminates a run and never yields to the user. Such a turn
|
||||
is nudged back into a tool call, bounded by a recovery limit.
|
||||
is nudged back into a tool call, bounded by a recovery limit. The same
|
||||
budget covers a ``wait_for_user`` call made before anything was said to the
|
||||
user since their last message: plain text is the only channel to them, so
|
||||
that park would hand them a silent turn, and the agent is sent back to
|
||||
write its reply instead.
|
||||
"""
|
||||
result: RunResultBase | None = None
|
||||
input_data: Any = initial_input
|
||||
recovery_limit = _INTERACTIVE_TOOL_RECOVERY_LIMIT if interactive else max(1, max_turns)
|
||||
said_to_user = False
|
||||
|
||||
while True:
|
||||
if coordinator.budget_stopped:
|
||||
|
|
@ -568,21 +573,26 @@ async def _run_until_lifecycle(
|
|||
input_data = []
|
||||
continue
|
||||
|
||||
said_to_user = said_to_user or _said_to_user(result)
|
||||
status = await _agent_status(coordinator, agent_id)
|
||||
if status != "running":
|
||||
silent_yield = (
|
||||
interactive and not said_to_user and await _parked_for_user(coordinator, agent_id)
|
||||
)
|
||||
if status != "running" and not silent_yield:
|
||||
await coordinator.reset_recovery(agent_id)
|
||||
return result
|
||||
|
||||
recoveries = await coordinator.record_recovery(agent_id)
|
||||
logger.warning(
|
||||
"agent %s ended a turn without a lifecycle tool call (interactive=%s); "
|
||||
"forcing tool continuation (%d/%d): %s",
|
||||
_log_recovery(
|
||||
agent_id,
|
||||
interactive,
|
||||
result,
|
||||
recoveries,
|
||||
recovery_limit,
|
||||
_final_output_preview(result),
|
||||
interactive=interactive,
|
||||
silent_yield=silent_yield,
|
||||
)
|
||||
if silent_yield:
|
||||
await coordinator.mark_running(agent_id)
|
||||
|
||||
if recoveries >= recovery_limit:
|
||||
return await _exhausted_recovery(coordinator, agent_id, result, interactive=interactive)
|
||||
|
|
@ -593,6 +603,7 @@ async def _run_until_lifecycle(
|
|||
attempt=recoveries,
|
||||
limit=recovery_limit,
|
||||
interactive=interactive,
|
||||
silent_yield=silent_yield,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -884,6 +895,51 @@ async def _agent_status(coordinator: AgentCoordinator, agent_id: str) -> Status
|
|||
return coordinator.statuses.get(agent_id)
|
||||
|
||||
|
||||
async def _parked_for_user(coordinator: AgentCoordinator, agent_id: str) -> bool:
|
||||
async with coordinator._lock:
|
||||
return (
|
||||
coordinator.statuses.get(agent_id) == "waiting"
|
||||
and coordinator.wait_kinds.get(agent_id) == "user"
|
||||
)
|
||||
|
||||
|
||||
def _log_recovery(
|
||||
agent_id: str,
|
||||
result: RunResultBase | None,
|
||||
attempt: int,
|
||||
limit: int,
|
||||
*,
|
||||
interactive: bool,
|
||||
silent_yield: bool,
|
||||
) -> None:
|
||||
if silent_yield:
|
||||
logger.warning(
|
||||
"agent %s called wait_for_user without saying anything to the user; "
|
||||
"sending it back to reply (%d/%d)",
|
||||
agent_id,
|
||||
attempt,
|
||||
limit,
|
||||
)
|
||||
return
|
||||
logger.warning(
|
||||
"agent %s ended a turn without a lifecycle tool call (interactive=%s); "
|
||||
"forcing tool continuation (%d/%d): %s",
|
||||
agent_id,
|
||||
interactive,
|
||||
attempt,
|
||||
limit,
|
||||
_final_output_preview(result),
|
||||
)
|
||||
|
||||
|
||||
def _said_to_user(result: RunResultBase | None) -> bool:
|
||||
"""Whether the run produced any assistant text, the only channel to the user."""
|
||||
for item in getattr(result, "new_items", ()) or ():
|
||||
if isinstance(item, MessageOutputItem) and ItemHelpers.text_message_output(item).strip():
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _final_output_preview(result: RunResultBase | None) -> str:
|
||||
final_output = getattr(result, "final_output", None)
|
||||
if final_output is None:
|
||||
|
|
@ -901,15 +957,23 @@ async def _append_tool_required_message(
|
|||
attempt: int,
|
||||
limit: int,
|
||||
interactive: bool,
|
||||
silent_yield: bool = False,
|
||||
) -> list[dict[str, str]]:
|
||||
finish_tool = "finish_scan" if context.get("parent_id") is None else "agent_finish"
|
||||
if interactive:
|
||||
if silent_yield:
|
||||
message = (
|
||||
"You called wait_for_user without having written anything to the user since "
|
||||
"their last message, so they would be handed a silent turn. Plain text is the "
|
||||
"only channel to the user: write your reply as plain text now, then call "
|
||||
f"wait_for_user. This is recovery attempt {attempt}/{limit}."
|
||||
)
|
||||
elif interactive:
|
||||
message = (
|
||||
"Your previous message ended a turn without a tool call. Plain text never ends "
|
||||
"execution and never hands control to the user: it is shown to the user, and the "
|
||||
"run continues. Continue immediately and call exactly one tool. "
|
||||
"If you have something to tell the user and nothing to do until they reply, "
|
||||
"call respond_to_user — with no message if you have already said it. "
|
||||
"If you have nothing to do until the user replies, call wait_for_user; your "
|
||||
"text already reached them, so do not repeat it. "
|
||||
"If you are blocked waiting for another agent, call wait_for_agents. "
|
||||
f"If the whole engagement is complete, call {finish_tool}. "
|
||||
"Otherwise use the appropriate execution or planning tool. "
|
||||
|
|
|
|||
|
|
@ -94,6 +94,8 @@ func Tool(data map[string]any) string {
|
|||
return renderListReports(result)
|
||||
case "get_report":
|
||||
return renderGetReport(result)
|
||||
case "wait_for_user":
|
||||
return renderWaitForUser()
|
||||
case "respond_to_user":
|
||||
return renderRespondToUser(args)
|
||||
case "finish_scan":
|
||||
|
|
|
|||
|
|
@ -136,7 +136,12 @@ func TestToolDispatchCoversKnownTools(t *testing.T) {
|
|||
[]string{"report read", "not found"},
|
||||
},
|
||||
{
|
||||
"respond_to_user",
|
||||
"wait_for_user",
|
||||
tool("wait_for_user", map[string]any{}, nil, "completed"),
|
||||
[]string{"waiting for your reply"},
|
||||
},
|
||||
{
|
||||
"respond_to_user (recorded runs)",
|
||||
tool("respond_to_user", map[string]any{"message": "Here is the answer"}, nil, "completed"),
|
||||
[]string{"Here is the answer", "waiting for your reply"},
|
||||
},
|
||||
|
|
@ -320,6 +325,9 @@ func TestCollapseToolOnlyOutputHeavyTools(t *testing.T) {
|
|||
if out, expandable := CollapseTool(full, "respond_to_user", false); expandable || out != full {
|
||||
t.Fatal("respond_to_user must never collapse")
|
||||
}
|
||||
if out, expandable := CollapseTool(full, "wait_for_user", false); expandable || out != full {
|
||||
t.Fatal("wait_for_user must never collapse")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReportSectionsRenderMarkdown(t *testing.T) {
|
||||
|
|
|
|||
|
|
@ -1,18 +0,0 @@
|
|||
package render
|
||||
|
||||
import "strings"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Direct replies (respond_renderer.py)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// renderRespondToUser shows the reply as the agent's own prose, since
|
||||
// respond_to_user carries the message the user is meant to read.
|
||||
func renderRespondToUser(args map[string]any) string {
|
||||
var b strings.Builder
|
||||
if message := StringValue(args["message"]); message != "" {
|
||||
b.WriteString(renderAssistantMarkdown(message) + "\n\n")
|
||||
}
|
||||
b.WriteString(Col(Gray).Render("○ ") + Dim().Render("waiting for your reply"))
|
||||
return b.String()
|
||||
}
|
||||
25
strix/interface/tui/internal/render/wait_user.go
Normal file
25
strix/interface/tui/internal/render/wait_user.go
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
package render
|
||||
|
||||
import "strings"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Handing the turn to the user
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// renderWaitForUser marks the agent parked for the user's reply. The call
|
||||
// carries no text: whatever the agent had to say was already shown as its
|
||||
// own prose.
|
||||
func renderWaitForUser() string {
|
||||
return Col(Gray).Render("○ ") + Dim().Render("waiting for your reply")
|
||||
}
|
||||
|
||||
// renderRespondToUser renders the retired respond_to_user call from recorded
|
||||
// runs, whose message argument was the reply the user meant to read.
|
||||
func renderRespondToUser(args map[string]any) string {
|
||||
var b strings.Builder
|
||||
if message := StringValue(args["message"]); message != "" {
|
||||
b.WriteString(renderAssistantMarkdown(message) + "\n\n")
|
||||
}
|
||||
b.WriteString(renderWaitForUser())
|
||||
return b.String()
|
||||
}
|
||||
|
|
@ -1,20 +0,0 @@
|
|||
"use client";
|
||||
|
||||
import type { ToolRendererProps } from "@/types/events";
|
||||
import Markdown from "./Markdown";
|
||||
|
||||
/**
|
||||
* `respond_to_user` carries the message the user is meant to read, so it renders
|
||||
* as the agent's own prose rather than as a tool call.
|
||||
*/
|
||||
export default function RespondRenderer({ args }: ToolRendererProps) {
|
||||
const message = (args.message as string) ?? "";
|
||||
if (!message) return null;
|
||||
|
||||
return (
|
||||
<div>
|
||||
<Markdown text={message} />
|
||||
<div className="mt-1.5 text-[#888] text-[13px]">waiting for your reply</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
|
@ -0,0 +1,22 @@
|
|||
"use client";
|
||||
|
||||
import type { ToolRendererProps } from "@/types/events";
|
||||
import Markdown from "./Markdown";
|
||||
|
||||
/**
|
||||
* `wait_for_user` hands the turn to the user and carries no text: whatever the
|
||||
* agent had to say was already shown as its own prose. Recorded runs still hold
|
||||
* `respond_to_user` calls, whose `message` argument was the reply itself.
|
||||
*/
|
||||
export default function WaitForUserRenderer({ args }: ToolRendererProps) {
|
||||
const message = typeof args.message === "string" ? args.message : "";
|
||||
|
||||
return (
|
||||
<div>
|
||||
{message && <Markdown text={message} />}
|
||||
<div className={message ? "mt-1.5 text-[#888] text-[13px]" : "text-[#888] text-[13px]"}>
|
||||
waiting for your reply
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
|
@ -25,7 +25,7 @@ import NotesRenderer from "./NotesRenderer";
|
|||
import TodoRenderer from "./TodoRenderer";
|
||||
import FallbackRenderer from "./FallbackRenderer";
|
||||
import LoadSkillRenderer from "./LoadSkillRenderer";
|
||||
import RespondRenderer from "./RespondRenderer";
|
||||
import WaitForUserRenderer from "./WaitForUserRenderer";
|
||||
import CoverageRenderer from "./CoverageRenderer";
|
||||
import ThreatModelRenderer from "./ThreatModelRenderer";
|
||||
import McpRenderer from "./McpRenderer";
|
||||
|
|
@ -122,7 +122,7 @@ const CATEGORY_TOOLS: Record<ToolCategory, readonly string[]> = {
|
|||
agents: ["create_agent", "agent_finish", "send_message_to_agent", "wait_for_agents", "view_agent_graph", "stop_agent"],
|
||||
search: ["web_search"],
|
||||
// scan_start_info / subagent_start_info are strix-app synthetic events; finish_scan is the engine's
|
||||
lifecycle: ["scan_start_info", "subagent_start_info", "finish_scan", "respond_to_user"],
|
||||
lifecycle: ["scan_start_info", "subagent_start_info", "finish_scan", "wait_for_user", "respond_to_user"],
|
||||
notes: ["create_note", "delete_note", "update_note", "list_notes", "get_note"],
|
||||
skills: ["load_skill"],
|
||||
todos: ["create_todo", "list_todos", "update_todo", "mark_todo_done", "mark_todo_pending", "delete_todo"],
|
||||
|
|
@ -147,7 +147,8 @@ const TOOL_CATEGORY: Record<string, ToolCategory> = Object.fromEntries(
|
|||
*/
|
||||
const RENDERER_OVERRIDES: Partial<Record<string, ComponentType<ToolRendererProps>>> = {
|
||||
finish_scan: FinishRenderer,
|
||||
respond_to_user: RespondRenderer,
|
||||
wait_for_user: WaitForUserRenderer,
|
||||
respond_to_user: WaitForUserRenderer,
|
||||
apply_patch: ApplyPatchRenderer,
|
||||
view_image: ViewImageRenderer,
|
||||
list_reports: ReportListRenderer,
|
||||
|
|
@ -163,6 +164,7 @@ const ICON_OVERRIDES: Partial<Record<string, ToolIconMeta>> = {
|
|||
agent_finish: { icon: Flag, color: "text-cyan-400" },
|
||||
send_message_to_agent: { icon: MessageCircle, color: "text-cyan-400" },
|
||||
wait_for_agents: { icon: MessageCircle, color: "text-cyan-400" },
|
||||
wait_for_user: { icon: MessageCircle, color: "text-emerald-400" },
|
||||
respond_to_user: { icon: MessageCircle, color: "text-emerald-400" },
|
||||
view_agent_graph: { icon: Eye, color: "text-cyan-400" },
|
||||
stop_agent: { icon: Ban, color: "text-red-400" },
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -6,7 +6,7 @@
|
|||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<meta name="color-scheme" content="dark" />
|
||||
<title>Strix Results</title>
|
||||
<script type="module" crossorigin src="./assets/index-DbPDzbcJ.js"></script>
|
||||
<script type="module" crossorigin src="./assets/index-B1JzV4Vu.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="./assets/index-CrLE9rrV.css">
|
||||
</head>
|
||||
<body>
|
||||
|
|
|
|||
|
|
@ -311,8 +311,8 @@ async def wait_for_agents( # noqa: PLR0911
|
|||
**This tool is only for waiting on other agents.** Two things it is
|
||||
NOT for:
|
||||
|
||||
- **Talking to the user.** Use ``respond_to_user``, which delivers
|
||||
your message and hands control back in one call.
|
||||
- **Waiting for the user.** Write your reply as plain text, then call
|
||||
``wait_for_user`` to hand control back.
|
||||
- **Waiting for a long-running command.** This tool does not watch
|
||||
processes at all — it sleeps until a *message* arrives, so it
|
||||
burns the full timeout even if your command finished a second
|
||||
|
|
|
|||
|
|
@ -1,6 +0,0 @@
|
|||
"""User-facing reply tool for interactive sessions."""
|
||||
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
|
||||
|
||||
__all__ = ["respond_to_user"]
|
||||
6
strix/tools/wait_for_user/__init__.py
Normal file
6
strix/tools/wait_for_user/__init__.py
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
"""Hand-the-turn-to-the-user tool for interactive sessions."""
|
||||
|
||||
from strix.tools.wait_for_user.tool import wait_for_user
|
||||
|
||||
|
||||
__all__ = ["wait_for_user"]
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
"""``respond_to_user`` — deliver a reply and hand control back to the user."""
|
||||
"""``wait_for_user`` — hand control back to the user and wait for their reply."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
|
@ -15,23 +15,23 @@ def _ctx(ctx: RunContextWrapper) -> dict[str, Any]:
|
|||
|
||||
|
||||
@function_tool
|
||||
async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
|
||||
"""Answer the user and hand control back to them.
|
||||
async def wait_for_user(ctx: RunContextWrapper) -> str:
|
||||
"""Hand control to the user and wait for their reply.
|
||||
|
||||
This is the ONLY way to yield to the user. Delivering the message and
|
||||
yielding are the same call on purpose: there is no way to answer and
|
||||
then forget to stop, and no way to stop without having answered.
|
||||
This is the ONLY way to yield to the user. It carries no text: everything
|
||||
you write as plain text is already shown to the user, so write your reply
|
||||
first, as plain text, then call this to stop and wait. Never restate in
|
||||
any form what you have already written.
|
||||
|
||||
Call it when you have something for the user and nothing to do until
|
||||
they reply — you answered their question, you need a decision or a
|
||||
credential only they can give, or you finished a chunk of work and
|
||||
want direction. You resume exactly where you left off when they
|
||||
reply, with everything you have done so far intact.
|
||||
Call it when you have nothing to do until the user replies — you answered
|
||||
their question, you need a decision or a credential only they can give,
|
||||
or you finished a chunk of work and want direction. You resume exactly
|
||||
where you left off when they reply, with everything you have done so far
|
||||
intact.
|
||||
|
||||
Do NOT call it to narrate progress or to think out loud. Plain text
|
||||
is still shown to the user as you work, so say whatever you like
|
||||
mid-task without stopping; ``respond_to_user`` is specifically the
|
||||
act of *waiting* for them. Every call costs the user their attention.
|
||||
Do NOT call it to narrate progress or to think out loud: plain text is
|
||||
shown to the user as you work, so say whatever you like mid-task without
|
||||
stopping. Every call costs the user their attention.
|
||||
|
||||
Not for these:
|
||||
|
||||
|
|
@ -39,16 +39,6 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
|
|||
use ``wait_for_agents``.
|
||||
- **Ending the engagement** — use ``finish_scan`` (root) or
|
||||
``agent_finish`` (subagent). Those are terminal; this is a pause.
|
||||
|
||||
Args:
|
||||
message: What to say to the user. Self-contained: they may not
|
||||
have followed the tool calls that led here. Lead with the
|
||||
answer or the decision you need, and if you are blocked, say
|
||||
exactly what you need from them.
|
||||
|
||||
Omit it when you have just said your piece as plain text and
|
||||
only need to wait: that text has already reached them, and
|
||||
repeating it makes them read the same answer twice.
|
||||
"""
|
||||
inner = _ctx(ctx)
|
||||
coordinator = coordinator_from_context(inner)
|
||||
|
|
@ -79,7 +69,7 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
|
|||
stopped = coordinator.statuses.get(me) == "stopped"
|
||||
if stopped:
|
||||
return json.dumps(
|
||||
{"success": True, "wait_outcome": "stopped", "message": message},
|
||||
{"success": True, "wait_outcome": "stopped"},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
|
@ -94,8 +84,7 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
|
|||
"success": True,
|
||||
"wait_outcome": "message_arrived",
|
||||
"pending_messages": pending,
|
||||
"message": message,
|
||||
"note": "Your reply was delivered; the user had already sent a new message.",
|
||||
"note": "The user had already sent a new message; keep going.",
|
||||
},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
|
|
@ -106,8 +95,7 @@ async def respond_to_user(ctx: RunContextWrapper, message: str = "") -> str:
|
|||
{
|
||||
"success": True,
|
||||
"wait_outcome": "waiting",
|
||||
"message": message,
|
||||
"note": "Reply delivered; parked until the user responds.",
|
||||
"note": "Parked until the user responds.",
|
||||
},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
|
|
@ -99,13 +99,13 @@ def test_no_override_renders_builtin_prompt() -> None:
|
|||
assert agent.instructions != ""
|
||||
|
||||
|
||||
def test_respond_to_user_is_interactive_only() -> None:
|
||||
def test_wait_for_user_is_interactive_only() -> None:
|
||||
"""Yielding to the user is meaningless when no user is attached."""
|
||||
interactive = factory.build_strix_agent(is_root=True, interactive=True)
|
||||
autonomous = factory.build_strix_agent(is_root=True, interactive=False)
|
||||
|
||||
assert "respond_to_user" in [t.name for t in interactive.tools]
|
||||
assert "respond_to_user" not in [t.name for t in autonomous.tools]
|
||||
assert "wait_for_user" in [t.name for t in interactive.tools]
|
||||
assert "wait_for_user" not in [t.name for t in autonomous.tools]
|
||||
|
||||
|
||||
def test_wait_for_agents_is_available_in_both_modes() -> None:
|
||||
|
|
|
|||
|
|
@ -13,10 +13,10 @@ from agents.exceptions import MaxTurnsExceeded
|
|||
from agents.items import MessageOutputItem
|
||||
from agents.memory import SQLiteSession
|
||||
from agents.tool_context import ToolContext
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputRefusal
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputRefusal, ResponseOutputText
|
||||
|
||||
from strix.core import execution
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.core.agents import AgentCoordinator, WaitKind
|
||||
from strix.core.execution import (
|
||||
_notify_root_on_budget_reserve,
|
||||
notify_parent_on_terminal,
|
||||
|
|
@ -1030,7 +1030,7 @@ async def test_interactive_text_only_turn_is_nudged_instead_of_parking(
|
|||
# The retry carries an explicit "call a tool" nudge rather than empty input.
|
||||
nudge = calls[1][0]["content"]
|
||||
assert "without a tool call" in nudge
|
||||
assert "respond_to_user" in nudge
|
||||
assert "wait_for_user" in nudge
|
||||
assert coordinator.statuses["root"] == "completed"
|
||||
|
||||
|
||||
|
|
@ -1038,7 +1038,7 @@ async def test_interactive_text_only_turn_is_nudged_instead_of_parking(
|
|||
async def test_interactive_explicit_park_gets_no_nudge(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""``waiting`` is only reachable via respond_to_user / wait_for_agents."""
|
||||
"""``waiting`` is only reachable via wait_for_user / wait_for_agents."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
|
|
@ -1161,7 +1161,7 @@ async def test_tool_required_message_is_persisted_to_the_session(tmp_path: Any)
|
|||
|
||||
stored = [cast("dict[str, Any]", i) for i in await session.get_items()]
|
||||
assert "finish_scan" in stored[0]["content"]
|
||||
assert "respond_to_user" in stored[0]["content"]
|
||||
assert "wait_for_user" in stored[0]["content"]
|
||||
session.close()
|
||||
|
||||
|
||||
|
|
@ -1277,13 +1277,9 @@ async def test_wait_kind_survives_a_snapshot_round_trip() -> None:
|
|||
async def test_interactive_nudge_offers_waiting_without_repeating() -> None:
|
||||
"""The nudge is the instruction an agent reads when it is stranded here.
|
||||
|
||||
It is where the option to wait on what was already said has to be, not only
|
||||
in the system prompt: an agent that ended a turn on plain text reasons off
|
||||
this text, and without the clause it restates its answer to reach a tool
|
||||
call, so the user reads it twice.
|
||||
|
||||
The clause holds whatever the turn did, because the agent is the one who
|
||||
knows whether it spoke — this fires for a turn that produced no text at all.
|
||||
An agent that ended a turn on plain text reasons off this text, and without
|
||||
the clause it restates its answer to reach a tool call, so the user reads it
|
||||
twice.
|
||||
"""
|
||||
items = await execution._append_tool_required_message(
|
||||
session=None,
|
||||
|
|
@ -1293,7 +1289,192 @@ async def test_interactive_nudge_offers_waiting_without_repeating() -> None:
|
|||
interactive=True,
|
||||
)
|
||||
|
||||
assert "with no message if you have already said it" in items[0]["content"]
|
||||
assert "call wait_for_user" in items[0]["content"]
|
||||
assert "do not repeat it" in items[0]["content"]
|
||||
|
||||
|
||||
def _cycle_with_items(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
script: list[tuple[str, WaitKind | None, list[Any]]],
|
||||
calls: list[Any],
|
||||
) -> Any:
|
||||
"""Fake run cycle scripted as (status, wait_kind, new_items) per call."""
|
||||
|
||||
async def _cycle(*_args: Any, **kwargs: Any) -> Any:
|
||||
calls.append(kwargs.get("input_data"))
|
||||
status, wait_kind, new_items = script[min(len(calls) - 1, len(script) - 1)]
|
||||
if wait_kind is not None:
|
||||
await coordinator.park_waiting(agent_id, wait_kind=wait_kind)
|
||||
else:
|
||||
await coordinator.set_status(agent_id, status)
|
||||
return MagicMock(final_output="", new_items=new_items)
|
||||
|
||||
return _cycle
|
||||
|
||||
|
||||
def _text_item(text: str) -> MessageOutputItem:
|
||||
return MessageOutputItem(
|
||||
agent=MagicMock(),
|
||||
raw_item=ResponseOutputMessage(
|
||||
id="msg-1",
|
||||
content=[ResponseOutputText(type="output_text", text=text, annotations=[])],
|
||||
role="assistant",
|
||||
status="completed",
|
||||
type="message",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_silent_wait_for_user_is_sent_back_to_reply(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""``wait_for_user`` carries no text, so a park with nothing said is a silent turn.
|
||||
|
||||
The driver undoes the park and nudges the agent to write its reply; the
|
||||
second cycle, which does say something, parks for real.
|
||||
"""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_cycle_with_items(
|
||||
coordinator,
|
||||
"root",
|
||||
[("waiting", "user", []), ("waiting", "user", [_text_item("Here is the answer.")])],
|
||||
calls,
|
||||
),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 2
|
||||
nudge = calls[1][0]["content"]
|
||||
assert "without having written anything" in nudge
|
||||
assert "wait_for_user" in nudge
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
assert coordinator.wait_kinds["root"] == "user"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wait_for_user_after_text_parks_without_a_nudge(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""Text written in the same turn as the call is the reply; the park stands."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_cycle_with_items(
|
||||
coordinator, "root", [("waiting", "user", [_text_item("Done, see above.")])], calls
|
||||
),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_text_from_an_earlier_nudged_turn_counts_as_the_reply(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A text-only turn is nudged; the bare ``wait_for_user`` that follows is right.
|
||||
|
||||
The words already reached the user in the first cycle, so the second must
|
||||
not be bounced back for repeating them.
|
||||
"""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_cycle_with_items(
|
||||
coordinator,
|
||||
"root",
|
||||
[
|
||||
("running", None, [_text_item("The scan found two issues.")]),
|
||||
("waiting", "user", []),
|
||||
],
|
||||
calls,
|
||||
),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 2
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_whitespace_only_text_does_not_count_as_a_reply(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_cycle_with_items(
|
||||
coordinator,
|
||||
"root",
|
||||
[("waiting", "user", [_text_item(" \n")]), ("waiting", "user", [_text_item("ok")])],
|
||||
calls,
|
||||
),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 2
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_wait_on_agents_is_never_a_silent_yield(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""Only the human wait needs words first; waiting on a child says nothing by design."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_cycle_with_items(coordinator, "root", [("waiting", "agents", [])], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_repeated_silent_yields_exhaust_the_recovery_budget(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""An agent that never writes anything is parked as stalled, not looped forever."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_cycle_with_items(coordinator, "root", [("waiting", "user", [])], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == execution._INTERACTIVE_TOOL_RECOVERY_LIMIT
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
assert coordinator.wait_kinds["root"] == "stalled"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -1307,4 +1488,4 @@ async def test_autonomous_nudge_does_not_offer_the_user() -> None:
|
|||
interactive=False,
|
||||
)
|
||||
|
||||
assert "respond_to_user" not in items[0]["content"]
|
||||
assert "wait_for_user" not in items[0]["content"]
|
||||
|
|
|
|||
|
|
@ -16,12 +16,12 @@ from strix.agents import factory
|
|||
from strix.tools.agents_graph.tools import agent_finish
|
||||
from strix.tools.finish.tool import finish_scan
|
||||
from strix.tools.reporting.tool import create_vulnerability_report
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
from strix.tools.wait_for_user.tool import wait_for_user
|
||||
|
||||
|
||||
_SCAN_AGENT_TOOLS = [
|
||||
tool
|
||||
for tool in (*factory._BASE_TOOLS, finish_scan, agent_finish, respond_to_user)
|
||||
for tool in (*factory._BASE_TOOLS, finish_scan, agent_finish, wait_for_user)
|
||||
if isinstance(tool, FunctionTool)
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
"""Tests for the ``respond_to_user`` yield tool."""
|
||||
"""Tests for the ``wait_for_user`` yield tool."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
|
@ -9,17 +9,17 @@ import pytest
|
|||
from agents.tool_context import ToolContext
|
||||
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
from strix.tools.wait_for_user.tool import wait_for_user
|
||||
|
||||
|
||||
async def _call(context: dict[str, Any], message: str = "here is what I found") -> dict[str, Any]:
|
||||
async def _call(context: dict[str, Any]) -> dict[str, Any]:
|
||||
ctx = ToolContext(
|
||||
context=context,
|
||||
tool_name="respond_to_user",
|
||||
tool_name="wait_for_user",
|
||||
tool_call_id="call-1",
|
||||
tool_arguments="{}",
|
||||
)
|
||||
raw = await respond_to_user.on_invoke_tool(ctx, json.dumps({"message": message}))
|
||||
raw = await wait_for_user.on_invoke_tool(ctx, "{}")
|
||||
return json.loads(raw) # type: ignore[no-any-return]
|
||||
|
||||
|
||||
|
|
@ -29,15 +29,24 @@ async def _context(*, interactive: bool, agent_id: str = "root") -> dict[str, An
|
|||
return {"coordinator": coordinator, "agent_id": agent_id, "interactive": interactive}
|
||||
|
||||
|
||||
def test_takes_no_arguments() -> None:
|
||||
"""Plain text is the only channel to the user, so the tool carries none.
|
||||
|
||||
A message parameter here is a second channel the model fills with the same
|
||||
words it already wrote, and the user reads its answer twice.
|
||||
"""
|
||||
assert wait_for_user.params_json_schema.get("properties", {}) == {}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parks_the_agent_and_carries_the_message() -> None:
|
||||
async def test_parks_the_agent_for_the_user() -> None:
|
||||
context = await _context(interactive=True)
|
||||
result = await _call(context)
|
||||
|
||||
coordinator = context["coordinator"]
|
||||
assert result["success"] is True
|
||||
assert result["wait_outcome"] == "waiting"
|
||||
assert result["message"] == "here is what I found"
|
||||
assert "message" not in result
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
# Recorded as a human wait, so the driver never auto-resumes it.
|
||||
assert coordinator.wait_kinds["root"] == "user"
|
||||
|
|
@ -66,29 +75,13 @@ async def test_a_message_that_already_arrived_is_taken_instead_of_parking() -> N
|
|||
assert coordinator.statuses["root"] == "running"
|
||||
|
||||
|
||||
async def _call_without_message(context: dict[str, Any]) -> dict[str, Any]:
|
||||
ctx = ToolContext(
|
||||
context=context,
|
||||
tool_name="respond_to_user",
|
||||
tool_call_id="call-1",
|
||||
tool_arguments="{}",
|
||||
)
|
||||
raw = await respond_to_user.on_invoke_tool(ctx, "{}")
|
||||
return json.loads(raw) # type: ignore[no-any-return]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parks_without_a_message() -> None:
|
||||
"""An agent that has already said its piece as plain text can just wait.
|
||||
|
||||
The nudge is what leaves it here, and while a message was required the only
|
||||
way to stop was to send the same answer a second time.
|
||||
"""
|
||||
async def test_a_stopped_agent_is_left_stopped() -> None:
|
||||
context = await _context(interactive=True)
|
||||
coordinator = context["coordinator"]
|
||||
await coordinator.set_status("root", "stopped")
|
||||
|
||||
result = await _call_without_message(context)
|
||||
result = await _call(context)
|
||||
|
||||
assert result["success"] is True
|
||||
assert result["wait_outcome"] == "waiting"
|
||||
assert result["message"] == ""
|
||||
assert context["coordinator"].statuses["root"] == "waiting"
|
||||
assert result["wait_outcome"] == "stopped"
|
||||
assert coordinator.statuses["root"] == "stopped"
|
||||
Loading…
Add table
Reference in a new issue