fix(guardrails): guard tools write-back under scan_only_tool_results and warn on role-filtered no-op scans

This commit is contained in:
mateo-berri 2026-08-05 20:49:15 -07:00
parent 28a277e99e
commit c2998dea75
8 changed files with 175 additions and 2 deletions

View file

@ -387,7 +387,7 @@ class AnthropicMessagesHandler(BaseTranslation):
guardrailed_texts: Final = guardrailed_inputs.get("texts", [])
guardrailed_tools: Final = guardrailed_inputs.get("tools")
if guardrailed_tools is not None:
if guardrailed_tools is not None and not scan_only_tool_results:
# Convert tools back from OpenAI format to Anthropic format
anthropic_config: Final = AnthropicConfig()
anthropic_tools: Final[list[AllAnthropicToolsValues]] = []

View file

@ -143,7 +143,7 @@ class OpenAIChatCompletionsHandler(BaseTranslation):
guardrailed_texts: Final = guardrailed_inputs.get("texts", [])
guardrailed_tool_calls: Final = guardrailed_inputs.get("tool_calls", [])
guardrailed_tools: Final = guardrailed_inputs.get("tools")
if guardrailed_tools is not None:
if guardrailed_tools is not None and not scan_only_tool_results:
data["tools"] = guardrailed_tools
guardrailed_structured_messages: Final = guardrailed_inputs.get("structured_messages")

View file

@ -26,6 +26,9 @@ from litellm.caching import DualCache
from litellm.exceptions import ModifyResponseException
from litellm.integrations.custom_guardrail import CustomGuardrail
from litellm.litellm_core_utils.core_helpers import redact_nested_match_and_regex_keys
from litellm.llms.base_llm.guardrail_translation.utils import (
effective_scan_only_tool_results_for_guardrail,
)
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.custom_httpx.http_handler import (
get_async_httpx_client,
@ -523,6 +526,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
latest_user_index: Final = self._find_latest_message_index(structured_messages, target_role="user")
if latest_user_index is None:
if effective_scan_only_tool_results_for_guardrail(self):
verbose_proxy_logger.warning(
"Bedrock Guardrail: experimental_use_latest_role_message_only scans only the latest "
"user message, so scan_only_tool_results leaves nothing to scan for this request"
)
verbose_proxy_logger.debug("Bedrock Guardrail: no user-role message in request, skipping INPUT scan")
return ApplyGuardrailMessageSelection(None, None, True, skip_scan=True)

View file

@ -22,6 +22,9 @@ from litellm.integrations.custom_guardrail import (
CustomGuardrail,
log_guardrail_information,
)
from litellm.llms.base_llm.guardrail_translation.utils import (
effective_scan_only_tool_results_for_guardrail,
)
from litellm.llms.custom_httpx.http_handler import (
get_async_httpx_client,
httpxSpecialProvider,
@ -1716,6 +1719,15 @@ class PanwPrismaAirsHandler(CustomGuardrail):
# - latest-user extraction returned None (no user / count mismatch)
if scannable_indices is None:
scannable_indices = self._get_scannable_text_indices(texts, structured_messages)
if (
scannable_indices is not None
and not scannable_indices
and effective_scan_only_tool_results_for_guardrail(self)
):
verbose_proxy_logger.warning(
"PANW Prisma AIRS scans only user, system, and developer messages, "
"so scan_only_tool_results leaves nothing to scan for this request"
)
for i, text in enumerate(texts):
if not text or not text.strip():

View file

@ -882,6 +882,40 @@ class TestAnthropicMessagesScanOnlyToolResults:
)
assert data["messages"][2]["content"][0]["text"] == "sibling POISON text"
@pytest.mark.asyncio
async def test_guardrail_synthesized_tools_never_replace_scoped_out_request_tools(self):
handler = AnthropicMessagesHandler()
guardrail = ToolAppendingGuardrail(guardrail_name="tool-appending")
guardrail.scan_only_tool_results = True
original_tools = [
{
"name": "get_weather",
"description": "Get the weather at a specific location",
"input_schema": {"type": "object", "properties": {"location": {"type": "string"}}},
}
]
data = {
"model": "claude-sonnet-4-5",
"tools": original_tools,
"messages": [
{"role": "user", "content": "what's the weather?"},
{
"role": "assistant",
"content": [{"type": "tool_use", "id": "tu1", "name": "get_weather", "input": {}}],
},
{
"role": "user",
"content": [{"type": "tool_result", "tool_use_id": "tu1", "content": "sunny"}],
},
],
}
await handler.process_input_messages(data=data, guardrail_to_apply=guardrail)
assert data["tools"] == original_tools, (
"tools the guardrail synthesized without seeing the request's tools must not replace them"
)
@pytest.mark.asyncio
async def test_guardrail_is_not_called_when_the_request_has_no_tool_results(self):
handler = AnthropicMessagesHandler()

View file

@ -1253,6 +1253,31 @@ class StructuredRedactionGuardrail(CustomGuardrail):
return inputs
class ToolSynthesizingGuardrail(CustomGuardrail):
"""Appends its own function tool to whatever tools it was given, like a
retrieval/recovery guardrail that injects a tool the model can later call."""
def __init__(self):
super().__init__(guardrail_name="tool-synthesizing")
async def apply_guardrail(
self,
inputs: GenericGuardrailAPIInputs,
request_data: dict,
input_type: Literal["request", "response"],
logging_obj: Optional[Any] = None,
) -> GenericGuardrailAPIInputs:
tools = list(inputs.get("tools") or [])
tools.append(
{
"type": "function",
"function": {"name": "injected_retrieve", "parameters": {"type": "object", "properties": {}}},
}
)
inputs["tools"] = tools
return inputs
class TestScanOnlyToolResults:
def _bedrock_guardrail(self):
from litellm.proxy.guardrails.guardrail_hooks.bedrock_guardrails import BedrockGuardrail
@ -1348,6 +1373,35 @@ class TestScanOnlyToolResults:
"function definitions must stay out of a tool-results-only scan"
)
@pytest.mark.parametrize("scan_only_tool_results", [True, False])
@pytest.mark.asyncio
async def test_guardrail_synthesized_tools_never_replace_scoped_out_request_tools(self, scan_only_tool_results):
handler = OpenAIChatCompletionsHandler()
guardrail = ToolSynthesizingGuardrail()
guardrail.scan_only_tool_results = scan_only_tool_results
original_tools = [
{
"type": "function",
"function": {"name": "read_file", "parameters": {"type": "object", "properties": {}}},
}
]
data = {
"messages": [
{"role": "user", "content": "read the report"},
{"role": "tool", "tool_call_id": "call_1", "content": "TOOL-RESULT"},
],
"tools": original_tools,
}
await handler.process_input_messages(data=data, guardrail_to_apply=guardrail)
if scan_only_tool_results:
assert data["tools"] == original_tools, (
"tools the guardrail synthesized without seeing the request's tools must not replace them"
)
else:
assert [t["function"]["name"] for t in data["tools"]] == ["read_file", "injected_retrieve"]
@pytest.mark.asyncio
async def test_structured_write_back_keeps_out_of_scope_messages(self):
handler = OpenAIChatCompletionsHandler()

View file

@ -3670,3 +3670,40 @@ async def test_moderation_hook_honors_the_mcp_event_type(mode, call_type, should
"the scan must be logged under the event it actually ran for, so guardrail logs, "
"OTel spans, and Langfuse metadata do not misclassify MCP enforcement as an LLM call"
)
class TestScanOnlyToolResultsWithLatestRoleFilter:
@pytest.mark.asyncio
async def test_warns_and_skips_when_scoped_payload_has_no_user_message(self):
"""scan_only_tool_results hands Bedrock a tool-role-only payload, but
experimental_use_latest_role_message_only scans only the latest user
message: the silent no-op must warn."""
guardrail = BedrockGuardrail(
guardrail_name="bedrock-latest-role-scoped",
guardrailIdentifier="test-guardrail",
guardrailVersion="DRAFT",
default_on=True,
experimental_use_latest_role_message_only=True,
)
guardrail.scan_only_tool_results = True
inputs = {
"texts": ["TOOL-RESULT"],
"structured_messages": [{"role": "tool", "tool_call_id": "call_1", "content": "TOOL-RESULT"}],
}
with (
patch.object(guardrail, "make_bedrock_api_request", new_callable=AsyncMock) as mock_api,
patch(
"litellm.proxy.guardrails.guardrail_hooks.bedrock_guardrails.verbose_proxy_logger.warning"
) as mock_warning,
):
result = await guardrail.apply_guardrail(
inputs=inputs,
request_data={"litellm_call_id": "test-call-id"},
input_type="request",
)
mock_api.assert_not_called()
assert result["texts"] == ["TOOL-RESULT"]
warning_text = " ".join(str(arg) for c in mock_warning.call_args_list for arg in c.args)
assert "scan_only_tool_results" in warning_text

View file

@ -1696,6 +1696,34 @@ class TestPanwAirsApplyGuardrail:
request_data=request_data, guardrail_name=handler.guardrail_name
)
@pytest.mark.asyncio
async def test_apply_guardrail_warns_when_tool_results_scope_leaves_nothing_scannable(self, handler):
"""scan_only_tool_results hands PANW a tool-role-only payload, but PANW's role
filter only scans user/system/developer rows: the silent no-op must warn."""
handler.scan_only_tool_results = True
inputs: GenericGuardrailAPIInputs = {
"texts": ["TOOL-RESULT"],
"structured_messages": [{"role": "tool", "tool_call_id": "call_1", "content": "TOOL-RESULT"}],
}
request_data = {"litellm_call_id": "test-call-id", "model": "gpt-4"}
with (
patch.object(handler, "_call_panw_api", new_callable=AsyncMock) as mock_api,
patch(
"litellm.proxy.guardrails.guardrail_hooks.panw_prisma_airs.panw_prisma_airs.verbose_proxy_logger.warning"
) as mock_warning,
):
result = await handler.apply_guardrail(
inputs=inputs,
request_data=request_data,
input_type="request",
)
mock_api.assert_not_called()
assert result["texts"] == ["TOOL-RESULT"]
warning_text = " ".join(str(arg) for c in mock_warning.call_args_list for arg in c.args)
assert "scan_only_tool_results" in warning_text
@pytest.mark.asyncio
async def test_apply_guardrail_block(self, handler):
"""Test block action raises HTTPException(400)."""