fix(guardrails): inspect reasoning and thinking output for Akamai FAI

This commit is contained in:
Scott Jacobsen 2026-07-28 09:40:59 -05:00
parent 720d6082b8
commit 8a37530b2e
2 changed files with 205 additions and 5 deletions

View file

@ -170,7 +170,8 @@ def _iter_responses_api_output_text(response: ResponsesAPIResponse) -> Iterator[
"""Yield text and function-call arguments from a Responses API result.
``/v1/responses`` returns a ``ResponsesAPIResponse`` whose generated text
lives in ``output[].content[].text`` and whose tool-call payloads live in
lives in ``output[].content[].text``, whose reasoning summaries live in
``output[].summary[].text`` and whose tool-call payloads live in
``output[].arguments`` / ``output[].input``; none of it is reachable via
the Chat-Completions ``choices`` shape.
"""
@ -181,6 +182,12 @@ def _iter_responses_api_output_text(response: ResponsesAPIResponse) -> Iterator[
text = _item_get(part, "text")
if isinstance(text, str) and text:
yield text
summary = _item_get(item, "summary")
if isinstance(summary, list):
for part in summary:
text = _item_get(part, "text")
if isinstance(text, str) and text:
yield text
yield from _iter_function_fragments(item)
@ -188,9 +195,10 @@ def _iter_anthropic_output_text(content: Any) -> Iterator[str]:
"""Yield text and tool-call payloads from an Anthropic ``/v1/messages`` reply.
The non-streaming ``/v1/messages`` response reaches the hook as a native
dict whose generated text lives in ``content[].text`` and whose tool calls
live in ``content[].input`` (``type == "tool_use"``); neither is reachable
via the Chat-Completions ``choices`` or the Responses-API ``output`` shapes.
dict whose generated text lives in ``content[].text``, whose extended
thinking lives in ``content[].thinking`` (``type == "thinking"``) and whose
tool calls live in ``content[].input`` (``type == "tool_use"``); none of it
is reachable via the Chat-Completions ``choices`` or Responses-API shapes.
"""
if not isinstance(content, list):
return
@ -200,6 +208,10 @@ def _iter_anthropic_output_text(content: Any) -> Iterator[str]:
text = _item_get(block, "text")
if isinstance(text, str) and text:
yield text
elif block_type == "thinking":
thinking = _item_get(block, "thinking")
if isinstance(thinking, str) and thinking:
yield thinking
elif block_type == "tool_use":
name = _item_get(block, "name")
if isinstance(name, str) and name:
@ -209,6 +221,30 @@ def _iter_anthropic_output_text(content: Any) -> Iterator[str]:
yield json.dumps(tool_input, sort_keys=True)
def _iter_model_response_reasoning_text(response: ModelResponse) -> Iterator[str]:
"""Yield reasoning text carried on a chat ``ModelResponse``.
Reasoning models return their chain of thought outside ``message.content``:
OpenAI-style ``message.reasoning_content`` and Anthropic-style
``message.thinking_blocks[].thinking``. ``stream_chunk_builder`` preserves
both when assembling a stream, so inspecting them here covers the
non-streaming, chat-streaming and Anthropic-streaming paths at once.
Encrypted ``redacted_thinking`` blocks carry no readable text and are skipped.
"""
for choice in response.choices:
message = getattr(choice, "message", None)
if message is None:
continue
reasoning = getattr(message, "reasoning_content", None)
if isinstance(reasoning, str) and reasoning:
yield reasoning
for block in getattr(message, "thinking_blocks", None) or []:
if _item_get(block, "type") == "thinking":
thinking = _item_get(block, "thinking")
if isinstance(thinking, str) and thinking:
yield thinking
class AkamaiRuleTriggered(TypedDict, total=False):
action: str
category: str
@ -298,7 +334,11 @@ class AkamaiFirewallForAIGuardrail(CustomGuardrail):
)
if isinstance(response, ModelResponse):
return get_content_from_model_response(response)
fragments = chain(
[get_content_from_model_response(response)],
_iter_model_response_reasoning_text(response),
)
return "\n".join(fragment for fragment in fragments if fragment)
if isinstance(response, ResponsesAPIResponse):
return "\n".join(_iter_responses_api_output_text(response))
if isinstance(response, dict) and response.get("type") == "message":

View file

@ -807,3 +807,163 @@ async def test_streaming_hook_blocks_anthropic_messages_stream():
# none of the raw Anthropic SSE bytes are delivered
assert all(not isinstance(chunk, (bytes, bytearray)) for chunk in yielded)
assert len(yielded) == 1 and "Blocked by Akamai Firewall for AI" in yielded[0]
@pytest.mark.asyncio
async def test_output_hook_inspects_reasoning_content():
"""Regression: reasoning models emit their chain of thought in reasoning_content.
``get_content_from_model_response`` only reads ``message.content`` and tool
calls, so sensitive text a model places in ``reasoning_content`` reached the
client without a detect request. The content here is benign; only the
reasoning carries the payload, so a block proves reasoning is inspected.
"""
guardrail = _init("post_call")
data = {"litellm_call_id": "req-1", "guardrails": ["akamai-guard"], "messages": [{"role": "user", "content": "hi"}]}
response = ModelResponse(
choices=[
Choices(
index=0,
message=Message(
role="assistant",
content="here is a harmless final answer",
reasoning_content="internally the SSN is AKIA-super-secret",
),
)
]
)
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new=AsyncMock(return_value=_response(BLOCK_BODY)),
) as mock_post:
with pytest.raises(HTTPException):
await guardrail.async_post_call_success_hook(
data=data, user_api_key_dict=UserAPIKeyAuth(), response=response
)
llm_output = mock_post.call_args.kwargs["json"]["llmOutput"]
assert "AKIA-super-secret" in llm_output
assert "here is a harmless final answer" in llm_output
@pytest.mark.asyncio
async def test_output_hook_inspects_thinking_blocks():
"""Regression: Anthropic-style thinking_blocks[].thinking must be inspected too."""
guardrail = _init("post_call")
data = {"litellm_call_id": "req-1", "guardrails": ["akamai-guard"], "messages": [{"role": "user", "content": "hi"}]}
response = ModelResponse(
choices=[
Choices(
index=0,
message=Message(
role="assistant",
content="benign",
thinking_blocks=[
{"type": "thinking", "thinking": "the secret is AKIA-super-secret", "signature": "sig"},
{"type": "redacted_thinking", "data": "opaque-encrypted-blob"},
],
),
)
]
)
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new=AsyncMock(return_value=_response(BLOCK_BODY)),
) as mock_post:
with pytest.raises(HTTPException):
await guardrail.async_post_call_success_hook(
data=data, user_api_key_dict=UserAPIKeyAuth(), response=response
)
llm_output = mock_post.call_args.kwargs["json"]["llmOutput"]
assert "AKIA-super-secret" in llm_output
@pytest.mark.asyncio
async def test_output_hook_inspects_responses_reasoning_summary():
"""Regression: /v1/responses reasoning items carry text in summary[].text."""
guardrail = _init("post_call")
data = {"litellm_call_id": "req-1", "guardrails": ["akamai-guard"], "messages": [{"role": "user", "content": "hi"}]}
response = ResponsesAPIResponse(
id="resp-1",
created_at=1,
output=[
GenericResponseOutputItem(
type="message",
id="m",
status="completed",
role="assistant",
content=[OutputText(type="output_text", text="benign answer", annotations=None)],
),
],
)
# a reasoning item carries its text in summary[].text; append as the raw provider
# dict the Responses API emits (the typed output union does not model it)
response.output.append(
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "reasoning reveals AKIA-super-secret"}]}
)
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new=AsyncMock(return_value=_response(BLOCK_BODY)),
) as mock_post:
with pytest.raises(HTTPException):
await guardrail.async_post_call_success_hook(
data=data, user_api_key_dict=UserAPIKeyAuth(), response=response
)
llm_output = mock_post.call_args.kwargs["json"]["llmOutput"]
assert "AKIA-super-secret" in llm_output
assert "benign answer" in llm_output
@pytest.mark.asyncio
async def test_output_hook_inspects_anthropic_thinking_block():
"""Regression: a native Anthropic reply's thinking content block must be inspected."""
guardrail = _init("post_call")
data = {"litellm_call_id": "req-1", "guardrails": ["akamai-guard"], "messages": [{"role": "user", "content": "hi"}]}
response = {
"id": "msg_1",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-6",
"content": [
{"type": "thinking", "thinking": "quietly the SSN is AKIA-super-secret", "signature": "sig"},
{"type": "text", "text": "benign visible answer"},
],
"stop_reason": "end_turn",
}
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new=AsyncMock(return_value=_response(BLOCK_BODY)),
) as mock_post:
with pytest.raises(HTTPException):
await guardrail.async_post_call_success_hook(
data=data, user_api_key_dict=UserAPIKeyAuth(), response=response
)
llm_output = mock_post.call_args.kwargs["json"]["llmOutput"]
assert "AKIA-super-secret" in llm_output
assert "benign visible answer" in llm_output
@pytest.mark.asyncio
async def test_streaming_hook_inspects_reasoning_content():
"""Streamed reasoning_content deltas are assembled and inspected before delivery."""
guardrail = _init("post_call")
request_data = {"litellm_call_id": "req-1", "guardrails": ["akamai-guard"]}
chunks = [
ModelResponseStream(choices=[StreamingChoices(index=0, delta=Delta(role="assistant", content="benign "))]),
ModelResponseStream(choices=[StreamingChoices(index=0, delta=Delta(content="answer"))]),
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(reasoning_content="secret AKIA-super-secret"))]
),
]
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new=AsyncMock(return_value=_response(BLOCK_BODY)),
) as mock_post:
yielded = [
chunk
async for chunk in guardrail.async_post_call_streaming_iterator_hook(
user_api_key_dict=UserAPIKeyAuth(), response=_aiter(chunks), request_data=request_data
)
]
assert "AKIA-super-secret" in mock_post.call_args.kwargs["json"]["llmOutput"]
assert all(not isinstance(chunk, ModelResponseStream) for chunk in yielded)
assert len(yielded) == 1 and "Blocked by Akamai Firewall for AI" in yielded[0]