fix(guardrails): mask Responses API input_text/output_text content parts

The Presidio Responses-input fix routed extraction/rewrite through
_content_utils, but iter_message_text / walk_user_text only recognised
Chat-Completions `{"type": "text"}` content parts. The Responses API
carries prompt text as `{"type": "input_text", "text": ...}` (and replays
model output as `output_text`), so that user text bypassed masking and
reached the provider unredacted (veria-ai review).

Recognise input_text/output_text alongside text via a shared
TEXT_CONTENT_PART_TYPES set used in the three content-part checks. This
also closes the same gap for the other guardrails using these helpers
(aim, lakera_v2, lasso, ibm).

Adds _content_utils unit tests (extract + in-place rewrite of
input_text/output_text) and a Presidio pre-call e2e test; non-text parts
(input_image) stay untouched.

Refs #30728

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yupeng Lin 2026-06-24 22:21:41 +08:00
parent 1fddd211cc
commit abdb3b6bf2
3 changed files with 113 additions and 3 deletions

View file

@ -35,6 +35,17 @@ def is_text_content_call_type(call_type: str) -> bool:
return call_type in TEXT_CONTENT_CALL_TYPES
# Content-part ``type`` values that carry inspectable text. Chat Completions
# (and Anthropic) use ``"text"``; the Responses API uses ``"input_text"`` for
# prompt input and ``"output_text"`` for prior model output replayed in
# conversation state. All three keep the fragment under the ``"text"`` key, so
# recognising the Responses variants here is enough to extract and rewrite them
# — otherwise ``{"type": "input_text", ...}`` parts slip past masking entirely.
TEXT_CONTENT_PART_TYPES: FrozenSet[str] = frozenset(
{"text", "input_text", "output_text"}
)
def _iter_text_parts_in_content(content: Any) -> Iterator[str]:
"""Yield text fragments from a ``message.content`` value (string or
multimodal list). Non-text parts (images, audio, …) are skipped."""
@ -51,7 +62,7 @@ def _iter_text_parts_in_content(content: Any) -> Iterator[str]:
continue
if not isinstance(part, dict):
continue
if part.get("type") == "text":
if part.get("type") in TEXT_CONTENT_PART_TYPES:
text = part.get("text")
if isinstance(text, str) and text:
yield text
@ -117,7 +128,7 @@ def walk_user_text(data: Dict[str, Any], visit: Callable[[str], str]) -> int:
new_parts.append(visit(part))
elif (
isinstance(part, dict)
and part.get("type") == "text"
and part.get("type") in TEXT_CONTENT_PART_TYPES
and isinstance(part.get("text"), str)
and part["text"]
):
@ -156,7 +167,7 @@ def walk_user_text(data: Dict[str, Any], visit: Callable[[str], str]) -> int:
input_value[idx] = visit(item)
elif (
isinstance(item, dict)
and item.get("type") == "text"
and item.get("type") in TEXT_CONTENT_PART_TYPES
and isinstance(item.get("text"), str)
and item["text"]
):

View file

@ -528,6 +528,44 @@ async def test_responses_api_input_role_messages_are_masked(
assert "test@example.com" not in result["input"][0]["content"]
@pytest.mark.asyncio
async def test_responses_api_input_text_content_parts_are_masked(
presidio_guardrail, mock_user_api_key, mock_cache
):
"""Responses API `input` carries prompt text as `input_text` content parts;
these must be masked too (issue #30728 / veria-ai review). The earlier
refactor only recognised Chat-Completions `text` parts, so `input_text`
slipped through unmasked while non-text parts stay untouched."""
test_data = {
"input": [
{
"role": "user",
"content": [
{"type": "input_text", "text": "card 4111-1111-1111-1111"},
{"type": "input_image", "image_url": "https://x/y.png"},
],
}
],
"model": "gpt-4o",
}
async def mock_check_pii(text, output_parse_pii, presidio_config, request_data):
return text.replace("4111-1111-1111-1111", "[CREDIT_CARD]")
presidio_guardrail.check_pii = mock_check_pii
result = await presidio_guardrail.async_pre_call_hook(
user_api_key_dict=mock_user_api_key,
cache=mock_cache,
data=test_data,
call_type="aresponses",
)
content = result["input"][0]["content"]
assert content[0] == {"type": "input_text", "text": "card [CREDIT_CARD]"}
# Non-text part must be left untouched.
assert content[1] == {"type": "input_image", "image_url": "https://x/y.png"}
assert "4111-1111-1111-1111" not in str(result["input"])
@pytest.mark.asyncio
async def test_both_messages_and_input_are_masked(
presidio_guardrail, mock_user_api_key, mock_cache

View file

@ -66,6 +66,30 @@ def test_iter_message_text_responses_api_list_input_content_parts():
assert list(iter_message_text(data)) == ["alpha", "beta"]
def test_iter_message_text_responses_api_input_text_parts():
"""veria-ai: Responses API carries prompt text as ``input_text`` parts (and
replays model output as ``output_text``); both must be inspected, not just
the Chat-Completions ``text`` type."""
data = {
"input": [
{"type": "input_text", "text": "ssn 078-05-1120"},
{"type": "input_image", "image_url": "..."},
{"type": "output_text", "text": "echoed 078-05-1120"},
]
}
assert list(iter_message_text(data)) == ["ssn 078-05-1120", "echoed 078-05-1120"]
def test_iter_message_text_responses_api_input_text_in_role_messages():
data = {
"input": [
{"role": "user", "content": [{"type": "input_text", "text": "alpha"}]},
{"role": "assistant", "content": [{"type": "output_text", "text": "beta"}]},
]
}
assert list(iter_message_text(data)) == ["alpha", "beta"]
def test_iter_message_text_responses_api_list_input_mixed_dicts_and_strings():
"""Greptile P2: mixed-list ``input`` with content-part dicts AND bare
strings must yield every text fragment — read helpers used to truncate
@ -160,6 +184,43 @@ def test_walk_user_text_redacts_responses_api_list_input():
assert data["input"][1] == {"type": "image_url", "image_url": {"url": "..."}}
def test_walk_user_text_redacts_input_text_parts():
"""veria-ai: ``input_text`` content parts (top-level Responses input list)
must be rewritten in place, not skipped, so the PII is actually masked."""
data = {
"input": [
{"type": "input_text", "text": "AKIAEXAMPLE"},
{"type": "input_image", "image_url": "..."},
]
}
visited = walk_user_text(data, lambda s: s.replace("AKIAEXAMPLE", "[REDACTED]"))
assert visited == 1
assert data["input"][0] == {"type": "input_text", "text": "[REDACTED]"}
assert data["input"][1] == {"type": "input_image", "image_url": "..."}
def test_walk_user_text_redacts_input_text_and_output_text_in_role_messages():
data = {
"input": [
{"role": "user", "content": [{"type": "input_text", "text": "AKIAEXAMPLE"}]},
{
"role": "assistant",
"content": [{"type": "output_text", "text": "saw AKIAEXAMPLE"}],
},
]
}
visited = walk_user_text(data, lambda s: s.replace("AKIAEXAMPLE", "[REDACTED]"))
assert visited == 2
assert data["input"][0]["content"][0] == {
"type": "input_text",
"text": "[REDACTED]",
}
assert data["input"][1]["content"][0] == {
"type": "output_text",
"text": "saw [REDACTED]",
}
def test_walk_user_text_redacts_mixed_list_input():
"""Read and write helpers must agree on coverage — bare strings inside
a mixed ``input`` list are inspected by both."""