fix(guardrails): mask role messages in mixed Responses input arrays

A Responses `input` array can interleave message items (with `role`) and
non-message items (function_call, function_call_output, reasoning) for tool
use. The shared coercion used an all-or-nothing `all("role" in item)` check:
a single tool item made it wrap the whole list as one message's content, so
the real message dict was treated as a content part and its text dropped —
leaving that user PII unmasked (veria-ai review).

Both _coerce_input_to_messages (read) and walk_user_text (write) now handle
each input item independently: message items have their content
extracted/rewritten, loose content-part dicts / bare strings are handled as
text, and non-message tool items are left untouched (their args/output
masking is the separate scope of #30723).

Adds _content_utils unit tests + a Presidio pre-call e2e test for the mixed
message + function_call_output shape.

Refs #30728

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yupeng Lin 2026-06-24 22:54:55 +08:00
parent abdb3b6bf2
commit 30c82526dd
3 changed files with 100 additions and 25 deletions

View file

@ -73,14 +73,25 @@ def _coerce_input_to_messages(input_value: Any) -> List[Dict[str, Any]]:
if isinstance(input_value, str):
return [{"role": "user", "content": input_value}]
if isinstance(input_value, list):
if input_value and all(
isinstance(item, dict) and "role" in item for item in input_value
):
return list(input_value)
# Mixed lists (content-part dicts + bare strings) and pure
# string/dict lists all become a single user message; the content
# iterator below handles each element type uniformly.
return [{"role": "user", "content": input_value}]
# Responses ``input`` is a heterogeneous item array: message items
# (with ``role``), loose content-part dicts / bare strings, and
# non-message items (function_call, function_call_output, reasoning, …).
# Handle each independently — an ``all(role)`` check would misclassify a
# message item the moment a tool item is mixed in, wrapping the whole
# list as one message's content and dropping the real message's text.
messages: List[Dict[str, Any]] = []
loose_parts: List[Any] = []
for item in input_value:
if isinstance(item, dict) and "role" in item:
messages.append(item)
elif isinstance(item, str) or (
isinstance(item, dict) and item.get("type") in TEXT_CONTENT_PART_TYPES
):
loose_parts.append(item)
# else: non-message input item carries no user/assistant prose.
if loose_parts:
messages.append({"role": "user", "content": loose_parts})
return messages
return []
@ -152,27 +163,25 @@ def walk_user_text(data: Dict[str, Any], visit: Callable[[str], str]) -> int:
data["input"] = visit(input_value)
return visited
if isinstance(input_value, list):
# List of full messages: rewrite each message's content.
if input_value and all(
isinstance(item, dict) and "role" in item for item in input_value
):
for item in input_value:
if "content" in item:
item["content"] = _rewrite_content(item["content"])
return visited
# List of content parts and/or bare strings: rewrite in place.
# Heterogeneous Responses item array — rewrite each item independently:
# message items (with ``role``) have their ``content`` rewritten, loose
# content-part dicts / bare strings are rewritten in place, and
# non-message items (function_call_output, …) are left untouched. A
# message item mixed with tool items must not be skipped.
for idx, item in enumerate(input_value):
if isinstance(item, str) and item:
visited += 1
input_value[idx] = visit(item)
elif (
isinstance(item, dict)
and item.get("type") in TEXT_CONTENT_PART_TYPES
and isinstance(item.get("text"), str)
and item["text"]
):
visited += 1
input_value[idx] = {**item, "text": visit(item["text"])}
elif isinstance(item, dict):
if "role" in item and "content" in item:
item["content"] = _rewrite_content(item["content"])
elif (
item.get("type") in TEXT_CONTENT_PART_TYPES
and isinstance(item.get("text"), str)
and item["text"]
):
visited += 1
input_value[idx] = {**item, "text": visit(item["text"])}
return visited
return visited

View file

@ -566,6 +566,40 @@ async def test_responses_api_input_text_content_parts_are_masked(
assert "4111-1111-1111-1111" not in str(result["input"])
@pytest.mark.asyncio
async def test_responses_api_mixed_input_items_are_masked(
presidio_guardrail, mock_user_api_key, mock_cache
):
"""Responses `input` can mix a role message with non-message items (e.g.
function_call_output) for tool use. The message's PII must still be masked
while the tool item is left untouched (issue #30728 / veria-ai review)."""
test_data = {
"input": [
{"role": "user", "content": "my card is 4111-1111-1111-1111"},
{"type": "function_call_output", "call_id": "c1", "output": "done"},
],
"model": "gpt-4o",
}
async def mock_check_pii(text, output_parse_pii, presidio_config, request_data):
return text.replace("4111-1111-1111-1111", "[CREDIT_CARD]")
presidio_guardrail.check_pii = mock_check_pii
result = await presidio_guardrail.async_pre_call_hook(
user_api_key_dict=mock_user_api_key,
cache=mock_cache,
data=test_data,
call_type="aresponses",
)
assert result["input"][0]["content"] == "my card is [CREDIT_CARD]"
assert result["input"][1] == {
"type": "function_call_output",
"call_id": "c1",
"output": "done",
}
assert "4111-1111-1111-1111" not in str(result["input"])
@pytest.mark.asyncio
async def test_both_messages_and_input_are_masked(
presidio_guardrail, mock_user_api_key, mock_cache

View file

@ -90,6 +90,19 @@ def test_iter_message_text_responses_api_input_text_in_role_messages():
assert list(iter_message_text(data)) == ["alpha", "beta"]
def test_iter_message_text_responses_mixed_message_and_tool_items():
"""veria-ai: a Responses ``input`` array mixing a role message with a
non-message item (e.g. function_call_output) must still surface the
message's text — the old all-or-nothing role check dropped it."""
data = {
"input": [
{"role": "user", "content": "card 4111-1111-1111-1111"},
{"type": "function_call_output", "call_id": "c1", "output": "done"},
]
}
assert list(iter_message_text(data)) == ["card 4111-1111-1111-1111"]
def test_iter_message_text_responses_api_list_input_mixed_dicts_and_strings():
"""Greptile P2: mixed-list ``input`` with content-part dicts AND bare
strings must yield every text fragment — read helpers used to truncate
@ -221,6 +234,25 @@ def test_walk_user_text_redacts_input_text_and_output_text_in_role_messages():
}
def test_walk_user_text_redacts_mixed_message_and_tool_items():
"""veria-ai: in a mixed Responses ``input`` array the role message's
content is masked in place while non-message items stay untouched."""
data = {
"input": [
{"role": "user", "content": "card 4111-1111-1111-1111"},
{"type": "function_call_output", "call_id": "c1", "output": "done"},
]
}
visited = walk_user_text(data, lambda s: s.replace("4111-1111-1111-1111", "[CARD]"))
assert visited == 1
assert data["input"][0]["content"] == "card [CARD]"
assert data["input"][1] == {
"type": "function_call_output",
"call_id": "c1",
"output": "done",
}
def test_walk_user_text_redacts_mixed_list_input():
"""Read and write helpers must agree on coverage — bare strings inside
a mixed ``input`` list are inspected by both."""