fix(guardrails): guard Presidio pre-call masking to text call types

The Responses-API refactor replaced the implicit `messages is None`
early-exit with an unconditional `iter_message_text(data)` call, which
also coerced embedding/moderation `input` payloads into maskable
fragments and rewrote them in place. Restore the gate via the shared
`is_text_content_call_type` helper (as banned_keywords and
azure_content_safety already do) so only chat / Responses /
Anthropic-messages traffic is masked.

Add `anthropic_messages` (/v1/messages) to TEXT_CONTENT_CALL_TYPES — it
carries chat text and was already expected to be masked by the existing
test; this closes the same gap for the other two text guardrails.

Refs #30728

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yupeng Lin 2026-06-23 16:18:09 +08:00
parent 8dcd6a45e7
commit 1fddd211cc
3 changed files with 68 additions and 1 deletions

View file

@ -18,13 +18,14 @@ from typing import Any, Callable, Dict, FrozenSet, Iterator, List
#
# /v1/chat/completions -> "acompletion"
# /v1/responses -> "aresponses"
# /v1/messages -> "anthropic_messages"
#
# ``"completion"`` is included for SDK / internal callers that invoke
# ``pre_call_hook`` directly with the sync name. Embedding, moderation,
# audio, and transcription endpoints are deliberately excluded — text
# guardrails on those paths are a separate scope.
TEXT_CONTENT_CALL_TYPES: FrozenSet[str] = frozenset(
{"completion", "acompletion", "aresponses"}
{"completion", "acompletion", "aresponses", "anthropic_messages"}
)

View file

@ -44,6 +44,7 @@ from litellm.integrations.custom_guardrail import (
)
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.guardrails._content_utils import (
is_text_content_call_type,
iter_message_text,
walk_user_text,
)
@ -747,6 +748,16 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
"""
try:
# Only mask request types that carry free-form prompt text
# (chat completions, Responses API, Anthropic messages). Embedding,
# moderation, and audio payloads also expose `input`, but the
# masking path below rewrites text fragments in place — running it
# on an embedding `input` would silently mutate the payload. This is
# the shared gate used by the other text guardrails; the pre-refactor
# hook got this implicitly by reading only `data["messages"]`.
if not is_text_content_call_type(call_type):
return data
content_safety = data.get("content_safety", None)
verbose_proxy_logger.debug("content_safety: %s", content_safety)
presidio_config = self.get_presidio_settings_from_request_data(data)

View file

@ -555,6 +555,61 @@ async def test_both_messages_and_input_are_masked(
assert result["input"] == "input email [EMAIL]"
@pytest.mark.asyncio
async def test_non_text_call_type_input_is_not_masked(
presidio_guardrail, mock_user_api_key, mock_cache
):
"""Embedding/moderation calls also expose `data['input']`, but the masking
path rewrites text in place and must not mutate non-text payloads. The
pre-refactor hook skipped these implicitly by reading only `messages`;
the call-type guard preserves that (issue #30728 review feedback)."""
test_data = {
"input": ["My email is test@example.com", "card 4111-1111-1111-1111"],
"model": "text-embedding-3-small",
}
async def mock_check_pii(text, output_parse_pii, presidio_config, request_data):
raise AssertionError("check_pii should not run for embedding call_type")
presidio_guardrail.check_pii = mock_check_pii
result = await presidio_guardrail.async_pre_call_hook(
user_api_key_dict=mock_user_api_key,
cache=mock_cache,
data=test_data,
call_type="aembedding",
)
# Payload returned untouched.
assert result["input"] == [
"My email is test@example.com",
"card 4111-1111-1111-1111",
]
@pytest.mark.asyncio
async def test_anthropic_messages_input_is_masked(
presidio_guardrail, mock_user_api_key, mock_cache
):
"""`/v1/messages` (call_type 'anthropic_messages') carries chat text and must
be masked — it is part of the text-content call-type set."""
test_data = {
"messages": [{"role": "user", "content": "Contact me at test@example.com"}],
"model": "claude-3-opus-20240229",
}
async def mock_check_pii(text, output_parse_pii, presidio_config, request_data):
return text.replace("test@example.com", "[EMAIL]")
presidio_guardrail.check_pii = mock_check_pii
result = await presidio_guardrail.async_pre_call_hook(
user_api_key_dict=mock_user_api_key,
cache=mock_cache,
data=test_data,
call_type="anthropic_messages",
)
assert result["messages"][0]["content"] == "Contact me at [EMAIL]"
assert "test@example.com" not in result["messages"][0]["content"]
@pytest.mark.asyncio
async def test_logging_hook_multimodal_message_format(presidio_guardrail):
"""