From 1fddd211cc4c63d5b49192277e971e0a33ae39dd Mon Sep 17 00:00:00 2001 From: Yupeng Lin Date: Tue, 23 Jun 2026 16:18:09 +0800 Subject: [PATCH] fix(guardrails): guard Presidio pre-call masking to text call types MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Responses-API refactor replaced the implicit `messages is None` early-exit with an unconditional `iter_message_text(data)` call, which also coerced embedding/moderation `input` payloads into maskable fragments and rewrote them in place. Restore the gate via the shared `is_text_content_call_type` helper (as banned_keywords and azure_content_safety already do) so only chat / Responses / Anthropic-messages traffic is masked. Add `anthropic_messages` (/v1/messages) to TEXT_CONTENT_CALL_TYPES — it carries chat text and was already expected to be masked by the existing test; this closes the same gap for the other two text guardrails. Refs #30728 Co-Authored-By: Claude Opus 4.8 (1M context) --- litellm/proxy/guardrails/_content_utils.py | 3 +- .../guardrails/guardrail_hooks/presidio.py | 11 ++++ .../guardrail_hooks/test_presidio.py | 55 +++++++++++++++++++ 3 files changed, 68 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/guardrails/_content_utils.py b/litellm/proxy/guardrails/_content_utils.py index 766ef0cf9f6..db1038e2bfa 100644 --- a/litellm/proxy/guardrails/_content_utils.py +++ b/litellm/proxy/guardrails/_content_utils.py @@ -18,13 +18,14 @@ from typing import Any, Callable, Dict, FrozenSet, Iterator, List # # /v1/chat/completions -> "acompletion" # /v1/responses -> "aresponses" +# /v1/messages -> "anthropic_messages" # # ``"completion"`` is included for SDK / internal callers that invoke # ``pre_call_hook`` directly with the sync name. Embedding, moderation, # audio, and transcription endpoints are deliberately excluded — text # guardrails on those paths are a separate scope. TEXT_CONTENT_CALL_TYPES: FrozenSet[str] = frozenset( - {"completion", "acompletion", "aresponses"} + {"completion", "acompletion", "aresponses", "anthropic_messages"} ) diff --git a/litellm/proxy/guardrails/guardrail_hooks/presidio.py b/litellm/proxy/guardrails/guardrail_hooks/presidio.py index 6b3ab0df21b..84d93566cfa 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/presidio.py +++ b/litellm/proxy/guardrails/guardrail_hooks/presidio.py @@ -44,6 +44,7 @@ from litellm.integrations.custom_guardrail import ( ) from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.guardrails._content_utils import ( + is_text_content_call_type, iter_message_text, walk_user_text, ) @@ -747,6 +748,16 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail): """ try: + # Only mask request types that carry free-form prompt text + # (chat completions, Responses API, Anthropic messages). Embedding, + # moderation, and audio payloads also expose `input`, but the + # masking path below rewrites text fragments in place — running it + # on an embedding `input` would silently mutate the payload. This is + # the shared gate used by the other text guardrails; the pre-refactor + # hook got this implicitly by reading only `data["messages"]`. + if not is_text_content_call_type(call_type): + return data + content_safety = data.get("content_safety", None) verbose_proxy_logger.debug("content_safety: %s", content_safety) presidio_config = self.get_presidio_settings_from_request_data(data) diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py index 3599b1098b7..1f4c31ab48d 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py @@ -555,6 +555,61 @@ async def test_both_messages_and_input_are_masked( assert result["input"] == "input email [EMAIL]" +@pytest.mark.asyncio +async def test_non_text_call_type_input_is_not_masked( + presidio_guardrail, mock_user_api_key, mock_cache +): + """Embedding/moderation calls also expose `data['input']`, but the masking + path rewrites text in place and must not mutate non-text payloads. The + pre-refactor hook skipped these implicitly by reading only `messages`; + the call-type guard preserves that (issue #30728 review feedback).""" + test_data = { + "input": ["My email is test@example.com", "card 4111-1111-1111-1111"], + "model": "text-embedding-3-small", + } + + async def mock_check_pii(text, output_parse_pii, presidio_config, request_data): + raise AssertionError("check_pii should not run for embedding call_type") + + presidio_guardrail.check_pii = mock_check_pii + result = await presidio_guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key, + cache=mock_cache, + data=test_data, + call_type="aembedding", + ) + # Payload returned untouched. + assert result["input"] == [ + "My email is test@example.com", + "card 4111-1111-1111-1111", + ] + + +@pytest.mark.asyncio +async def test_anthropic_messages_input_is_masked( + presidio_guardrail, mock_user_api_key, mock_cache +): + """`/v1/messages` (call_type 'anthropic_messages') carries chat text and must + be masked — it is part of the text-content call-type set.""" + test_data = { + "messages": [{"role": "user", "content": "Contact me at test@example.com"}], + "model": "claude-3-opus-20240229", + } + + async def mock_check_pii(text, output_parse_pii, presidio_config, request_data): + return text.replace("test@example.com", "[EMAIL]") + + presidio_guardrail.check_pii = mock_check_pii + result = await presidio_guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key, + cache=mock_cache, + data=test_data, + call_type="anthropic_messages", + ) + assert result["messages"][0]["content"] == "Contact me at [EMAIL]" + assert "test@example.com" not in result["messages"][0]["content"] + + @pytest.mark.asyncio async def test_logging_hook_multimodal_message_format(presidio_guardrail): """