From 842c423ccd73e3161cde284829cd5612db90353b Mon Sep 17 00:00:00 2001 From: yucheng-berri Date: Sat, 29 Aug 2026 17:07:39 -0700 Subject: [PATCH] fix(guardrails): stop Lakera monitor mode forwarding unmasked PII on Responses-API bodies (#38841) #34940 widened the mask-in-place safety guard so a Responses-API `instructions` field (and a combined messages+input body) skips the PII masking branch. With `on_flagged: "monitor"` that fell straight through to "allow", so PII that used to be masked now reaches the model unredacted. Monitor means "don't block", not "don't redact". Recover the one shape whose payload is still fully writable: mask it and write the redacted instructions back into `data["instructions"]` directly, since apply_redacted_messages_back has no path for that field and would otherwise fold the instructions text into `data["input"]`. The combined messages+input and multimodal shapes stay unmasked - both are unsafe to write back, not merely unwritable - and now log an error naming the reason instead of passing silently. No block/allow decision changes: block and inject_system_message keep the exact outcomes #34940 shipped. --- .../guardrail_hooks/lakera_ai_v2.py | 83 ++++++- .../guardrail_hooks/test_lakera_ai_v2.py | 215 ++++++++++++++++++ 2 files changed, 288 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py b/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py index bcaffa8e91c..2f98a9afbd8 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py +++ b/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py @@ -191,6 +191,25 @@ def _breakdown_has_pii_violation(lakera_response: LakeraAIResponse | None) -> bo ) +def _unmaskable_reason( + guardrail: "LakeraAIGuardrail", + data: dict[str, object], + lakera_response: LakeraAIResponse | None, +) -> str | None: + """Why a PII-only violation on ``data`` can't be masked in place, or None when it can.""" + if has_non_string_content(data): + return "multimodal content, masking would drop the image/audio parts" + if _has_combined_messages_and_input(data): + return "messages and input are both present, so the write-back is positionally ambiguous" + if "messages" in data and not isinstance(data.get("messages"), list): + return "a messages key that isn't a list, so there's nothing to merge the redacted content into" + if not _has_responses_instructions(guardrail, data): + return "no write-back path for the redacted content" + if not (lakera_response or {}).get("payload"): + return "Lakera reported no locations to redact, so payload=true is likely off" + return None + + def _build_lakera_inspection_messages(data: Mapping[str, object]) -> Sequence[Mapping[str, str]]: """Like build_inspection_messages, but also covers the Responses-API ``instructions`` field, placed first since litellm later converts it @@ -473,6 +492,32 @@ class LakeraAIGuardrail(CustomGuardrail): msg["content"] = content return messages + def _mask_unwritable_instructions_pii_in_place( + self, + data: dict[str, object], # mutable-ok: writes the redacted result back into the caller's request dict in place + inspected_messages: Sequence[AllMessageValues], + lakera_response: LakeraAIResponse | None, + masked_entity_count: dict[str, int], + ) -> bool: + """Mask a body whose only obstacle to mask-in-place is the Responses-API + ``instructions`` field, writing the redacted instructions straight into + ``data["instructions"]``: apply_redacted_messages_back has no path for + that field and would fold the instructions text into ``data["input"]``. + Returns False without masking anything when _unmaskable_reason names an + obstacle this can't get around.""" + if _unmaskable_reason(self, data, lakera_response) is not None: + return False + redacted: Final = self._mask_pii_in_messages( + messages=inspected_messages, + lakera_response=lakera_response, + masked_entity_count=masked_entity_count, + ) + # _build_lakera_inspection_messages puts instructions first and + # _filter_skipped_messages kept it, so index 0 is the instructions. + data["instructions"] = redacted[0]["content"] + _apply_redacted_messages_back_preserving_fields(self, data, redacted[1:]) + return True + async def async_pre_call_hook( self, user_api_key_dict: UserAPIKeyAuth, @@ -537,11 +582,12 @@ class LakeraAIGuardrail(CustomGuardrail): ########## 2. Handle flagged content ########## ######################################################### if lakera_guardrail_response.get("flagged") is True: + is_pii_only_violation: Final = self._is_only_pii_violation(lakera_guardrail_response) # PII-only violations get masked in place regardless of on_flagged: there's # no reason to expose raw PII to satisfy an advisory note, and masking is # strictly safer than either blocking or appending an advisory message next # to unredacted PII. - if self._is_only_pii_violation(lakera_guardrail_response) and not is_multimodal_input: + if is_pii_only_violation and not is_multimodal_input: redacted_messages: Final = self._mask_pii_in_messages( messages=new_messages, lakera_response=lakera_guardrail_response, @@ -583,18 +629,35 @@ class LakeraAIGuardrail(CustomGuardrail): # blocking rather than silently letting the flagged request # through with no advisory ever reaching the model. raise self._get_http_exception_for_blocked_guardrail(lakera_guardrail_response) - else: - # Check on_flagged setting - if self.on_flagged == "monitor": + elif self.on_flagged == "monitor": + # Monitor means "don't block", not "don't redact": until the mask + # branch above started skipping shapes it can't write back to, a + # PII-only violation was masked whatever on_flagged said. + masked_in_place: Final = is_pii_only_violation and self._mask_unwritable_instructions_pii_in_place( + data=data, + inspected_messages=new_messages, + lakera_response=lakera_guardrail_response, + masked_entity_count=masked_entity_count, + ) + if masked_in_place: + verbose_proxy_logger.warning( + "Lakera Guardrail: Monitoring mode - PII detected, masked in place and allowing request" + ) + elif is_pii_only_violation: + verbose_proxy_logger.error( + "Lakera Guardrail: Monitoring mode - PII detected but NOT masked, forwarding unredacted " + "content to the model (reason: %s)", + _unmaskable_reason(self, data, lakera_guardrail_response), + ) + else: verbose_proxy_logger.warning( "Lakera Guardrail: Monitoring mode - violation detected but allowing request" ) - # Log violation but continue - elif self.on_flagged == "block": - # Either non-PII violations, or PII on multimodal input - # (which cannot be masked in place without dropping - # image/audio parts) — raise the standard block error. - raise self._get_http_exception_for_blocked_guardrail(lakera_guardrail_response) + elif self.on_flagged == "block": + # Either non-PII violations, or PII on multimodal input + # (which cannot be masked in place without dropping + # image/audio parts) — raise the standard block error. + raise self._get_http_exception_for_blocked_guardrail(lakera_guardrail_response) ######################################################### ########## 3. Add the guardrail to the applied guardrails header ########## diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_lakera_ai_v2.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_lakera_ai_v2.py index 712cf0c2e5a..ee3f8659d51 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_lakera_ai_v2.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_lakera_ai_v2.py @@ -5,6 +5,7 @@ PR checklist requires at least one test in tests/test_litellm/. Additional tests live in tests/guardrails_tests/test_lakera_v2.py. """ +import logging from unittest.mock import AsyncMock, MagicMock, patch import pytest @@ -538,6 +539,220 @@ class TestPiiMaskingSafetyGuard: assert result["messages"][0] == SYSTEM_MSG assert "[MASKED" in result["messages"][1]["content"] + async def test_monitor_mode_masks_responses_input_when_instructions_present(self): + """ + Regression: #34940 added `instructions` to the mask-in-place safety guard, + which skips the mask branch for every Responses-API body carrying one. In + on_flagged="monitor" that dropped through to "allow", so PII in `input` + that was masked before the PR now reached the model unredacted. Monitor + means "don't block", not "don't redact" -- the input is still writable, so + it must still be masked. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "instructions": "be nice", + "input": "a@b.com", + "model": "gpt-3.5-turbo", + "metadata": {}, + } + lakera_response = { + "flagged": True, + "breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}], + "payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 1}], + } + with patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call: + mock_call.return_value = (lakera_response, {}) + result = await guardrail.async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test_key"), + cache=MagicMock(), + data=data, + call_type="responses", + ) + assert result["input"] == "[MASKED EMAIL]" + assert result["instructions"] == "be nice" + + async def test_monitor_mode_masks_pii_carried_in_responses_instructions(self): + """ + Regression: `instructions` is inspected as a synthetic leading system + message but apply_redacted_messages_back has no path to rewrite it, so + monitor mode forwarded the flagged instructions text verbatim. The + redacted instructions must be written straight back into + data["instructions"], and must not be folded into data["input"]. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "instructions": "a@b.com is the contact", + "input": "hi", + "model": "gpt-3.5-turbo", + "metadata": {}, + } + lakera_response = { + "flagged": True, + "breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 0}], + "payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 0}], + } + with patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call: + mock_call.return_value = (lakera_response, {}) + result = await guardrail.async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test_key"), + cache=MagicMock(), + data=data, + call_type="responses", + ) + assert result["instructions"] == "[MASKED EMAIL] is the contact" + assert "a@b.com" not in result["instructions"] + assert result["input"] == "hi" + + async def test_monitor_mode_masks_messages_when_instructions_present(self): + """ + Regression: a chat body that also carries `instructions` hit the same + guard. The messages list has a write-back path, so it must still be + masked in monitor mode, with the untouched instructions preserved. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "instructions": "be nice", + "messages": [{"role": "user", "content": "a@b.com", "name": "u1"}], + "model": "gpt-3.5-turbo", + "metadata": {}, + } + lakera_response = { + "flagged": True, + "breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}], + "payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 1}], + } + with patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call: + mock_call.return_value = (lakera_response, {}) + result = await guardrail.async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test_key"), + cache=MagicMock(), + data=data, + call_type="completion", + ) + assert result["messages"][0]["content"] == "[MASKED EMAIL]" + assert result["messages"][0]["name"] == "u1" + assert result["instructions"] == "be nice" + + async def _monitor_unmasked(self, guardrail, data, lakera_response, caplog, call_type="completion"): + """Drive the monitor path and hand back the result plus the ERROR records it logged.""" + with ( + patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call, + caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"), + ): + mock_call.return_value = (lakera_response, {}) + result = await guardrail.async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test_key"), + cache=MagicMock(), + data=data, + call_type=call_type, + ) + return result, [r.getMessage() for r in caplog.records if r.levelno == logging.ERROR] + + async def test_monitor_mode_leaves_combined_messages_and_input_unmasked(self, caplog): + """ + The combined messages+input shape stays unmasked in monitor mode on + purpose: build_inspection_messages flattens both into one list, so + writing the redacted result back is positionally ambiguous (Greptile P1 + on #34940, see + test_pii_only_violation_with_combined_messages_and_input_blocks_instead_of_masking). + Monitor still must not block, so the request goes through untouched and + the guardrail logs an error naming that reason. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "messages": [{"role": "user", "content": ""}, {"role": "user", "content": "a@b.com"}], + "input": "responses-api content", + "model": "gpt-3.5-turbo", + "metadata": {}, + } + result, errors = await self._monitor_unmasked(guardrail, data, PII_ONLY_LAKERA_RESPONSE, caplog) + assert result["messages"][1]["content"] == "a@b.com" + assert result["input"] == "responses-api content" + assert any("messages and input are both present" in e for e in errors) + + async def test_monitor_mode_multimodal_logs_the_multimodal_reason(self, caplog): + """The multimodal shape was already unmasked before this branch existed; + it must stay that way and say which obstacle it hit, not a generic one.""" + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "messages": [{"role": "user", "content": [{"type": "text", "text": "a@b.com"}]}], + "model": "gpt-3.5-turbo", + "metadata": {}, + } + result, errors = await self._monitor_unmasked(guardrail, data, PII_ONLY_LAKERA_RESPONSE, caplog) + assert result["messages"][0]["content"] == [{"type": "text", "text": "a@b.com"}] + assert any("multimodal content" in e for e in errors) + + async def test_monitor_mode_does_not_claim_masking_when_lakera_sent_no_locations(self, caplog): + """ + payload=false is a supported config for block/monitor, and it makes + Lakera report the violation without the offsets masking needs. Masking + must not silently no-op and report success -- the request goes out + unredacted, so it has to be logged as unredacted. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor", payload=False) + data = { + "instructions": "be nice", + "input": "a@b.com", + "model": "gpt-3.5-turbo", + "metadata": {}, + } + lakera_response = { + "flagged": True, + "breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}], + } + result, errors = await self._monitor_unmasked(guardrail, data, lakera_response, caplog, call_type="responses") + assert result["input"] == "a@b.com" + assert any("no locations to redact" in e for e in errors) + + async def test_monitor_mode_does_not_invent_a_messages_list(self, caplog): + """ + A Responses body carrying a falsy non-list `messages` key must not come + out of the guardrail with a fabricated chat messages list -- the shared + write-back helper keys off `"messages" in data`, not off it being a list. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "instructions": "be nice", + "input": "a@b.com", + "messages": None, + "model": "gpt-3.5-turbo", + "metadata": {}, + } + lakera_response = { + "flagged": True, + "breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}], + "payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 1}], + } + result, errors = await self._monitor_unmasked(guardrail, data, lakera_response, caplog, call_type="responses") + assert result["messages"] is None + assert result["input"] == "a@b.com" + assert any("isn't a list" in e for e in errors) + + async def test_monitor_mode_mixed_violation_is_not_logged_as_an_error(self, caplog): + """ + A PII-plus-prompt-injection violation on an ordinary chat body behaves + exactly as it did before this branch existed, so it must keep logging at + warning level rather than adding error volume to every mixed detection. + """ + guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor") + data = { + "messages": [{"role": "user", "content": "a@b.com"}], + "model": "gpt-3.5-turbo", + "metadata": {}, + } + lakera_response = { + "flagged": True, + "breakdown": [ + {"detector_type": "pii/email", "detected": True, "message_id": 0}, + {"detector_type": "prompt_attack", "detected": True, "message_id": 0}, + ], + "payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 0}], + } + result, errors = await self._monitor_unmasked(guardrail, data, lakera_response, caplog) + assert result["messages"][0]["content"] == "a@b.com" + assert errors == [] + async def test_pii_only_violation_with_uppercase_skipped_role_masks_without_raising(self): """ Greptile finding on BerriAI/litellm#34940: filter_messages_by_skip_flags