mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-06 08:16:43 +00:00
Merge remote-tracking branch 'origin/litellm_internal_staging' into litellm_/rbac-batches-action-items-c00ade
This commit is contained in:
commit
c6e95ecc76
2 changed files with 288 additions and 10 deletions
|
|
@ -191,6 +191,25 @@ def _breakdown_has_pii_violation(lakera_response: LakeraAIResponse | None) -> bo
|
|||
)
|
||||
|
||||
|
||||
def _unmaskable_reason(
|
||||
guardrail: "LakeraAIGuardrail",
|
||||
data: dict[str, object],
|
||||
lakera_response: LakeraAIResponse | None,
|
||||
) -> str | None:
|
||||
"""Why a PII-only violation on ``data`` can't be masked in place, or None when it can."""
|
||||
if has_non_string_content(data):
|
||||
return "multimodal content, masking would drop the image/audio parts"
|
||||
if _has_combined_messages_and_input(data):
|
||||
return "messages and input are both present, so the write-back is positionally ambiguous"
|
||||
if "messages" in data and not isinstance(data.get("messages"), list):
|
||||
return "a messages key that isn't a list, so there's nothing to merge the redacted content into"
|
||||
if not _has_responses_instructions(guardrail, data):
|
||||
return "no write-back path for the redacted content"
|
||||
if not (lakera_response or {}).get("payload"):
|
||||
return "Lakera reported no locations to redact, so payload=true is likely off"
|
||||
return None
|
||||
|
||||
|
||||
def _build_lakera_inspection_messages(data: Mapping[str, object]) -> Sequence[Mapping[str, str]]:
|
||||
"""Like build_inspection_messages, but also covers the Responses-API
|
||||
``instructions`` field, placed first since litellm later converts it
|
||||
|
|
@ -473,6 +492,32 @@ class LakeraAIGuardrail(CustomGuardrail):
|
|||
msg["content"] = content
|
||||
return messages
|
||||
|
||||
def _mask_unwritable_instructions_pii_in_place(
|
||||
self,
|
||||
data: dict[str, object], # mutable-ok: writes the redacted result back into the caller's request dict in place
|
||||
inspected_messages: Sequence[AllMessageValues],
|
||||
lakera_response: LakeraAIResponse | None,
|
||||
masked_entity_count: dict[str, int],
|
||||
) -> bool:
|
||||
"""Mask a body whose only obstacle to mask-in-place is the Responses-API
|
||||
``instructions`` field, writing the redacted instructions straight into
|
||||
``data["instructions"]``: apply_redacted_messages_back has no path for
|
||||
that field and would fold the instructions text into ``data["input"]``.
|
||||
Returns False without masking anything when _unmaskable_reason names an
|
||||
obstacle this can't get around."""
|
||||
if _unmaskable_reason(self, data, lakera_response) is not None:
|
||||
return False
|
||||
redacted: Final = self._mask_pii_in_messages(
|
||||
messages=inspected_messages,
|
||||
lakera_response=lakera_response,
|
||||
masked_entity_count=masked_entity_count,
|
||||
)
|
||||
# _build_lakera_inspection_messages puts instructions first and
|
||||
# _filter_skipped_messages kept it, so index 0 is the instructions.
|
||||
data["instructions"] = redacted[0]["content"]
|
||||
_apply_redacted_messages_back_preserving_fields(self, data, redacted[1:])
|
||||
return True
|
||||
|
||||
async def async_pre_call_hook(
|
||||
self,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
|
|
@ -537,11 +582,12 @@ class LakeraAIGuardrail(CustomGuardrail):
|
|||
########## 2. Handle flagged content ##########
|
||||
#########################################################
|
||||
if lakera_guardrail_response.get("flagged") is True:
|
||||
is_pii_only_violation: Final = self._is_only_pii_violation(lakera_guardrail_response)
|
||||
# PII-only violations get masked in place regardless of on_flagged: there's
|
||||
# no reason to expose raw PII to satisfy an advisory note, and masking is
|
||||
# strictly safer than either blocking or appending an advisory message next
|
||||
# to unredacted PII.
|
||||
if self._is_only_pii_violation(lakera_guardrail_response) and not is_multimodal_input:
|
||||
if is_pii_only_violation and not is_multimodal_input:
|
||||
redacted_messages: Final = self._mask_pii_in_messages(
|
||||
messages=new_messages,
|
||||
lakera_response=lakera_guardrail_response,
|
||||
|
|
@ -583,18 +629,35 @@ class LakeraAIGuardrail(CustomGuardrail):
|
|||
# blocking rather than silently letting the flagged request
|
||||
# through with no advisory ever reaching the model.
|
||||
raise self._get_http_exception_for_blocked_guardrail(lakera_guardrail_response)
|
||||
else:
|
||||
# Check on_flagged setting
|
||||
if self.on_flagged == "monitor":
|
||||
elif self.on_flagged == "monitor":
|
||||
# Monitor means "don't block", not "don't redact": until the mask
|
||||
# branch above started skipping shapes it can't write back to, a
|
||||
# PII-only violation was masked whatever on_flagged said.
|
||||
masked_in_place: Final = is_pii_only_violation and self._mask_unwritable_instructions_pii_in_place(
|
||||
data=data,
|
||||
inspected_messages=new_messages,
|
||||
lakera_response=lakera_guardrail_response,
|
||||
masked_entity_count=masked_entity_count,
|
||||
)
|
||||
if masked_in_place:
|
||||
verbose_proxy_logger.warning(
|
||||
"Lakera Guardrail: Monitoring mode - PII detected, masked in place and allowing request"
|
||||
)
|
||||
elif is_pii_only_violation:
|
||||
verbose_proxy_logger.error(
|
||||
"Lakera Guardrail: Monitoring mode - PII detected but NOT masked, forwarding unredacted "
|
||||
"content to the model (reason: %s)",
|
||||
_unmaskable_reason(self, data, lakera_guardrail_response),
|
||||
)
|
||||
else:
|
||||
verbose_proxy_logger.warning(
|
||||
"Lakera Guardrail: Monitoring mode - violation detected but allowing request"
|
||||
)
|
||||
# Log violation but continue
|
||||
elif self.on_flagged == "block":
|
||||
# Either non-PII violations, or PII on multimodal input
|
||||
# (which cannot be masked in place without dropping
|
||||
# image/audio parts) — raise the standard block error.
|
||||
raise self._get_http_exception_for_blocked_guardrail(lakera_guardrail_response)
|
||||
elif self.on_flagged == "block":
|
||||
# Either non-PII violations, or PII on multimodal input
|
||||
# (which cannot be masked in place without dropping
|
||||
# image/audio parts) — raise the standard block error.
|
||||
raise self._get_http_exception_for_blocked_guardrail(lakera_guardrail_response)
|
||||
|
||||
#########################################################
|
||||
########## 3. Add the guardrail to the applied guardrails header ##########
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ PR checklist requires at least one test in tests/test_litellm/.
|
|||
Additional tests live in tests/guardrails_tests/test_lakera_v2.py.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
|
@ -538,6 +539,220 @@ class TestPiiMaskingSafetyGuard:
|
|||
assert result["messages"][0] == SYSTEM_MSG
|
||||
assert "[MASKED" in result["messages"][1]["content"]
|
||||
|
||||
async def test_monitor_mode_masks_responses_input_when_instructions_present(self):
|
||||
"""
|
||||
Regression: #34940 added `instructions` to the mask-in-place safety guard,
|
||||
which skips the mask branch for every Responses-API body carrying one. In
|
||||
on_flagged="monitor" that dropped through to "allow", so PII in `input`
|
||||
that was masked before the PR now reached the model unredacted. Monitor
|
||||
means "don't block", not "don't redact" -- the input is still writable, so
|
||||
it must still be masked.
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"instructions": "be nice",
|
||||
"input": "a@b.com",
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
lakera_response = {
|
||||
"flagged": True,
|
||||
"breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}],
|
||||
"payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 1}],
|
||||
}
|
||||
with patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call:
|
||||
mock_call.return_value = (lakera_response, {})
|
||||
result = await guardrail.async_pre_call_hook(
|
||||
user_api_key_dict=UserAPIKeyAuth(api_key="test_key"),
|
||||
cache=MagicMock(),
|
||||
data=data,
|
||||
call_type="responses",
|
||||
)
|
||||
assert result["input"] == "[MASKED EMAIL]"
|
||||
assert result["instructions"] == "be nice"
|
||||
|
||||
async def test_monitor_mode_masks_pii_carried_in_responses_instructions(self):
|
||||
"""
|
||||
Regression: `instructions` is inspected as a synthetic leading system
|
||||
message but apply_redacted_messages_back has no path to rewrite it, so
|
||||
monitor mode forwarded the flagged instructions text verbatim. The
|
||||
redacted instructions must be written straight back into
|
||||
data["instructions"], and must not be folded into data["input"].
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"instructions": "a@b.com is the contact",
|
||||
"input": "hi",
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
lakera_response = {
|
||||
"flagged": True,
|
||||
"breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 0}],
|
||||
"payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 0}],
|
||||
}
|
||||
with patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call:
|
||||
mock_call.return_value = (lakera_response, {})
|
||||
result = await guardrail.async_pre_call_hook(
|
||||
user_api_key_dict=UserAPIKeyAuth(api_key="test_key"),
|
||||
cache=MagicMock(),
|
||||
data=data,
|
||||
call_type="responses",
|
||||
)
|
||||
assert result["instructions"] == "[MASKED EMAIL] is the contact"
|
||||
assert "a@b.com" not in result["instructions"]
|
||||
assert result["input"] == "hi"
|
||||
|
||||
async def test_monitor_mode_masks_messages_when_instructions_present(self):
|
||||
"""
|
||||
Regression: a chat body that also carries `instructions` hit the same
|
||||
guard. The messages list has a write-back path, so it must still be
|
||||
masked in monitor mode, with the untouched instructions preserved.
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"instructions": "be nice",
|
||||
"messages": [{"role": "user", "content": "a@b.com", "name": "u1"}],
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
lakera_response = {
|
||||
"flagged": True,
|
||||
"breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}],
|
||||
"payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 1}],
|
||||
}
|
||||
with patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call:
|
||||
mock_call.return_value = (lakera_response, {})
|
||||
result = await guardrail.async_pre_call_hook(
|
||||
user_api_key_dict=UserAPIKeyAuth(api_key="test_key"),
|
||||
cache=MagicMock(),
|
||||
data=data,
|
||||
call_type="completion",
|
||||
)
|
||||
assert result["messages"][0]["content"] == "[MASKED EMAIL]"
|
||||
assert result["messages"][0]["name"] == "u1"
|
||||
assert result["instructions"] == "be nice"
|
||||
|
||||
async def _monitor_unmasked(self, guardrail, data, lakera_response, caplog, call_type="completion"):
|
||||
"""Drive the monitor path and hand back the result plus the ERROR records it logged."""
|
||||
with (
|
||||
patch.object(guardrail, "call_v2_guard", new_callable=AsyncMock) as mock_call,
|
||||
caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"),
|
||||
):
|
||||
mock_call.return_value = (lakera_response, {})
|
||||
result = await guardrail.async_pre_call_hook(
|
||||
user_api_key_dict=UserAPIKeyAuth(api_key="test_key"),
|
||||
cache=MagicMock(),
|
||||
data=data,
|
||||
call_type=call_type,
|
||||
)
|
||||
return result, [r.getMessage() for r in caplog.records if r.levelno == logging.ERROR]
|
||||
|
||||
async def test_monitor_mode_leaves_combined_messages_and_input_unmasked(self, caplog):
|
||||
"""
|
||||
The combined messages+input shape stays unmasked in monitor mode on
|
||||
purpose: build_inspection_messages flattens both into one list, so
|
||||
writing the redacted result back is positionally ambiguous (Greptile P1
|
||||
on #34940, see
|
||||
test_pii_only_violation_with_combined_messages_and_input_blocks_instead_of_masking).
|
||||
Monitor still must not block, so the request goes through untouched and
|
||||
the guardrail logs an error naming that reason.
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"messages": [{"role": "user", "content": ""}, {"role": "user", "content": "a@b.com"}],
|
||||
"input": "responses-api content",
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
result, errors = await self._monitor_unmasked(guardrail, data, PII_ONLY_LAKERA_RESPONSE, caplog)
|
||||
assert result["messages"][1]["content"] == "a@b.com"
|
||||
assert result["input"] == "responses-api content"
|
||||
assert any("messages and input are both present" in e for e in errors)
|
||||
|
||||
async def test_monitor_mode_multimodal_logs_the_multimodal_reason(self, caplog):
|
||||
"""The multimodal shape was already unmasked before this branch existed;
|
||||
it must stay that way and say which obstacle it hit, not a generic one."""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"messages": [{"role": "user", "content": [{"type": "text", "text": "a@b.com"}]}],
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
result, errors = await self._monitor_unmasked(guardrail, data, PII_ONLY_LAKERA_RESPONSE, caplog)
|
||||
assert result["messages"][0]["content"] == [{"type": "text", "text": "a@b.com"}]
|
||||
assert any("multimodal content" in e for e in errors)
|
||||
|
||||
async def test_monitor_mode_does_not_claim_masking_when_lakera_sent_no_locations(self, caplog):
|
||||
"""
|
||||
payload=false is a supported config for block/monitor, and it makes
|
||||
Lakera report the violation without the offsets masking needs. Masking
|
||||
must not silently no-op and report success -- the request goes out
|
||||
unredacted, so it has to be logged as unredacted.
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor", payload=False)
|
||||
data = {
|
||||
"instructions": "be nice",
|
||||
"input": "a@b.com",
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
lakera_response = {
|
||||
"flagged": True,
|
||||
"breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}],
|
||||
}
|
||||
result, errors = await self._monitor_unmasked(guardrail, data, lakera_response, caplog, call_type="responses")
|
||||
assert result["input"] == "a@b.com"
|
||||
assert any("no locations to redact" in e for e in errors)
|
||||
|
||||
async def test_monitor_mode_does_not_invent_a_messages_list(self, caplog):
|
||||
"""
|
||||
A Responses body carrying a falsy non-list `messages` key must not come
|
||||
out of the guardrail with a fabricated chat messages list -- the shared
|
||||
write-back helper keys off `"messages" in data`, not off it being a list.
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"instructions": "be nice",
|
||||
"input": "a@b.com",
|
||||
"messages": None,
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
lakera_response = {
|
||||
"flagged": True,
|
||||
"breakdown": [{"detector_type": "pii/email", "detected": True, "message_id": 1}],
|
||||
"payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 1}],
|
||||
}
|
||||
result, errors = await self._monitor_unmasked(guardrail, data, lakera_response, caplog, call_type="responses")
|
||||
assert result["messages"] is None
|
||||
assert result["input"] == "a@b.com"
|
||||
assert any("isn't a list" in e for e in errors)
|
||||
|
||||
async def test_monitor_mode_mixed_violation_is_not_logged_as_an_error(self, caplog):
|
||||
"""
|
||||
A PII-plus-prompt-injection violation on an ordinary chat body behaves
|
||||
exactly as it did before this branch existed, so it must keep logging at
|
||||
warning level rather than adding error volume to every mixed detection.
|
||||
"""
|
||||
guardrail = LakeraAIGuardrail(api_key="test_key", on_flagged="monitor")
|
||||
data = {
|
||||
"messages": [{"role": "user", "content": "a@b.com"}],
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {},
|
||||
}
|
||||
lakera_response = {
|
||||
"flagged": True,
|
||||
"breakdown": [
|
||||
{"detector_type": "pii/email", "detected": True, "message_id": 0},
|
||||
{"detector_type": "prompt_attack", "detected": True, "message_id": 0},
|
||||
],
|
||||
"payload": [{"detector_type": "pii/email", "start": 0, "end": 7, "message_id": 0}],
|
||||
}
|
||||
result, errors = await self._monitor_unmasked(guardrail, data, lakera_response, caplog)
|
||||
assert result["messages"][0]["content"] == "a@b.com"
|
||||
assert errors == []
|
||||
|
||||
async def test_pii_only_violation_with_uppercase_skipped_role_masks_without_raising(self):
|
||||
"""
|
||||
Greptile finding on BerriAI/litellm#34940: filter_messages_by_skip_flags
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue