mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(guardrails): degrade Lasso/Aim mask paths to block on multimodal
Two more in-place rewrite paths exhibit the same regression as Lakera v2: overwriting ``data["messages"]`` with text-only redacted versions silently strips image/audio parts from multimodal requests. - ``LassoGuardrail._run_lasso_guardrail``: when ``mask=True`` AND input is multimodal/Responses-API list, fall back to the classify endpoint (which raises on BLOCK actions but never overwrites the payload). - ``AimGuardrail._anonymize_request``: when input is multimodal, raise the standard 400 instead of replacing ``data["messages"]`` with the text-only ``redacted_chat`` from Aim. The error message tells the user to either send plain string content or rely on block-mode. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
9fcf234750
commit
40817caa4a
3 changed files with 77 additions and 7 deletions
|
|
@ -22,7 +22,10 @@ from litellm.llms.custom_httpx.http_handler import (
|
|||
httpxSpecialProvider,
|
||||
)
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.guardrails._content_utils import build_inspection_messages
|
||||
from litellm.proxy.guardrails._content_utils import (
|
||||
build_inspection_messages,
|
||||
has_non_string_content,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
CallTypesLiteral,
|
||||
Choices,
|
||||
|
|
@ -139,6 +142,20 @@ class AimGuardrail(CustomGuardrail):
|
|||
redacted_chat = res.get("redacted_chat")
|
||||
if not redacted_chat:
|
||||
return data
|
||||
# Aim returns text-only redacted messages. Overwriting
|
||||
# ``data["messages"]`` with that would silently strip image/audio
|
||||
# parts from a multimodal request — degrade to block so the
|
||||
# multimodal payload is never silently rewritten.
|
||||
if has_non_string_content(data):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
"Aim: anonymize action requested for multimodal input "
|
||||
"but mask-in-place would drop non-text parts. Send the "
|
||||
"request with plain string content to use anonymize, "
|
||||
"or rely on block-mode policies."
|
||||
),
|
||||
)
|
||||
data["messages"] = [
|
||||
{
|
||||
"role": message["role"],
|
||||
|
|
|
|||
|
|
@ -50,7 +50,10 @@ from litellm.llms.custom_httpx.http_handler import (
|
|||
httpxSpecialProvider,
|
||||
)
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.guardrails._content_utils import build_inspection_messages
|
||||
from litellm.proxy.guardrails._content_utils import (
|
||||
build_inspection_messages,
|
||||
has_non_string_content,
|
||||
)
|
||||
from litellm.types.guardrails import GuardrailEventHooks
|
||||
import litellm
|
||||
|
||||
|
|
@ -372,12 +375,14 @@ class LassoGuardrail(CustomGuardrail):
|
|||
if not messages:
|
||||
return data
|
||||
|
||||
if self.mask:
|
||||
# Lasso's classifix endpoint returns masked text that we copy back
|
||||
# into ``data["messages"]``. For multimodal/Responses-API input we
|
||||
# would silently strip image/audio parts, so fall back to the
|
||||
# classify endpoint (which still raises on BLOCK actions) and
|
||||
# leave the original payload intact.
|
||||
if self.mask and not has_non_string_content(data):
|
||||
return await self._handle_masking(data, cache, message_type, messages)
|
||||
else:
|
||||
return await self._handle_classification(
|
||||
data, cache, message_type, messages
|
||||
)
|
||||
return await self._handle_classification(data, cache, message_type, messages)
|
||||
|
||||
async def _handle_classification(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -251,6 +251,54 @@ async def test_lakera_v2_inspects_multimodal_list_content(user_api_key, monkeypa
|
|||
# ── Lasso ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_lasso_multimodal_falls_back_to_classify(user_api_key, monkeypatch):
|
||||
"""Lasso's classifix (mask) endpoint returns text that overwrites
|
||||
``data["messages"]``. For multimodal input that would silently strip
|
||||
image parts — the hook must use the classify endpoint instead and
|
||||
leave the original payload intact."""
|
||||
monkeypatch.setenv("LASSO_API_KEY", "ls-test")
|
||||
from litellm.proxy.guardrails.guardrail_hooks.lasso.lasso import LassoGuardrail
|
||||
|
||||
guard = LassoGuardrail(lasso_api_key="ls-test", mask=True)
|
||||
|
||||
masking_called = False
|
||||
classify_called = False
|
||||
|
||||
async def fake_masking(data, cache, message_type, messages):
|
||||
nonlocal masking_called
|
||||
masking_called = True
|
||||
return data
|
||||
|
||||
async def fake_classification(data, cache, message_type, messages):
|
||||
nonlocal classify_called
|
||||
classify_called = True
|
||||
return data
|
||||
|
||||
with (
|
||||
patch.object(guard, "_handle_masking", side_effect=fake_masking),
|
||||
patch.object(guard, "_handle_classification", side_effect=fake_classification),
|
||||
):
|
||||
await guard._run_lasso_guardrail(
|
||||
data={
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "hello"},
|
||||
{"type": "image_url", "image_url": {"url": "..."}},
|
||||
],
|
||||
}
|
||||
]
|
||||
},
|
||||
cache=DualCache(),
|
||||
message_type="PROMPT",
|
||||
)
|
||||
|
||||
assert classify_called is True
|
||||
assert masking_called is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_lasso_inspects_responses_api_input(user_api_key, monkeypatch):
|
||||
monkeypatch.setenv("LASSO_API_KEY", "ls-test")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue