fix(guardrails): degrade Lasso/Aim mask paths to block on multimodal

Two more in-place rewrite paths exhibit the same regression as Lakera v2:
overwriting ``data["messages"]`` with text-only redacted versions silently
strips image/audio parts from multimodal requests.

- ``LassoGuardrail._run_lasso_guardrail``: when ``mask=True`` AND input
  is multimodal/Responses-API list, fall back to the classify endpoint
  (which raises on BLOCK actions but never overwrites the payload).
- ``AimGuardrail._anonymize_request``: when input is multimodal, raise
  the standard 400 instead of replacing ``data["messages"]`` with the
  text-only ``redacted_chat`` from Aim. The error message tells the
  user to either send plain string content or rely on block-mode.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
user 2026-05-01 04:28:22 +00:00
parent 9fcf234750
commit 40817caa4a
No known key found for this signature in database
3 changed files with 77 additions and 7 deletions

View file

@ -22,7 +22,10 @@ from litellm.llms.custom_httpx.http_handler import (
httpxSpecialProvider,
)
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.guardrails._content_utils import build_inspection_messages
from litellm.proxy.guardrails._content_utils import (
build_inspection_messages,
has_non_string_content,
)
from litellm.types.utils import (
CallTypesLiteral,
Choices,
@ -139,6 +142,20 @@ class AimGuardrail(CustomGuardrail):
redacted_chat = res.get("redacted_chat")
if not redacted_chat:
return data
# Aim returns text-only redacted messages. Overwriting
# ``data["messages"]`` with that would silently strip image/audio
# parts from a multimodal request — degrade to block so the
# multimodal payload is never silently rewritten.
if has_non_string_content(data):
raise HTTPException(
status_code=400,
detail=(
"Aim: anonymize action requested for multimodal input "
"but mask-in-place would drop non-text parts. Send the "
"request with plain string content to use anonymize, "
"or rely on block-mode policies."
),
)
data["messages"] = [
{
"role": message["role"],

View file

@ -50,7 +50,10 @@ from litellm.llms.custom_httpx.http_handler import (
httpxSpecialProvider,
)
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.guardrails._content_utils import build_inspection_messages
from litellm.proxy.guardrails._content_utils import (
build_inspection_messages,
has_non_string_content,
)
from litellm.types.guardrails import GuardrailEventHooks
import litellm
@ -372,12 +375,14 @@ class LassoGuardrail(CustomGuardrail):
if not messages:
return data
if self.mask:
# Lasso's classifix endpoint returns masked text that we copy back
# into ``data["messages"]``. For multimodal/Responses-API input we
# would silently strip image/audio parts, so fall back to the
# classify endpoint (which still raises on BLOCK actions) and
# leave the original payload intact.
if self.mask and not has_non_string_content(data):
return await self._handle_masking(data, cache, message_type, messages)
else:
return await self._handle_classification(
data, cache, message_type, messages
)
return await self._handle_classification(data, cache, message_type, messages)
async def _handle_classification(
self,

View file

@ -251,6 +251,54 @@ async def test_lakera_v2_inspects_multimodal_list_content(user_api_key, monkeypa
# ── Lasso ─────────────────────────────────────────────────────────────────────
@pytest.mark.asyncio
async def test_lasso_multimodal_falls_back_to_classify(user_api_key, monkeypatch):
"""Lasso's classifix (mask) endpoint returns text that overwrites
``data["messages"]``. For multimodal input that would silently strip
image parts — the hook must use the classify endpoint instead and
leave the original payload intact."""
monkeypatch.setenv("LASSO_API_KEY", "ls-test")
from litellm.proxy.guardrails.guardrail_hooks.lasso.lasso import LassoGuardrail
guard = LassoGuardrail(lasso_api_key="ls-test", mask=True)
masking_called = False
classify_called = False
async def fake_masking(data, cache, message_type, messages):
nonlocal masking_called
masking_called = True
return data
async def fake_classification(data, cache, message_type, messages):
nonlocal classify_called
classify_called = True
return data
with (
patch.object(guard, "_handle_masking", side_effect=fake_masking),
patch.object(guard, "_handle_classification", side_effect=fake_classification),
):
await guard._run_lasso_guardrail(
data={
"messages": [
{
"role": "user",
"content": [
{"type": "text", "text": "hello"},
{"type": "image_url", "image_url": {"url": "..."}},
],
}
]
},
cache=DualCache(),
message_type="PROMPT",
)
assert classify_called is True
assert masking_called is False
@pytest.mark.asyncio
async def test_lasso_inspects_responses_api_input(user_api_key, monkeypatch):
monkeypatch.setenv("LASSO_API_KEY", "ls-test")