From 40817caa4a1008149dd132c09d8edfa5f66f835b Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Fri, 1 May 2026 04:28:22 +0000 Subject: [PATCH] fix(guardrails): degrade Lasso/Aim mask paths to block on multimodal Two more in-place rewrite paths exhibit the same regression as Lakera v2: overwriting ``data["messages"]`` with text-only redacted versions silently strips image/audio parts from multimodal requests. - ``LassoGuardrail._run_lasso_guardrail``: when ``mask=True`` AND input is multimodal/Responses-API list, fall back to the classify endpoint (which raises on BLOCK actions but never overwrites the payload). - ``AimGuardrail._anonymize_request``: when input is multimodal, raise the standard 400 instead of replacing ``data["messages"]`` with the text-only ``redacted_chat`` from Aim. The error message tells the user to either send plain string content or rely on block-mode. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../guardrails/guardrail_hooks/aim/aim.py | 19 +++++++- .../guardrails/guardrail_hooks/lasso/lasso.py | 17 ++++--- .../guardrails/test_guardrail_coverage.py | 48 +++++++++++++++++++ 3 files changed, 77 insertions(+), 7 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/aim/aim.py b/litellm/proxy/guardrails/guardrail_hooks/aim/aim.py index a3866174286..749bd4acf2b 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/aim/aim.py +++ b/litellm/proxy/guardrails/guardrail_hooks/aim/aim.py @@ -22,7 +22,10 @@ from litellm.llms.custom_httpx.http_handler import ( httpxSpecialProvider, ) from litellm.proxy._types import UserAPIKeyAuth -from litellm.proxy.guardrails._content_utils import build_inspection_messages +from litellm.proxy.guardrails._content_utils import ( + build_inspection_messages, + has_non_string_content, +) from litellm.types.utils import ( CallTypesLiteral, Choices, @@ -139,6 +142,20 @@ class AimGuardrail(CustomGuardrail): redacted_chat = res.get("redacted_chat") if not redacted_chat: return data + # Aim returns text-only redacted messages. Overwriting + # ``data["messages"]`` with that would silently strip image/audio + # parts from a multimodal request — degrade to block so the + # multimodal payload is never silently rewritten. + if has_non_string_content(data): + raise HTTPException( + status_code=400, + detail=( + "Aim: anonymize action requested for multimodal input " + "but mask-in-place would drop non-text parts. Send the " + "request with plain string content to use anonymize, " + "or rely on block-mode policies." + ), + ) data["messages"] = [ { "role": message["role"], diff --git a/litellm/proxy/guardrails/guardrail_hooks/lasso/lasso.py b/litellm/proxy/guardrails/guardrail_hooks/lasso/lasso.py index a7719124b29..04697981085 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/lasso/lasso.py +++ b/litellm/proxy/guardrails/guardrail_hooks/lasso/lasso.py @@ -50,7 +50,10 @@ from litellm.llms.custom_httpx.http_handler import ( httpxSpecialProvider, ) from litellm.proxy._types import UserAPIKeyAuth -from litellm.proxy.guardrails._content_utils import build_inspection_messages +from litellm.proxy.guardrails._content_utils import ( + build_inspection_messages, + has_non_string_content, +) from litellm.types.guardrails import GuardrailEventHooks import litellm @@ -372,12 +375,14 @@ class LassoGuardrail(CustomGuardrail): if not messages: return data - if self.mask: + # Lasso's classifix endpoint returns masked text that we copy back + # into ``data["messages"]``. For multimodal/Responses-API input we + # would silently strip image/audio parts, so fall back to the + # classify endpoint (which still raises on BLOCK actions) and + # leave the original payload intact. + if self.mask and not has_non_string_content(data): return await self._handle_masking(data, cache, message_type, messages) - else: - return await self._handle_classification( - data, cache, message_type, messages - ) + return await self._handle_classification(data, cache, message_type, messages) async def _handle_classification( self, diff --git a/tests/test_litellm/proxy/guardrails/test_guardrail_coverage.py b/tests/test_litellm/proxy/guardrails/test_guardrail_coverage.py index a4fb995ed54..df321b80749 100644 --- a/tests/test_litellm/proxy/guardrails/test_guardrail_coverage.py +++ b/tests/test_litellm/proxy/guardrails/test_guardrail_coverage.py @@ -251,6 +251,54 @@ async def test_lakera_v2_inspects_multimodal_list_content(user_api_key, monkeypa # ── Lasso ───────────────────────────────────────────────────────────────────── +@pytest.mark.asyncio +async def test_lasso_multimodal_falls_back_to_classify(user_api_key, monkeypatch): + """Lasso's classifix (mask) endpoint returns text that overwrites + ``data["messages"]``. For multimodal input that would silently strip + image parts — the hook must use the classify endpoint instead and + leave the original payload intact.""" + monkeypatch.setenv("LASSO_API_KEY", "ls-test") + from litellm.proxy.guardrails.guardrail_hooks.lasso.lasso import LassoGuardrail + + guard = LassoGuardrail(lasso_api_key="ls-test", mask=True) + + masking_called = False + classify_called = False + + async def fake_masking(data, cache, message_type, messages): + nonlocal masking_called + masking_called = True + return data + + async def fake_classification(data, cache, message_type, messages): + nonlocal classify_called + classify_called = True + return data + + with ( + patch.object(guard, "_handle_masking", side_effect=fake_masking), + patch.object(guard, "_handle_classification", side_effect=fake_classification), + ): + await guard._run_lasso_guardrail( + data={ + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": "hello"}, + {"type": "image_url", "image_url": {"url": "..."}}, + ], + } + ] + }, + cache=DualCache(), + message_type="PROMPT", + ) + + assert classify_called is True + assert masking_called is False + + @pytest.mark.asyncio async def test_lasso_inspects_responses_api_input(user_api_key, monkeypatch): monkeypatch.setenv("LASSO_API_KEY", "ls-test")