From e679caaed8d054c4cd10e3e9cad9d60ba1956b77 Mon Sep 17 00:00:00 2001 From: lior-k Date: Wed, 6 May 2026 22:47:15 +0300 Subject: [PATCH] fix(guardrails): propagate Alice WonderFence MASK to texts slot OpenAI chat translation populates both `structured_messages` and `texts` on guardrail input but reads back only `texts` after apply_guardrail returns. MASK was writing only to `structured_messages` when that was the analyzed source, so the unmasked `texts` slot won downstream and the original prompt reached the LLM while the response header still claimed the guardrail applied. MASK now also overwrites `texts[-1]` whenever `texts` is populated, keeping both slots consistent. --- .../alice_wonderfence/alice_wonderfence.py | 18 ++++++---- .../guardrail_hooks/test_alice_wonderfence.py | 35 +++++++++++++++++++ 2 files changed, 46 insertions(+), 7 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/alice_wonderfence/alice_wonderfence.py b/litellm/proxy/guardrails/guardrail_hooks/alice_wonderfence/alice_wonderfence.py index 70ff0a26c12..979d71520c7 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/alice_wonderfence/alice_wonderfence.py +++ b/litellm/proxy/guardrails/guardrail_hooks/alice_wonderfence/alice_wonderfence.py @@ -484,19 +484,23 @@ class WonderFenceGuardrail(CustomGuardrail): raise WonderFenceBlockedError(detail) if action == "MASK": masked_text = result.action_text or "[MASKED]" + wrote = False if text_source == "structured_messages": inputs["structured_messages"] = set_last_user_message( inputs.get("structured_messages", []), masked_text ) - elif text_source == "texts": - texts = inputs.get("texts", []) + wrote = True + # Always also overwrite texts[-1] when texts is populated. The + # OpenAI chat translation layer reads back only `texts` after + # apply_guardrail returns and maps it onto messages — masking + # only `structured_messages` lets the unmasked `texts` slot win + # and the original prompt reaches the LLM. + texts = inputs.get("texts") + if texts: texts[-1] = masked_text inputs["texts"] = texts - else: # pragma: no cover - # Should be unreachable: apply_guardrail short-circuits on no - # text. Raise rather than silently drop the mask, which would - # send the original prompt to the LLM while the header still - # claims the guardrail applied. + wrote = True + if not wrote: # pragma: no cover raise RuntimeError( "Alice WonderFence MASK requested but no text source — refusing " "to silently no-op." diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_alice_wonderfence.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_alice_wonderfence.py index 487812a80be..828188a3a95 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_alice_wonderfence.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_alice_wonderfence.py @@ -385,6 +385,41 @@ async def test_apply_guardrail_mask_replaces_structured_messages(guardrail_and_c assert last_user["content"] == "[REDACTED]" +@pytest.mark.asyncio +async def test_apply_guardrail_mask_rewrites_texts_when_both_slots_present( + guardrail_and_client, +): + """OpenAI chat translation populates both `structured_messages` and `texts`, + then reads back only `texts`. MASK must overwrite `texts[-1]` even when + the analyzed text was extracted from `structured_messages`, otherwise the + unmasked `texts` slot wins downstream and the original prompt reaches the + LLM while the response header still claims the guardrail applied.""" + guardrail, client = guardrail_and_client + result_obj = Mock() + result_obj.action = "MASK" + result_obj.action_text = "[REDACTED]" + result_obj.detections = [] + result_obj.correlation_id = None + client.evaluate_prompt.return_value = result_obj + + inputs = { + "structured_messages": [ + {"role": "user", "content": "first"}, + {"role": "assistant", "content": "ack"}, + {"role": "user", "content": "sensitive content"}, + ], + "texts": ["first", "ack", "sensitive content"], + } + out = await guardrail.apply_guardrail( + inputs=inputs, + request_data=_request_data(), + input_type="request", + ) + assert out["texts"] == ["first", "ack", "[REDACTED]"] + last_user = [m for m in out["structured_messages"] if m.get("role") == "user"][-1] + assert last_user["content"] == "[REDACTED]" + + @pytest.mark.asyncio async def test_apply_guardrail_mask_replaces_last_text_response(guardrail_and_client): guardrail, client = guardrail_and_client