mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(guardrails): propagate Alice WonderFence MASK to texts slot
OpenAI chat translation populates both `structured_messages` and `texts` on guardrail input but reads back only `texts` after apply_guardrail returns. MASK was writing only to `structured_messages` when that was the analyzed source, so the unmasked `texts` slot won downstream and the original prompt reached the LLM while the response header still claimed the guardrail applied. MASK now also overwrites `texts[-1]` whenever `texts` is populated, keeping both slots consistent.
This commit is contained in:
parent
61125912eb
commit
e679caaed8
2 changed files with 46 additions and 7 deletions
|
|
@ -484,19 +484,23 @@ class WonderFenceGuardrail(CustomGuardrail):
|
|||
raise WonderFenceBlockedError(detail)
|
||||
if action == "MASK":
|
||||
masked_text = result.action_text or "[MASKED]"
|
||||
wrote = False
|
||||
if text_source == "structured_messages":
|
||||
inputs["structured_messages"] = set_last_user_message(
|
||||
inputs.get("structured_messages", []), masked_text
|
||||
)
|
||||
elif text_source == "texts":
|
||||
texts = inputs.get("texts", [])
|
||||
wrote = True
|
||||
# Always also overwrite texts[-1] when texts is populated. The
|
||||
# OpenAI chat translation layer reads back only `texts` after
|
||||
# apply_guardrail returns and maps it onto messages — masking
|
||||
# only `structured_messages` lets the unmasked `texts` slot win
|
||||
# and the original prompt reaches the LLM.
|
||||
texts = inputs.get("texts")
|
||||
if texts:
|
||||
texts[-1] = masked_text
|
||||
inputs["texts"] = texts
|
||||
else: # pragma: no cover
|
||||
# Should be unreachable: apply_guardrail short-circuits on no
|
||||
# text. Raise rather than silently drop the mask, which would
|
||||
# send the original prompt to the LLM while the header still
|
||||
# claims the guardrail applied.
|
||||
wrote = True
|
||||
if not wrote: # pragma: no cover
|
||||
raise RuntimeError(
|
||||
"Alice WonderFence MASK requested but no text source — refusing "
|
||||
"to silently no-op."
|
||||
|
|
|
|||
|
|
@ -385,6 +385,41 @@ async def test_apply_guardrail_mask_replaces_structured_messages(guardrail_and_c
|
|||
assert last_user["content"] == "[REDACTED]"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_guardrail_mask_rewrites_texts_when_both_slots_present(
|
||||
guardrail_and_client,
|
||||
):
|
||||
"""OpenAI chat translation populates both `structured_messages` and `texts`,
|
||||
then reads back only `texts`. MASK must overwrite `texts[-1]` even when
|
||||
the analyzed text was extracted from `structured_messages`, otherwise the
|
||||
unmasked `texts` slot wins downstream and the original prompt reaches the
|
||||
LLM while the response header still claims the guardrail applied."""
|
||||
guardrail, client = guardrail_and_client
|
||||
result_obj = Mock()
|
||||
result_obj.action = "MASK"
|
||||
result_obj.action_text = "[REDACTED]"
|
||||
result_obj.detections = []
|
||||
result_obj.correlation_id = None
|
||||
client.evaluate_prompt.return_value = result_obj
|
||||
|
||||
inputs = {
|
||||
"structured_messages": [
|
||||
{"role": "user", "content": "first"},
|
||||
{"role": "assistant", "content": "ack"},
|
||||
{"role": "user", "content": "sensitive content"},
|
||||
],
|
||||
"texts": ["first", "ack", "sensitive content"],
|
||||
}
|
||||
out = await guardrail.apply_guardrail(
|
||||
inputs=inputs,
|
||||
request_data=_request_data(),
|
||||
input_type="request",
|
||||
)
|
||||
assert out["texts"] == ["first", "ack", "[REDACTED]"]
|
||||
last_user = [m for m in out["structured_messages"] if m.get("role") == "user"][-1]
|
||||
assert last_user["content"] == "[REDACTED]"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_guardrail_mask_replaces_last_text_response(guardrail_and_client):
|
||||
guardrail, client = guardrail_and_client
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue