fix(guardrails): propagate Alice WonderFence MASK to texts slot

OpenAI chat translation populates both `structured_messages` and `texts`
on guardrail input but reads back only `texts` after apply_guardrail
returns. MASK was writing only to `structured_messages` when that was
the analyzed source, so the unmasked `texts` slot won downstream and
the original prompt reached the LLM while the response header still
claimed the guardrail applied.

MASK now also overwrites `texts[-1]` whenever `texts` is populated,
keeping both slots consistent.
This commit is contained in:
lior-k 2026-05-06 22:47:15 +03:00
parent 61125912eb
commit e679caaed8
No known key found for this signature in database
2 changed files with 46 additions and 7 deletions

View file

@ -484,19 +484,23 @@ class WonderFenceGuardrail(CustomGuardrail):
raise WonderFenceBlockedError(detail)
if action == "MASK":
masked_text = result.action_text or "[MASKED]"
wrote = False
if text_source == "structured_messages":
inputs["structured_messages"] = set_last_user_message(
inputs.get("structured_messages", []), masked_text
)
elif text_source == "texts":
texts = inputs.get("texts", [])
wrote = True
# Always also overwrite texts[-1] when texts is populated. The
# OpenAI chat translation layer reads back only `texts` after
# apply_guardrail returns and maps it onto messages — masking
# only `structured_messages` lets the unmasked `texts` slot win
# and the original prompt reaches the LLM.
texts = inputs.get("texts")
if texts:
texts[-1] = masked_text
inputs["texts"] = texts
else: # pragma: no cover
# Should be unreachable: apply_guardrail short-circuits on no
# text. Raise rather than silently drop the mask, which would
# send the original prompt to the LLM while the header still
# claims the guardrail applied.
wrote = True
if not wrote: # pragma: no cover
raise RuntimeError(
"Alice WonderFence MASK requested but no text source — refusing "
"to silently no-op."

View file

@ -385,6 +385,41 @@ async def test_apply_guardrail_mask_replaces_structured_messages(guardrail_and_c
assert last_user["content"] == "[REDACTED]"
@pytest.mark.asyncio
async def test_apply_guardrail_mask_rewrites_texts_when_both_slots_present(
guardrail_and_client,
):
"""OpenAI chat translation populates both `structured_messages` and `texts`,
then reads back only `texts`. MASK must overwrite `texts[-1]` even when
the analyzed text was extracted from `structured_messages`, otherwise the
unmasked `texts` slot wins downstream and the original prompt reaches the
LLM while the response header still claims the guardrail applied."""
guardrail, client = guardrail_and_client
result_obj = Mock()
result_obj.action = "MASK"
result_obj.action_text = "[REDACTED]"
result_obj.detections = []
result_obj.correlation_id = None
client.evaluate_prompt.return_value = result_obj
inputs = {
"structured_messages": [
{"role": "user", "content": "first"},
{"role": "assistant", "content": "ack"},
{"role": "user", "content": "sensitive content"},
],
"texts": ["first", "ack", "sensitive content"],
}
out = await guardrail.apply_guardrail(
inputs=inputs,
request_data=_request_data(),
input_type="request",
)
assert out["texts"] == ["first", "ack", "[REDACTED]"]
last_user = [m for m in out["structured_messages"] if m.get("role") == "user"][-1]
assert last_user["content"] == "[REDACTED]"
@pytest.mark.asyncio
async def test_apply_guardrail_mask_replaces_last_text_response(guardrail_and_client):
guardrail, client = guardrail_and_client