mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
perf(guardrails): stop sending the conversation twice in the noma v2 payload (#36764)
The Noma guardrail sends the conversation to the scanner in `inputs`. It also forwarded `request_data` whole, which repeats that same conversation under `messages` (or `input` on the responses API), and attached `logging_obj.model_call_details`, which repeats it a third time. For image-heavy calls that duplication is most of the request. A production scan of a request carrying base64 images measured 100MB total, of which 94.8MB was `request_data` against 5.1MB of `inputs` - the proxy was uploading ~95% redundant bytes, and paying to serialize them. Drop `messages` and `input` from `request_data` and from `model_call_details`. This is a denylist rather than an allowlist on purpose: every other key is still forwarded untouched, so a scanner-side change that starts reading a new `request_data` key needs no matching release of this hook. The removed keys are ones the scanner never reads - it takes context only from metadata, litellm_metadata, provider_specific_header, litellm_session_id/trace_id/call_id, stream, response/responses ids, and litellm_logging_obj.complete_streaming_response, all of which still pass through. The conversation still reaches the scanner in full via `inputs`, so no detection coverage changes. Trimming happens before serialization, so the duplicate is never encoded. Existing payload tests asserted the duplication; they now assert the trim while keeping what they originally guarded - deep-copy semantics and the unpicklable-object (uvloop.Loop) regression.
This commit is contained in:
parent
ce20c282e9
commit
fbd09ca27d
2 changed files with 90 additions and 9 deletions
|
|
@ -36,6 +36,13 @@ _AIDR_SCAN_ENDPOINT: Final = "/litellm/guardrail"
|
|||
_INTERVENED_INPUT_FIELDS: Final = ("texts", "images", "tools", "tool_calls")
|
||||
_DEFAULT_API_BASE_HOSTNAME: Final = urlparse(_DEFAULT_API_BASE).hostname
|
||||
|
||||
_KEYS_DUPLICATING_SCAN_INPUTS: Final = ("messages", "input")
|
||||
_LOGGING_KEYS_DUPLICATING_SCAN_INPUTS: Final = _KEYS_DUPLICATING_SCAN_INPUTS + (
|
||||
"additional_args",
|
||||
"standard_logging_object",
|
||||
"original_response",
|
||||
)
|
||||
|
||||
|
||||
class _Action(str, enum.Enum):
|
||||
BLOCKED = "BLOCKED"
|
||||
|
|
@ -131,9 +138,20 @@ class NomaV2Guardrail(CustomGuardrail):
|
|||
logging_obj: Optional["LiteLLMLoggingObj"],
|
||||
application_id: str | None,
|
||||
) -> dict:
|
||||
payload_request_data: Final = self._sanitize_payload_for_transport(request_data)
|
||||
payload_request_data: Final = self._sanitize_payload_for_transport(
|
||||
{key: value for key, value in request_data.items() if key not in _KEYS_DUPLICATING_SCAN_INPUTS}
|
||||
)
|
||||
if logging_obj is not None:
|
||||
payload_request_data["litellm_logging_obj"] = getattr(logging_obj, "model_call_details", None)
|
||||
model_call_details: Final = getattr(logging_obj, "model_call_details", None)
|
||||
payload_request_data["litellm_logging_obj"] = (
|
||||
{
|
||||
key: value
|
||||
for key, value in model_call_details.items()
|
||||
if key not in _LOGGING_KEYS_DUPLICATING_SCAN_INPUTS
|
||||
}
|
||||
if isinstance(model_call_details, dict)
|
||||
else model_call_details
|
||||
)
|
||||
|
||||
payload: Final[dict[str, Any]] = {
|
||||
"inputs": inputs,
|
||||
|
|
|
|||
|
|
@ -129,7 +129,11 @@ class TestNomaV2Configuration:
|
|||
)
|
||||
|
||||
assert payload["inputs"] == inputs
|
||||
assert payload["request_data"] == request_data
|
||||
# Everything except the duplicated conversation is forwarded untouched, so a scanner-side
|
||||
# change that starts reading a new request_data key needs no hook release.
|
||||
assert payload["request_data"] == {
|
||||
key: value for key, value in request_data.items() if key != "messages"
|
||||
}
|
||||
assert payload["input_type"] == "request"
|
||||
assert payload["monitor_mode"] is False
|
||||
assert payload["application_id"] == "dynamic-app"
|
||||
|
|
@ -137,6 +141,39 @@ class TestNomaV2Configuration:
|
|||
assert "x-noma-context" not in payload
|
||||
assert "input" not in payload
|
||||
|
||||
def test_build_scan_payload_drops_conversation_duplicated_in_request_data(
|
||||
self, noma_v2_guardrail
|
||||
):
|
||||
"""The scan reads the conversation from `inputs`; repeating it in `request_data` uploaded
|
||||
the whole thing - base64 images included - a second time."""
|
||||
inputs = {"texts": ["hello"]}
|
||||
request_data = {
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"input": [{"role": "user", "content": "hello"}],
|
||||
"metadata": {"headers": {"x-noma-application-id": "header-app"}},
|
||||
"litellm_call_id": "call-id-1",
|
||||
"stream": True,
|
||||
}
|
||||
|
||||
payload = noma_v2_guardrail._build_scan_payload(
|
||||
inputs=inputs,
|
||||
request_data=request_data,
|
||||
input_type="request",
|
||||
logging_obj=None,
|
||||
application_id="dynamic-app",
|
||||
)
|
||||
|
||||
assert "messages" not in payload["request_data"]
|
||||
assert "input" not in payload["request_data"]
|
||||
# Keys the scanner reads for context must survive.
|
||||
assert payload["request_data"]["metadata"] == request_data["metadata"]
|
||||
assert payload["request_data"]["litellm_call_id"] == "call-id-1"
|
||||
assert payload["request_data"]["stream"] is True
|
||||
# The conversation still reaches the scanner through `inputs`.
|
||||
assert payload["inputs"] == inputs
|
||||
# The caller's dict is untouched.
|
||||
assert "messages" in request_data
|
||||
|
||||
def test_build_scan_payload_deep_copies_request_data(self, noma_v2_guardrail):
|
||||
request_data = {
|
||||
"metadata": {"headers": {"x-noma-application-id": "header-app"}},
|
||||
|
|
@ -153,11 +190,11 @@ class TestNomaV2Configuration:
|
|||
payload["request_data"]["metadata"]["headers"][
|
||||
"x-noma-application-id"
|
||||
] = "mutated-value"
|
||||
payload["request_data"]["messages"][0]["content"] = "changed-content"
|
||||
|
||||
assert (
|
||||
request_data["metadata"]["headers"]["x-noma-application-id"] == "header-app"
|
||||
)
|
||||
# Trimming the duplicated conversation must not mutate the caller's dict either.
|
||||
assert request_data["messages"][0]["content"] == "hello"
|
||||
|
||||
def test_build_scan_payload_survives_unpicklable_request_data(
|
||||
|
|
@ -191,14 +228,13 @@ class TestNomaV2Configuration:
|
|||
|
||||
assert isinstance(payload["request_data"], dict)
|
||||
assert payload["request_data"]["event_loop"] == "<fake-uvloop-loop>"
|
||||
assert payload["request_data"]["messages"] == [
|
||||
{"role": "user", "content": "hello"}
|
||||
]
|
||||
assert "messages" not in payload["request_data"]
|
||||
|
||||
# Original request_data must not have been mutated by the copy.
|
||||
assert request_data["event_loop"] is unpicklable
|
||||
assert request_data["messages"] == [{"role": "user", "content": "hello"}]
|
||||
|
||||
def test_build_scan_payload_passes_model_call_details_as_is(
|
||||
def test_build_scan_payload_passes_model_call_details_without_conversation(
|
||||
self, noma_v2_guardrail
|
||||
):
|
||||
class _LoggingObj:
|
||||
|
|
@ -206,6 +242,16 @@ class TestNomaV2Configuration:
|
|||
self.model_call_details = {
|
||||
"model": "gpt-4.1-mini",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"input": [{"role": "user", "content": "hello"}],
|
||||
"additional_args": {
|
||||
"complete_input_dict": {"messages": [{"role": "user", "content": "hello"}]}
|
||||
},
|
||||
"standard_logging_object": {
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"response": {"choices": []},
|
||||
},
|
||||
"original_response": {"choices": []},
|
||||
"complete_streaming_response": {"status": "completed"},
|
||||
"stream": False,
|
||||
"call_type": "acompletion",
|
||||
"litellm_call_id": "call-id-123",
|
||||
|
|
@ -224,8 +270,8 @@ class TestNomaV2Configuration:
|
|||
)
|
||||
|
||||
assert payload["request_data"]["litellm_logging_obj"] == {
|
||||
"complete_streaming_response": {"status": "completed"},
|
||||
"model": "gpt-4.1-mini",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"stream": False,
|
||||
"call_type": "acompletion",
|
||||
"litellm_call_id": "call-id-123",
|
||||
|
|
@ -236,6 +282,23 @@ class TestNomaV2Configuration:
|
|||
assert "logging_obj" not in payload
|
||||
assert request_data["litellm_logging_obj"] == "<Logging object>"
|
||||
|
||||
def test_build_scan_payload_forwards_non_dict_model_call_details_unchanged(
|
||||
self, noma_v2_guardrail
|
||||
):
|
||||
class _LoggingObjWithoutDetails:
|
||||
def __init__(self) -> None:
|
||||
self.model_call_details = None
|
||||
|
||||
payload = noma_v2_guardrail._build_scan_payload(
|
||||
inputs={"texts": ["hello"]},
|
||||
request_data={"litellm_call_id": "call-id-1"},
|
||||
input_type="request",
|
||||
logging_obj=_LoggingObjWithoutDetails(),
|
||||
application_id="test-app",
|
||||
)
|
||||
|
||||
assert payload["request_data"]["litellm_logging_obj"] is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_call_noma_scan_sanitizes_response_model_dump_object(
|
||||
self, noma_v2_guardrail
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue