diff --git a/litellm/proxy/spend_tracking/spend_tracking_utils.py b/litellm/proxy/spend_tracking/spend_tracking_utils.py index d3642cb12a4..b38d5e39800 100644 --- a/litellm/proxy/spend_tracking/spend_tracking_utils.py +++ b/litellm/proxy/spend_tracking/spend_tracking_utils.py @@ -136,7 +136,7 @@ def _get_spend_logs_metadata( clean_metadata["vector_store_request_metadata"] = _get_vector_store_request_for_spend_logs_payload( vector_store_request_metadata ) - clean_metadata["guardrail_information"] = guardrail_information + clean_metadata["guardrail_information"] = _sanitize_guardrail_information_for_spend_logs(guardrail_information) clean_metadata["usage_object"] = usage_object clean_metadata["model_map_information"] = model_map_information clean_metadata["cold_storage_object_key"] = cold_storage_object_key @@ -868,6 +868,51 @@ def _redact_prompt_leaks_in_error_string(text: str) -> str: return "".join(out) +def _sanitize_guardrail_information_for_spend_logs( + guardrail_information: Optional[List[StandardLoggingGuardrailInformation]], +) -> Optional[List[StandardLoggingGuardrailInformation]]: + """ + When ``store_prompts_in_spend_logs`` is False, redact prompt-carrying fields + (``guardrail_request``, ``guardrail_response``, ``match_details``, + ``classification``) before they land in ``LiteLLM_SpendLogs.metadata``. + + Guardrail hooks may echo the LLM request payload back into + ``guardrail_response``, and two first-party hooks + (``block_code_execution``, ``litellm_content_filter``) inline user-prompt + substrings into ``match_details`` / ``classification`` too, so the flag + must cover all four fields. Every other typed field on the entry (name, + provider, mode, status, timings, action, violation_categories, risk_score, + masked_entity_count, ...) is preserved so guardrail dashboards keep + working. + + ``guardrail_information`` is typed ``Optional[List[...]]`` but at least + one writer (``xecguard``) assigns a bare dict, so normalize to a list + here to match OTEL's defensive read pattern; otherwise iteration would + yield the dict's keys and crash the whole spend-log write. + """ + if guardrail_information is None or _should_store_prompts_and_responses_in_spend_logs(): + return guardrail_information + entries = [guardrail_information] if isinstance(guardrail_information, dict) else guardrail_information + return [_redact_prompt_fields_in_guardrail_entry(entry) for entry in entries if isinstance(entry, dict)] + + +_PROMPT_CARRYING_GUARDRAIL_FIELDS = ( + "guardrail_request", + "guardrail_response", + "match_details", + "classification", +) + + +def _redact_prompt_fields_in_guardrail_entry( + entry: StandardLoggingGuardrailInformation, +) -> StandardLoggingGuardrailInformation: + return { + **entry, + **{key: REDACTED_BY_LITELM_STRING for key in _PROMPT_CARRYING_GUARDRAIL_FIELDS if key in entry}, + } + + def _sanitize_error_information_for_spend_logs( error_information: Optional[StandardLoggingPayloadErrorInformation], ) -> Optional[StandardLoggingPayloadErrorInformation]: diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 8d474e2e880..8a5acc24d19 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2698,7 +2698,7 @@ class StandardLoggingGuardrailInformation(TypedDict, total=False): guardrail_name: Optional[str] guardrail_provider: Optional[str] guardrail_mode: Optional[Union[GuardrailEventHooks, List[GuardrailEventHooks], GuardrailMode]] - guardrail_request: Optional[dict] + guardrail_request: Optional[Union[str, dict]] guardrail_response: Optional[Union[dict, str, List[dict]]] guardrail_status: GuardrailStatus start_time: Optional[float] @@ -2729,10 +2729,10 @@ class StandardLoggingGuardrailInformation(TypedDict, total=False): confidence_score: Optional[float] """For LLM-judge guardrails: confidence score 0.0-1.0""" - classification: Optional[dict] + classification: Optional[Union[str, dict]] """For LLM-judge guardrails: structured classification output""" - match_details: Optional[List[dict]] + match_details: Optional[Union[str, List[dict]]] """Detailed match information for each detected pattern""" patterns_checked: Optional[int] diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py index 39b5e120c48..9a8f8146d6f 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py @@ -33,6 +33,7 @@ from litellm.proxy.spend_tracking.spend_tracking_utils import ( _is_master_key, _redact_prompt_leaks_in_error_string, _sanitize_error_information_for_spend_logs, + _sanitize_guardrail_information_for_spend_logs, _sanitize_request_body_for_spend_logs_payload, _should_store_prompts_and_responses_in_spend_logs, get_logging_payload, @@ -1263,6 +1264,260 @@ def test_get_spend_logs_metadata_guardrail_info_fallback_from_metadata(): assert result["guardrail_information"] is None +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_redacts_all_prompt_carrying_fields_when_flag_false( + mock_should_store, +): + """ + match_details and classification are declared as structured metadata but + in-tree writers (litellm_content_filter, block_code_execution) inline + raw prompt content into them, so they leak the same way + guardrail_request/guardrail_response do. Redaction must cover all four. + """ + mock_should_store.return_value = False + guardrail_info = [ + { + "guardrail_name": "demo-echo-guard", + "guardrail_status": "success", + "guardrail_request": {"messages": [{"role": "user", "content": "hi"}]}, + "guardrail_response": {"evaluated_input": "hi"}, + "match_details": [{"type": "pattern", "snippet": "hi", "action_taken": "log"}], + "classification": {"intent": "x", "evidence": [{"match": "hi"}]}, + "guardrail_action": "NONE", + } + ] + + result = _sanitize_guardrail_information_for_spend_logs(guardrail_info) + + assert result is not None + entry = result[0] + assert entry["guardrail_request"] == REDACTED_BY_LITELM_STRING + assert entry["guardrail_response"] == REDACTED_BY_LITELM_STRING + assert entry["match_details"] == REDACTED_BY_LITELM_STRING + assert entry["classification"] == REDACTED_BY_LITELM_STRING + assert entry["guardrail_name"] == "demo-echo-guard" + assert entry["guardrail_status"] == "success" + assert entry["guardrail_action"] == "NONE" + + +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_redacts_prompt_fields_when_flag_false( + mock_should_store, +): + """ + LIT-4314 Issue A regression: with store_prompts_in_spend_logs=False, + guardrail_request and guardrail_response must be redacted before they + land in LiteLLM_SpendLogs.metadata, while every other field on the + entry is preserved bit-for-bit. + """ + mock_should_store.return_value = False + guardrail_info = [ + { + "guardrail_name": "demo-echo-guard", + "guardrail_provider": "custom", + "guardrail_mode": "pre_call", + "guardrail_status": "success", + "guardrail_request": { + "messages": [{"role": "user", "content": "Say hi in 3 words"}], + }, + "guardrail_response": { + "evaluated_input": "Say hi in 3 words", + "verdict": "allow", + }, + "start_time": 1_700_000_000.0, + "end_time": 1_700_000_000.5, + "duration": 0.5, + "guardrail_id": "gd-42", + "masked_entity_count": {"EMAIL": 1}, + "violation_categories": ["prompt_injection"], + "risk_score": 3.5, + "guardrail_action": "NONE", + } + ] + + result = _sanitize_guardrail_information_for_spend_logs(guardrail_info) + + assert result is not None + assert len(result) == 1 + entry = result[0] + assert entry["guardrail_request"] == REDACTED_BY_LITELM_STRING + assert entry["guardrail_response"] == REDACTED_BY_LITELM_STRING + assert entry["guardrail_name"] == "demo-echo-guard" + assert entry["guardrail_provider"] == "custom" + assert entry["guardrail_mode"] == "pre_call" + assert entry["guardrail_status"] == "success" + assert entry["start_time"] == 1_700_000_000.0 + assert entry["end_time"] == 1_700_000_000.5 + assert entry["duration"] == 0.5 + assert entry["guardrail_id"] == "gd-42" + assert entry["masked_entity_count"] == {"EMAIL": 1} + assert entry["violation_categories"] == ["prompt_injection"] + assert entry["risk_score"] == 3.5 + assert entry["guardrail_action"] == "NONE" + + assert guardrail_info[0]["guardrail_request"] == { + "messages": [{"role": "user", "content": "Say hi in 3 words"}], + } + assert guardrail_info[0]["guardrail_response"] == { + "evaluated_input": "Say hi in 3 words", + "verdict": "allow", + } + + +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_passthrough_when_flag_true( + mock_should_store, +): + """ + When store_prompts_in_spend_logs=True the sanitizer must be a no-op so + operators who explicitly opted in still see full guardrail payloads. + """ + mock_should_store.return_value = True + guardrail_info = [ + { + "guardrail_name": "content_filter", + "guardrail_status": "success", + "guardrail_request": {"messages": [{"role": "user", "content": "hi"}]}, + "guardrail_response": {"verdict": "allow"}, + } + ] + + result = _sanitize_guardrail_information_for_spend_logs(guardrail_info) + + assert result == guardrail_info + + +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_none_passthrough(mock_should_store): + mock_should_store.return_value = False + assert _sanitize_guardrail_information_for_spend_logs(None) is None + + +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_normalizes_bare_dict_input(mock_should_store): + """ + Regression: xecguard (xecguard.py:246) assigns a bare dict to + standard_logging_object["guardrail_information"] even though the typed + contract is Optional[List[...]]. Without defensive normalization here, + the for-loop would iterate the dict's string keys and _redact... + would TypeError on {**"guardrail_name"}, taking down the entire + spend-log write via update_database's broad except. + """ + mock_should_store.return_value = False + bare_dict_entry = { + "guardrail_name": "xecguard", + "guardrail_status": "success", + "guardrail_response": {"decision": "SAFE", "raw_prompt": "hi"}, + "start_time": 1.0, + "end_time": 2.0, + "duration": 1.0, + } + + result = _sanitize_guardrail_information_for_spend_logs(bare_dict_entry) + + assert result is not None + assert isinstance(result, list) + assert len(result) == 1 + entry = result[0] + assert entry["guardrail_response"] == REDACTED_BY_LITELM_STRING + assert entry["guardrail_name"] == "xecguard" + assert entry["guardrail_status"] == "success" + assert entry["start_time"] == 1.0 + + +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_drops_non_dict_items_in_list(mock_should_store): + """ + A stray non-dict item in the list (e.g. from a buggy caller that + accidentally appends a string) should be silently skipped instead of + crashing the spend-log write. + """ + mock_should_store.return_value = False + mixed_input = [ + {"guardrail_name": "x", "guardrail_response": {"leak": "hi"}}, + "not-a-dict", + None, + ] + + result = _sanitize_guardrail_information_for_spend_logs(mixed_input) + + assert result == [{"guardrail_name": "x", "guardrail_response": REDACTED_BY_LITELM_STRING}] + + +@patch("litellm.proxy.spend_tracking.spend_tracking_utils._should_store_prompts_and_responses_in_spend_logs") +def test_sanitize_guardrail_information_preserves_absent_prompt_fields(mock_should_store): + """ + Entries that never carried guardrail_request or guardrail_response must + not gain those keys after sanitization; consumers keying on presence + (`"guardrail_request" in entry`) would otherwise flip from absent to + the sentinel string. + """ + mock_should_store.return_value = False + guardrail_info = [ + { + "guardrail_name": "demo-echo-guard", + "guardrail_status": "success", + "guardrail_response": {"verdict": "allow", "evaluated_input": "hi"}, + } + ] + + result = _sanitize_guardrail_information_for_spend_logs(guardrail_info) + + assert result is not None + entry = result[0] + assert "guardrail_request" not in entry + assert entry["guardrail_response"] == REDACTED_BY_LITELM_STRING + assert entry["guardrail_name"] == "demo-echo-guard" + assert entry["guardrail_status"] == "success" + + +@patch("litellm.proxy.proxy_server.master_key", "sk-master") +@patch( + "litellm.proxy.proxy_server.general_settings", + {"store_prompts_in_spend_logs": False}, +) +def test_get_logging_payload_redacts_guardrail_prompt_fields_when_flag_false(): + """ + End-to-end wire-in check: get_logging_payload -> _get_spend_logs_metadata + -> sanitizer. Without the wire-in at line 139, the raw guardrail_response + lands in payload["metadata"] verbatim. + """ + guardrail_info = [ + { + "guardrail_name": "demo-echo-guard", + "guardrail_provider": "custom", + "guardrail_status": "success", + "guardrail_request": {"messages": [{"role": "user", "content": "secret"}]}, + "guardrail_response": {"evaluated_input": "secret"}, + } + ] + kwargs = { + "model": "gpt-4o-mini", + "litellm_call_id": "test-call-id", + "litellm_params": { + "metadata": { + "user_api_key": "test-key", + "standard_logging_guardrail_information": guardrail_info, + }, + "proxy_server_request": {}, + }, + } + + payload = get_logging_payload( + kwargs=kwargs, + response_obj={}, + start_time=datetime.datetime.now(tz=timezone.utc), + end_time=datetime.datetime.now(tz=timezone.utc), + ) + + metadata_result = json.loads(payload["metadata"]) + stored = metadata_result["guardrail_information"][0] + assert stored["guardrail_request"] == REDACTED_BY_LITELM_STRING + assert stored["guardrail_response"] == REDACTED_BY_LITELM_STRING + assert stored["guardrail_name"] == "demo-echo-guard" + assert stored["guardrail_status"] == "success" + + def test_get_logging_payload_guardrail_info_when_no_standard_logging_payload(): """ When a guardrail blocks a request before the LLM call, the standard_logging_object @@ -1295,7 +1550,10 @@ def test_get_logging_payload_guardrail_info_when_no_standard_logging_payload(): } with patch("litellm.proxy.proxy_server.master_key", "sk-master"): - with patch("litellm.proxy.proxy_server.general_settings", {}): + with patch( + "litellm.proxy.proxy_server.general_settings", + {"store_prompts_in_spend_logs": True}, + ): payload = get_logging_payload( kwargs=kwargs, response_obj={},