From 5c375b23ae2421ffe57499aeba33b4f5b27c3e77 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 28 Oct 2025 16:40:49 -0700 Subject: [PATCH] [Fix] Guardrails - Ensure Key Guardrails are applied (#16025) * _add_guardrails_from_key_or_team_metadata * test_team_guardrails_append_to_key_guardrails * fix move_guardrails_to_metadata * fix _add_guardrails_from_key_or_team_metadata --- ...odel_prices_and_context_window_backup.json | 30 +++++- litellm/proxy/litellm_pre_call_utils.py | 53 +++++++---- .../proxy/test_litellm_pre_call_utils.py | 94 +++++++++++++++++++ 3 files changed, 156 insertions(+), 21 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8297cc42ff9..0623bc8a904 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -1106,6 +1106,7 @@ "supports_vision": true }, "azure/eu/gpt-4o-2024-08-06": { + "deprecation_date": "2026-02-27", "cache_read_input_token_cost": 1.375e-06, "input_cost_per_token": 2.75e-06, "litellm_provider": "azure", @@ -1122,6 +1123,7 @@ "supports_vision": true }, "azure/eu/gpt-4o-2024-11-20": { + "deprecation_date": "2026-03-01", "cache_creation_input_token_cost": 1.38e-06, "input_cost_per_token": 2.75e-06, "litellm_provider": "azure", @@ -1280,7 +1282,7 @@ }, "azure/global-standard/gpt-4o-2024-08-06": { "cache_read_input_token_cost": 1.25e-06, - "deprecation_date": "2025-08-20", + "deprecation_date": "2026-02-27", "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", "max_input_tokens": 128000, @@ -1297,7 +1299,7 @@ }, "azure/global-standard/gpt-4o-2024-11-20": { "cache_read_input_token_cost": 1.25e-06, - "deprecation_date": "2025-12-20", + "deprecation_date": "2026-03-01", "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", "max_input_tokens": 128000, @@ -1326,6 +1328,7 @@ "supports_vision": true }, "azure/global/gpt-4o-2024-08-06": { + "deprecation_date": "2026-02-27", "cache_read_input_token_cost": 1.25e-06, "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", @@ -1342,6 +1345,7 @@ "supports_vision": true }, "azure/global/gpt-4o-2024-11-20": { + "deprecation_date": "2026-03-01", "cache_read_input_token_cost": 1.25e-06, "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", @@ -1625,6 +1629,7 @@ "supports_web_search": false }, "azure/gpt-4.1-2025-04-14": { + "deprecation_date": "2026-11-04", "cache_read_input_token_cost": 5e-07, "input_cost_per_token": 2e-06, "input_cost_per_token_batches": 1e-06, @@ -1691,6 +1696,7 @@ "supports_web_search": false }, "azure/gpt-4.1-mini-2025-04-14": { + "deprecation_date": "2026-11-04", "cache_read_input_token_cost": 1e-07, "input_cost_per_token": 4e-07, "input_cost_per_token_batches": 2e-07, @@ -1756,6 +1762,7 @@ "supports_vision": true }, "azure/gpt-4.1-nano-2025-04-14": { + "deprecation_date": "2026-11-04", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -1837,6 +1844,7 @@ "supports_vision": true }, "azure/gpt-4o-2024-08-06": { + "deprecation_date": "2026-02-27", "cache_read_input_token_cost": 1.25e-06, "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", @@ -1853,6 +1861,7 @@ "supports_vision": true }, "azure/gpt-4o-2024-11-20": { + "deprecation_date": "2026-03-01", "cache_read_input_token_cost": 1.25e-06, "input_cost_per_token": 2.75e-06, "litellm_provider": "azure", @@ -2604,6 +2613,7 @@ "supports_vision": true }, "azure/o3-2025-04-16": { + "deprecation_date": "2026-04-16", "cache_read_input_token_cost": 2.5e-06, "input_cost_per_token": 1e-05, "litellm_provider": "azure", @@ -2832,6 +2842,7 @@ "output_cost_per_token": 0.0 }, "azure/text-embedding-3-small": { + "deprecation_date": "2026-04-30", "input_cost_per_token": 2e-08, "litellm_provider": "azure", "max_input_tokens": 8191, @@ -2870,6 +2881,7 @@ "mode": "audio_speech" }, "azure/us/gpt-4o-2024-08-06": { + "deprecation_date": "2026-02-27", "cache_read_input_token_cost": 1.375e-06, "input_cost_per_token": 2.75e-06, "litellm_provider": "azure", @@ -2886,6 +2898,7 @@ "supports_vision": true }, "azure/us/gpt-4o-2024-11-20": { + "deprecation_date": "2026-03-01", "cache_creation_input_token_cost": 1.38e-06, "input_cost_per_token": 2.75e-06, "litellm_provider": "azure", @@ -4911,7 +4924,7 @@ "cache_creation_input_token_cost": 3.75e-06, "cache_creation_input_token_cost_above_1hr": 6e-06, "cache_read_input_token_cost": 3e-07, - "deprecation_date": "2026-02-01", + "deprecation_date": "2026-02-19", "input_cost_per_token": 3e-06, "litellm_provider": "anthropic", "max_input_tokens": 200000, @@ -4968,7 +4981,6 @@ "cache_creation_input_token_cost": 3e-07, "cache_creation_input_token_cost_above_1hr": 6e-06, "cache_read_input_token_cost": 3e-08, - "deprecation_date": "2025-03-01", "input_cost_per_token": 2.5e-07, "litellm_provider": "anthropic", "max_input_tokens": 200000, @@ -4988,7 +5000,7 @@ "cache_creation_input_token_cost": 1.875e-05, "cache_creation_input_token_cost_above_1hr": 6e-06, "cache_read_input_token_cost": 1.5e-06, - "deprecation_date": "2025-03-01", + "deprecation_date": "2026-05-01", "input_cost_per_token": 1.5e-05, "litellm_provider": "anthropic", "max_input_tokens": 200000, @@ -5172,6 +5184,7 @@ "cache_creation_input_token_cost_above_1hr": 3e-05, "cache_read_input_token_cost": 1.5e-06, "input_cost_per_token": 1.5e-05, + "deprecation_date": "2026-08-05", "litellm_provider": "anthropic", "max_input_tokens": 200000, "max_output_tokens": 32000, @@ -5199,6 +5212,7 @@ "cache_creation_input_token_cost_above_1hr": 3e-05, "cache_read_input_token_cost": 1.5e-06, "input_cost_per_token": 1.5e-05, + "deprecation_date": "2026-05-14", "litellm_provider": "anthropic", "max_input_tokens": 200000, "max_output_tokens": 32000, @@ -5222,6 +5236,7 @@ "tool_use_system_prompt_tokens": 159 }, "claude-sonnet-4-20250514": { + "deprecation_date": "2026-05-14", "cache_creation_input_token_cost": 3.75e-06, "cache_creation_input_token_cost_above_1hr": 6e-06, "cache_read_input_token_cost": 3e-07, @@ -7993,6 +8008,7 @@ "cache_creation_input_token_cost": 1.375e-06, "cache_read_input_token_cost": 1.1e-07, "input_cost_per_token": 1.1e-06, + "deprecation_date": "2026-10-15", "litellm_provider": "bedrock", "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -12397,6 +12413,7 @@ "supports_tool_choice": true }, "gpt-3.5-turbo-1106": { + "deprecation_date": "2026-09-28", "input_cost_per_token": 1e-06, "litellm_provider": "openai", "max_input_tokens": 16385, @@ -12466,6 +12483,7 @@ "supports_tool_choice": true }, "gpt-4-0125-preview": { + "deprecation_date": "2026-03-26", "input_cost_per_token": 1e-05, "litellm_provider": "openai", "max_input_tokens": 128000, @@ -12506,6 +12524,7 @@ "supports_tool_choice": true }, "gpt-4-1106-preview": { + "deprecation_date": "2026-03-26", "input_cost_per_token": 1e-05, "litellm_provider": "openai", "max_input_tokens": 128000, @@ -16563,6 +16582,7 @@ "supports_vision": true }, "o1-mini-2024-09-12": { + "deprecation_date": "2025-10-27", "cache_read_input_token_cost": 1.5e-06, "input_cost_per_token": 3e-06, "litellm_provider": "openai", diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 1e53655dd0b..7dd5ba2b9ad 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1,7 +1,7 @@ import asyncio import copy import time -from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Set, Union from fastapi import Request from starlette.datastructures import Headers @@ -1191,6 +1191,8 @@ def _add_guardrails_from_key_or_team_metadata( ) -> None: """ Helper add guardrails from key or team metadata to request data + + Key guardrails are set first, then team guardrails are appended (without duplicates). Args: key_metadata: The key metadata dictionary to check for guardrails @@ -1201,14 +1203,24 @@ def _add_guardrails_from_key_or_team_metadata( """ from litellm.proxy.utils import _premium_user_check - for _management_object_metadata in [key_metadata, team_metadata]: - if _management_object_metadata and "guardrails" in _management_object_metadata: - if len(_management_object_metadata["guardrails"]) > 0: - _premium_user_check() - - data[metadata_variable_name]["guardrails"] = _management_object_metadata[ - "guardrails" - ] + # Initialize guardrails set (avoiding duplicates) + combined_guardrails = set() + + # Add key-level guardrails first + if key_metadata and "guardrails" in key_metadata: + if isinstance(key_metadata["guardrails"], list) and len(key_metadata["guardrails"]) > 0: + _premium_user_check() + combined_guardrails.update(key_metadata["guardrails"]) + + # Add team-level guardrails (set automatically handles duplicates) + if team_metadata and "guardrails" in team_metadata: + if isinstance(team_metadata["guardrails"], list) and len(team_metadata["guardrails"]) > 0: + _premium_user_check() + combined_guardrails.update(team_metadata["guardrails"]) + + # Set combined guardrails in metadata as list + if combined_guardrails: + data[metadata_variable_name]["guardrails"] = list(combined_guardrails) def move_guardrails_to_metadata( @@ -1230,15 +1242,24 @@ def move_guardrails_to_metadata( metadata_variable_name=_metadata_variable_name, ) - # Check request-level guardrails + ######################################################################################### + # User's might send "guardrails" in the request body, we need to add them to the request metadata. + # Since downstream logic requires "guardrails" to be in the request metadata + ######################################################################################### if "guardrails" in data: - data[_metadata_variable_name]["guardrails"] = data["guardrails"] - del data["guardrails"] - + request_body_guardrails = data.pop("guardrails") + if "guardrails" in data[_metadata_variable_name] and isinstance(data[_metadata_variable_name]["guardrails"], list): + data[_metadata_variable_name]["guardrails"].extend(request_body_guardrails) + else: + data[_metadata_variable_name]["guardrails"] = request_body_guardrails + + ######################################################################################### if "guardrail_config" in data: - data[_metadata_variable_name]["guardrail_config"] = data["guardrail_config"] - del data["guardrail_config"] - + request_body_guardrail_config = data.pop("guardrail_config") + if "guardrail_config" in data[_metadata_variable_name] and isinstance(data[_metadata_variable_name]["guardrail_config"], dict): + data[_metadata_variable_name]["guardrail_config"].update(request_body_guardrail_config) + else: + data[_metadata_variable_name]["guardrail_config"] = request_body_guardrail_config def add_provider_specific_headers_to_request( data: dict, diff --git a/tests/test_litellm/proxy/test_litellm_pre_call_utils.py b/tests/test_litellm/proxy/test_litellm_pre_call_utils.py index bcc0e5ac2d6..6379ea704ce 100644 --- a/tests/test_litellm/proxy/test_litellm_pre_call_utils.py +++ b/tests/test_litellm/proxy/test_litellm_pre_call_utils.py @@ -1140,3 +1140,97 @@ def test_get_sanitized_user_information_from_key_includes_guardrails_metadata(): assert "guardrails" in result["user_api_key_auth_metadata"] assert result["user_api_key_auth_metadata"]["guardrails"] == ["presidio", "aporia"] assert result["user_api_key_auth_metadata"]["other_field"] == "value" + + +@pytest.mark.asyncio +async def test_team_guardrails_append_to_key_guardrails(): + """ + Test that team guardrails are appended to key guardrails instead of overriding them. + Team guardrails should only be added if they are not already present in key guardrails. + """ + request_mock = MagicMock(spec=Request) + request_mock.url.path = "/chat/completions" + request_mock.url = MagicMock() + request_mock.url.__str__.return_value = "http://localhost/chat/completions" + request_mock.method = "POST" + request_mock.query_params = {} + request_mock.headers = {"Content-Type": "application/json"} + request_mock.client = MagicMock() + request_mock.client.host = "127.0.0.1" + + data = { + "model": "gpt-3.5-turbo", + "messages": [{"role": "user", "content": "test"}], + } + + user_api_key_dict = UserAPIKeyAuth( + api_key="test-key", + metadata={"guardrails": ["key-guardrail-1", "key-guardrail-2"]}, + team_metadata={"guardrails": ["team-guardrail-1", "key-guardrail-1"]}, + ) + + with patch("litellm.proxy.utils._premium_user_check"): + updated_data = await add_litellm_data_to_request( + data=data, + request=request_mock, + user_api_key_dict=user_api_key_dict, + proxy_config=MagicMock(), + general_settings={}, + version="test-version", + ) + + metadata = updated_data.get("metadata", {}) + guardrails = metadata.get("guardrails", []) + + assert "key-guardrail-1" in guardrails + assert "key-guardrail-2" in guardrails + assert "team-guardrail-1" in guardrails + assert guardrails.count("key-guardrail-1") == 1 + + +@pytest.mark.asyncio +async def test_request_guardrails_do_not_override_key_guardrails(): + """ + Test that request-level guardrails do not override key-level guardrails. + + Key guardrails should be preserved when request contains guardrails (including empty array). + """ + request_mock = MagicMock(spec=Request) + request_mock.url.path = "/chat/completions" + request_mock.url = MagicMock() + request_mock.url.__str__.return_value = "http://localhost/chat/completions" + request_mock.method = "POST" + request_mock.query_params = {} + request_mock.headers = {"Content-Type": "application/json"} + request_mock.client = MagicMock() + request_mock.client.host = "127.0.0.1" + + user_api_key_dict = UserAPIKeyAuth( + api_key="test-key", + metadata={"guardrails": ["key-guardrail-1"]}, + team_metadata={}, + ) + + # Test case: Request with empty guardrails should not result in empty guardrails + data_with_empty = { + "model": "gpt-3.5-turbo", + "messages": [{"role": "user", "content": "test"}], + "guardrails": [], + } + + with patch("litellm.proxy.utils._premium_user_check"): + updated_data_empty = await add_litellm_data_to_request( + data=data_with_empty, + request=request_mock, + user_api_key_dict=user_api_key_dict, + proxy_config=MagicMock(), + general_settings={}, + version="test-version", + ) + + _metadata = updated_data_empty.get("metadata", {}) + requested_guardrails = _metadata.get("guardrails", []) + + assert "guardrails" not in updated_data_empty + assert "key-guardrail-1" in requested_guardrails + assert len(requested_guardrails) == 1