mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
[Fix] Guardrails - Ensure Key Guardrails are applied (#16025)
* _add_guardrails_from_key_or_team_metadata * test_team_guardrails_append_to_key_guardrails * fix move_guardrails_to_metadata * fix _add_guardrails_from_key_or_team_metadata
This commit is contained in:
parent
12de66dad6
commit
5c375b23ae
3 changed files with 156 additions and 21 deletions
|
|
@ -1106,6 +1106,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/eu/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2026-02-27",
|
||||
"cache_read_input_token_cost": 1.375e-06,
|
||||
"input_cost_per_token": 2.75e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -1122,6 +1123,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/eu/gpt-4o-2024-11-20": {
|
||||
"deprecation_date": "2026-03-01",
|
||||
"cache_creation_input_token_cost": 1.38e-06,
|
||||
"input_cost_per_token": 2.75e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -1280,7 +1282,7 @@
|
|||
},
|
||||
"azure/global-standard/gpt-4o-2024-08-06": {
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"deprecation_date": "2025-08-20",
|
||||
"deprecation_date": "2026-02-27",
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -1297,7 +1299,7 @@
|
|||
},
|
||||
"azure/global-standard/gpt-4o-2024-11-20": {
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"deprecation_date": "2025-12-20",
|
||||
"deprecation_date": "2026-03-01",
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -1326,6 +1328,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/global/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2026-02-27",
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -1342,6 +1345,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/global/gpt-4o-2024-11-20": {
|
||||
"deprecation_date": "2026-03-01",
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -1625,6 +1629,7 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"azure/gpt-4.1-2025-04-14": {
|
||||
"deprecation_date": "2026-11-04",
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
|
|
@ -1691,6 +1696,7 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"azure/gpt-4.1-mini-2025-04-14": {
|
||||
"deprecation_date": "2026-11-04",
|
||||
"cache_read_input_token_cost": 1e-07,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"input_cost_per_token_batches": 2e-07,
|
||||
|
|
@ -1756,6 +1762,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/gpt-4.1-nano-2025-04-14": {
|
||||
"deprecation_date": "2026-11-04",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"input_cost_per_token_batches": 5e-08,
|
||||
|
|
@ -1837,6 +1844,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2026-02-27",
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -1853,6 +1861,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/gpt-4o-2024-11-20": {
|
||||
"deprecation_date": "2026-03-01",
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"input_cost_per_token": 2.75e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -2604,6 +2613,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/o3-2025-04-16": {
|
||||
"deprecation_date": "2026-04-16",
|
||||
"cache_read_input_token_cost": 2.5e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -2832,6 +2842,7 @@
|
|||
"output_cost_per_token": 0.0
|
||||
},
|
||||
"azure/text-embedding-3-small": {
|
||||
"deprecation_date": "2026-04-30",
|
||||
"input_cost_per_token": 2e-08,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 8191,
|
||||
|
|
@ -2870,6 +2881,7 @@
|
|||
"mode": "audio_speech"
|
||||
},
|
||||
"azure/us/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2026-02-27",
|
||||
"cache_read_input_token_cost": 1.375e-06,
|
||||
"input_cost_per_token": 2.75e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -2886,6 +2898,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"azure/us/gpt-4o-2024-11-20": {
|
||||
"deprecation_date": "2026-03-01",
|
||||
"cache_creation_input_token_cost": 1.38e-06,
|
||||
"input_cost_per_token": 2.75e-06,
|
||||
"litellm_provider": "azure",
|
||||
|
|
@ -4911,7 +4924,7 @@
|
|||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
"deprecation_date": "2026-02-01",
|
||||
"deprecation_date": "2026-02-19",
|
||||
"input_cost_per_token": 3e-06,
|
||||
"litellm_provider": "anthropic",
|
||||
"max_input_tokens": 200000,
|
||||
|
|
@ -4968,7 +4981,6 @@
|
|||
"cache_creation_input_token_cost": 3e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"deprecation_date": "2025-03-01",
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
"litellm_provider": "anthropic",
|
||||
"max_input_tokens": 200000,
|
||||
|
|
@ -4988,7 +5000,7 @@
|
|||
"cache_creation_input_token_cost": 1.875e-05,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 1.5e-06,
|
||||
"deprecation_date": "2025-03-01",
|
||||
"deprecation_date": "2026-05-01",
|
||||
"input_cost_per_token": 1.5e-05,
|
||||
"litellm_provider": "anthropic",
|
||||
"max_input_tokens": 200000,
|
||||
|
|
@ -5172,6 +5184,7 @@
|
|||
"cache_creation_input_token_cost_above_1hr": 3e-05,
|
||||
"cache_read_input_token_cost": 1.5e-06,
|
||||
"input_cost_per_token": 1.5e-05,
|
||||
"deprecation_date": "2026-08-05",
|
||||
"litellm_provider": "anthropic",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 32000,
|
||||
|
|
@ -5199,6 +5212,7 @@
|
|||
"cache_creation_input_token_cost_above_1hr": 3e-05,
|
||||
"cache_read_input_token_cost": 1.5e-06,
|
||||
"input_cost_per_token": 1.5e-05,
|
||||
"deprecation_date": "2026-05-14",
|
||||
"litellm_provider": "anthropic",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 32000,
|
||||
|
|
@ -5222,6 +5236,7 @@
|
|||
"tool_use_system_prompt_tokens": 159
|
||||
},
|
||||
"claude-sonnet-4-20250514": {
|
||||
"deprecation_date": "2026-05-14",
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -7993,6 +8008,7 @@
|
|||
"cache_creation_input_token_cost": 1.375e-06,
|
||||
"cache_read_input_token_cost": 1.1e-07,
|
||||
"input_cost_per_token": 1.1e-06,
|
||||
"deprecation_date": "2026-10-15",
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 8192,
|
||||
|
|
@ -12397,6 +12413,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-3.5-turbo-1106": {
|
||||
"deprecation_date": "2026-09-28",
|
||||
"input_cost_per_token": 1e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 16385,
|
||||
|
|
@ -12466,6 +12483,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-4-0125-preview": {
|
||||
"deprecation_date": "2026-03-26",
|
||||
"input_cost_per_token": 1e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -12506,6 +12524,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-4-1106-preview": {
|
||||
"deprecation_date": "2026-03-26",
|
||||
"input_cost_per_token": 1e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -16563,6 +16582,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"o1-mini-2024-09-12": {
|
||||
"deprecation_date": "2025-10-27",
|
||||
"cache_read_input_token_cost": 1.5e-06,
|
||||
"input_cost_per_token": 3e-06,
|
||||
"litellm_provider": "openai",
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
import asyncio
|
||||
import copy
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Set, Union
|
||||
|
||||
from fastapi import Request
|
||||
from starlette.datastructures import Headers
|
||||
|
|
@ -1191,6 +1191,8 @@ def _add_guardrails_from_key_or_team_metadata(
|
|||
) -> None:
|
||||
"""
|
||||
Helper add guardrails from key or team metadata to request data
|
||||
|
||||
Key guardrails are set first, then team guardrails are appended (without duplicates).
|
||||
|
||||
Args:
|
||||
key_metadata: The key metadata dictionary to check for guardrails
|
||||
|
|
@ -1201,14 +1203,24 @@ def _add_guardrails_from_key_or_team_metadata(
|
|||
"""
|
||||
from litellm.proxy.utils import _premium_user_check
|
||||
|
||||
for _management_object_metadata in [key_metadata, team_metadata]:
|
||||
if _management_object_metadata and "guardrails" in _management_object_metadata:
|
||||
if len(_management_object_metadata["guardrails"]) > 0:
|
||||
_premium_user_check()
|
||||
|
||||
data[metadata_variable_name]["guardrails"] = _management_object_metadata[
|
||||
"guardrails"
|
||||
]
|
||||
# Initialize guardrails set (avoiding duplicates)
|
||||
combined_guardrails = set()
|
||||
|
||||
# Add key-level guardrails first
|
||||
if key_metadata and "guardrails" in key_metadata:
|
||||
if isinstance(key_metadata["guardrails"], list) and len(key_metadata["guardrails"]) > 0:
|
||||
_premium_user_check()
|
||||
combined_guardrails.update(key_metadata["guardrails"])
|
||||
|
||||
# Add team-level guardrails (set automatically handles duplicates)
|
||||
if team_metadata and "guardrails" in team_metadata:
|
||||
if isinstance(team_metadata["guardrails"], list) and len(team_metadata["guardrails"]) > 0:
|
||||
_premium_user_check()
|
||||
combined_guardrails.update(team_metadata["guardrails"])
|
||||
|
||||
# Set combined guardrails in metadata as list
|
||||
if combined_guardrails:
|
||||
data[metadata_variable_name]["guardrails"] = list(combined_guardrails)
|
||||
|
||||
|
||||
def move_guardrails_to_metadata(
|
||||
|
|
@ -1230,15 +1242,24 @@ def move_guardrails_to_metadata(
|
|||
metadata_variable_name=_metadata_variable_name,
|
||||
)
|
||||
|
||||
# Check request-level guardrails
|
||||
#########################################################################################
|
||||
# User's might send "guardrails" in the request body, we need to add them to the request metadata.
|
||||
# Since downstream logic requires "guardrails" to be in the request metadata
|
||||
#########################################################################################
|
||||
if "guardrails" in data:
|
||||
data[_metadata_variable_name]["guardrails"] = data["guardrails"]
|
||||
del data["guardrails"]
|
||||
|
||||
request_body_guardrails = data.pop("guardrails")
|
||||
if "guardrails" in data[_metadata_variable_name] and isinstance(data[_metadata_variable_name]["guardrails"], list):
|
||||
data[_metadata_variable_name]["guardrails"].extend(request_body_guardrails)
|
||||
else:
|
||||
data[_metadata_variable_name]["guardrails"] = request_body_guardrails
|
||||
|
||||
#########################################################################################
|
||||
if "guardrail_config" in data:
|
||||
data[_metadata_variable_name]["guardrail_config"] = data["guardrail_config"]
|
||||
del data["guardrail_config"]
|
||||
|
||||
request_body_guardrail_config = data.pop("guardrail_config")
|
||||
if "guardrail_config" in data[_metadata_variable_name] and isinstance(data[_metadata_variable_name]["guardrail_config"], dict):
|
||||
data[_metadata_variable_name]["guardrail_config"].update(request_body_guardrail_config)
|
||||
else:
|
||||
data[_metadata_variable_name]["guardrail_config"] = request_body_guardrail_config
|
||||
|
||||
def add_provider_specific_headers_to_request(
|
||||
data: dict,
|
||||
|
|
|
|||
|
|
@ -1140,3 +1140,97 @@ def test_get_sanitized_user_information_from_key_includes_guardrails_metadata():
|
|||
assert "guardrails" in result["user_api_key_auth_metadata"]
|
||||
assert result["user_api_key_auth_metadata"]["guardrails"] == ["presidio", "aporia"]
|
||||
assert result["user_api_key_auth_metadata"]["other_field"] == "value"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_team_guardrails_append_to_key_guardrails():
|
||||
"""
|
||||
Test that team guardrails are appended to key guardrails instead of overriding them.
|
||||
Team guardrails should only be added if they are not already present in key guardrails.
|
||||
"""
|
||||
request_mock = MagicMock(spec=Request)
|
||||
request_mock.url.path = "/chat/completions"
|
||||
request_mock.url = MagicMock()
|
||||
request_mock.url.__str__.return_value = "http://localhost/chat/completions"
|
||||
request_mock.method = "POST"
|
||||
request_mock.query_params = {}
|
||||
request_mock.headers = {"Content-Type": "application/json"}
|
||||
request_mock.client = MagicMock()
|
||||
request_mock.client.host = "127.0.0.1"
|
||||
|
||||
data = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "test"}],
|
||||
}
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key="test-key",
|
||||
metadata={"guardrails": ["key-guardrail-1", "key-guardrail-2"]},
|
||||
team_metadata={"guardrails": ["team-guardrail-1", "key-guardrail-1"]},
|
||||
)
|
||||
|
||||
with patch("litellm.proxy.utils._premium_user_check"):
|
||||
updated_data = await add_litellm_data_to_request(
|
||||
data=data,
|
||||
request=request_mock,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
proxy_config=MagicMock(),
|
||||
general_settings={},
|
||||
version="test-version",
|
||||
)
|
||||
|
||||
metadata = updated_data.get("metadata", {})
|
||||
guardrails = metadata.get("guardrails", [])
|
||||
|
||||
assert "key-guardrail-1" in guardrails
|
||||
assert "key-guardrail-2" in guardrails
|
||||
assert "team-guardrail-1" in guardrails
|
||||
assert guardrails.count("key-guardrail-1") == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_request_guardrails_do_not_override_key_guardrails():
|
||||
"""
|
||||
Test that request-level guardrails do not override key-level guardrails.
|
||||
|
||||
Key guardrails should be preserved when request contains guardrails (including empty array).
|
||||
"""
|
||||
request_mock = MagicMock(spec=Request)
|
||||
request_mock.url.path = "/chat/completions"
|
||||
request_mock.url = MagicMock()
|
||||
request_mock.url.__str__.return_value = "http://localhost/chat/completions"
|
||||
request_mock.method = "POST"
|
||||
request_mock.query_params = {}
|
||||
request_mock.headers = {"Content-Type": "application/json"}
|
||||
request_mock.client = MagicMock()
|
||||
request_mock.client.host = "127.0.0.1"
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key="test-key",
|
||||
metadata={"guardrails": ["key-guardrail-1"]},
|
||||
team_metadata={},
|
||||
)
|
||||
|
||||
# Test case: Request with empty guardrails should not result in empty guardrails
|
||||
data_with_empty = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "test"}],
|
||||
"guardrails": [],
|
||||
}
|
||||
|
||||
with patch("litellm.proxy.utils._premium_user_check"):
|
||||
updated_data_empty = await add_litellm_data_to_request(
|
||||
data=data_with_empty,
|
||||
request=request_mock,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
proxy_config=MagicMock(),
|
||||
general_settings={},
|
||||
version="test-version",
|
||||
)
|
||||
|
||||
_metadata = updated_data_empty.get("metadata", {})
|
||||
requested_guardrails = _metadata.get("guardrails", [])
|
||||
|
||||
assert "guardrails" not in updated_data_empty
|
||||
assert "key-guardrail-1" in requested_guardrails
|
||||
assert len(requested_guardrails) == 1
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue