Merge branch 'main' into litellm_litellm_anthropic_remote_url3

This commit is contained in:
Sameer Kankute 2026-02-13 18:40:00 +05:30 • committed by GitHub
commit 4a9d851dc4
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
53 changed files with 4013 additions and 467 deletions

View file

@ -40,38 +40,33 @@ outputs:
runs:
using: composite
steps:
- name: Helm | Setup
uses: azure/setup-helm@v4
with:
version: v3.20.0
- name: Helm | Login
shell: bash
run: echo ${{ inputs.registry_password }} | helm registry login -u ${{ inputs.registry_username }} --password-stdin ${{ inputs.registry }}
env:
HELM_EXPERIMENTAL_OCI: '1'
- name: Helm | Dependency
if: inputs.update_dependencies == 'true'
shell: bash
run: helm dependency update ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
env:
HELM_EXPERIMENTAL_OCI: '1'
- name: Helm | Package
shell: bash
run: helm package ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} --version ${{ inputs.tag }} --app-version ${{ inputs.app_version }}
env:
HELM_EXPERIMENTAL_OCI: '1'
- name: Helm | Push
shell: bash
run: helm push ${{ inputs.name }}-${{ inputs.tag }}.tgz oci://${{ inputs.registry }}/${{ inputs.repository }}
env:
HELM_EXPERIMENTAL_OCI: '1'
- name: Helm | Logout
shell: bash
run: helm registry logout ${{ inputs.registry }}
env:
HELM_EXPERIMENTAL_OCI: '1'
- name: Helm | Output
id: output
shell: bash
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT

View file

@ -26,6 +26,10 @@ version: 1.1.0
# It is recommended to use it with quotes.
appVersion: v1.80.12
annotations:
org.opencontainers.image.source: "https://github.com/BerriAI/litellm"
org.opencontainers.image.url: "https://docs.litellm.ai/"
dependencies:
- name: "postgresql"
version: ">=13.3.0"

View file

@ -18,7 +18,7 @@ Each provider uses their own search backend:
| Provider | Search Engine | Notes |
|----------|---------------|-------|
| **OpenAI** (`gpt-4o-search-preview`, `gpt-4o-mini-search-preview`, `gpt-5-search-api`) | OpenAI's internal search | Real-time web data |
| **OpenAI** (`gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview`) | OpenAI's internal search | Real-time web data |
| **xAI** (`grok-3`) | xAI's search + X/Twitter | Real-time social media data |
| **Google AI/Vertex** (`gemini-2.0-flash`) | **Google Search** | Uses actual Google search results |
| **Anthropic** (`claude-3-5-sonnet`) | Anthropic's web search | Real-time web data |
@ -45,6 +45,19 @@ Use `web_search_options` when you need to:
**Anthropic Web Search Models**: Claude models that support web search: `claude-3-5-sonnet-latest`, `claude-3-5-sonnet-20241022`, `claude-3-5-haiku-latest`, `claude-3-5-haiku-20241022`, `claude-3-7-sonnet-20250219`
:::
## OpenAI Web Search: Two Approaches
OpenAI offers two distinct ways to use web search depending on the endpoint and model:
| Approach | Endpoint | Models | How to enable |
|----------|----------|--------|---------------|
| **Search Models** | `/chat/completions` | `gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview` | Pass `web_search_options` parameter |
| **Web Search Tool** | `/responses` | `gpt-5`, `gpt-4.1`, `gpt-4o`, and other regular models | Pass `web_search_preview` tool |
:::tip Search models search automatically
Search models like `gpt-5-search-api` **automatically search the web** even without the `web_search_options` parameter. Use `web_search_options` to set `search_context_size` (`"low"`, `"medium"`, `"high"`) or specify `user_location` for localized results.
:::
## `/chat/completions` (litellm.completion)
### Quick Start
@ -56,7 +69,7 @@ Use `web_search_options` when you need to:
from litellm import completion
response = completion(
model="openai/gpt-4o-search-preview",
model="openai/gpt-5-search-api",
messages=[
{
"role": "user",
@ -76,31 +89,36 @@ response = completion(
```yaml
model_list:
# OpenAI
# OpenAI search models
- model_name: gpt-5-search-api
litellm_params:
model: openai/gpt-5-search-api
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-4o-search-preview
litellm_params:
model: openai/gpt-4o-search-preview
api_key: os.environ/OPENAI_API_KEY
# xAI
- model_name: grok-3
litellm_params:
model: xai/grok-3
api_key: os.environ/XAI_API_KEY
# Anthropic
- model_name: claude-3-5-sonnet-latest
litellm_params:
model: anthropic/claude-3-5-sonnet-latest
api_key: os.environ/ANTHROPIC_API_KEY
# VertexAI
- model_name: gemini-2-flash
litellm_params:
model: gemini-2.0-flash
vertex_project: your-project-id
vertex_location: us-central1
# Google AI Studio
- model_name: gemini-2-flash-studio
litellm_params:
@ -108,13 +126,13 @@ model_list:
api_key: os.environ/GOOGLE_API_KEY
```
2. Start the proxy
2. Start the proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
3. Test it!
```python showLineNumbers
from openai import OpenAI
@ -126,13 +144,18 @@ client = OpenAI(
)
response = client.chat.completions.create(
model="grok-3", # or any other web search enabled model
model="gpt-5-search-api", # or any other web search enabled model
messages=[
{
"role": "user",
"content": "What was a positive news story from today?"
}
]
],
extra_body={
"web_search_options": {
"search_context_size": "medium"
}
}
)
```
</TabItem>
@ -149,7 +172,7 @@ from litellm import completion
# Customize search context size
response = completion(
model="openai/gpt-4o-search-preview",
model="openai/gpt-5-search-api",
messages=[
{
"role": "user",
@ -257,6 +280,12 @@ response = client.chat.completions.create(
## `/responses` (litellm.responses)
Use the `web_search_preview` tool with models like `gpt-5`, `gpt-4.1`, `gpt-4o`, etc.
:::info
Search-dedicated models like `gpt-5-search-api` and `gpt-4o-search-preview` do **not** support the `/responses` endpoint. Use them with `/chat/completions` + `web_search_options` instead (see above).
:::
### Quick Start
<Tabs>
@ -266,18 +295,14 @@ response = client.chat.completions.create(
from litellm import responses
response = responses(
model="openai/gpt-4o",
input=[
{
"role": "user",
"content": "What was a positive news story from today?"
}
],
model="openai/gpt-5",
input="What is the capital of France?",
tools=[{
"type": "web_search_preview" # enables web search with default medium context size
}]
)
```
</TabItem>
<TabItem value="proxy" label="PROXY">
@ -285,19 +310,24 @@ response = responses(
```yaml
model_list:
- model_name: gpt-4o
- model_name: gpt-5
litellm_params:
model: openai/gpt-4o
model: openai/gpt-5
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-4.1
litellm_params:
model: openai/gpt-4.1
api_key: os.environ/OPENAI_API_KEY
```
2. Start the proxy
2. Start the proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
3. Test it!
```python showLineNumbers
from openai import OpenAI
@ -309,11 +339,11 @@ client = OpenAI(
)
response = client.responses.create(
model="gpt-4o",
model="gpt-5",
tools=[{
"type": "web_search_preview"
}],
input="What was a positive news story from today?",
input="What is the capital of France?",
)
print(response.output_text)
@ -331,13 +361,8 @@ from litellm import responses
# Customize search context size
response = responses(
model="openai/gpt-4o",
input=[
{
"role": "user",
"content": "What was a positive news story from today?"
}
],
model="openai/gpt-5",
input="What is the capital of France?",
tools=[{
"type": "web_search_preview",
"search_context_size": "low" # Options: "low", "medium" (default), "high"
@ -358,12 +383,12 @@ client = OpenAI(
# Customize search context size
response = client.responses.create(
model="gpt-4o",
model="gpt-5",
tools=[{
"type": "web_search_preview",
"search_context_size": "low" # Options: "low", "medium" (default), "high"
}],
input="What was a positive news story from today?",
input="What is the capital of France?",
)
print(response.output_text)
@ -417,14 +442,14 @@ model_list:
web_search_options:
search_context_size: "high" # Options: "low", "medium", "high"
# Different context size for different models
- model_name: gpt-4o-search-preview
# OpenAI search model with custom context size
- model_name: gpt-5-search-api
litellm_params:
model: openai/gpt-4o-search-preview
model: openai/gpt-5-search-api
api_key: os.environ/OPENAI_API_KEY
web_search_options:
search_context_size: "low"
# Gemini with medium context (default)
- model_name: gemini-2-flash
litellm_params:
@ -449,6 +474,7 @@ Use `litellm.supports_web_search(model="model_name")` -> returns `True` if model
```python showLineNumbers
# Check OpenAI models
assert litellm.supports_web_search(model="openai/gpt-5-search-api") == True
assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
# Check xAI models
@ -472,13 +498,20 @@ assert litellm.supports_web_search(model="gemini/gemini-2.0-flash") == True
```yaml
model_list:
# OpenAI
- model_name: gpt-5-search-api
litellm_params:
model: openai/gpt-5-search-api
api_key: os.environ/OPENAI_API_KEY
model_info:
supports_web_search: True
- model_name: gpt-4o-search-preview
litellm_params:
model: openai/gpt-4o-search-preview
api_key: os.environ/OPENAI_API_KEY
model_info:
supports_web_search: True
# xAI
- model_name: grok-3
litellm_params:
@ -533,6 +566,12 @@ Expected Response
```json showLineNumbers
{
"data": [
{
"model_group": "gpt-5-search-api",
"providers": ["openai"],
"max_tokens": 128000,
"supports_web_search": true
},
{
"model_group": "gpt-4o-search-preview",
"providers": ["openai"],

View file

@ -230,7 +230,70 @@ os.environ["OPENAI_BASE_URL"] = "https://your_host/v1" # OPTIONAL
These also support the `OPENAI_BASE_URL` environment variable, which can be used to specify a custom API endpoint.
## OpenAI Vision Models
### OpenAI Web Search Models
OpenAI has two ways to use web search, depending on the endpoint:
| Approach | Endpoint | Models | How to enable |
|----------|----------|--------|---------------|
| **Search Models** | `/chat/completions` | `gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview` | Pass `web_search_options` parameter |
| **Web Search Tool** | `/responses` | `gpt-5`, `gpt-4.1`, `gpt-4o`, and other regular models | Pass `web_search_preview` tool |
<Tabs>
<TabItem value="sdk-completion" label="SDK - /chat/completions">
```python showLineNumbers
from litellm import completion
response = completion(
model="openai/gpt-5-search-api",
messages=[{"role": "user", "content": "What is the capital of France?"}],
web_search_options={
"search_context_size": "medium" # Options: "low", "medium", "high"
}
)
```
</TabItem>
<TabItem value="sdk-responses" label="SDK - /responses">
```python showLineNumbers
from litellm import responses
response = responses(
model="openai/gpt-5",
input="What is the capital of France?",
tools=[{
"type": "web_search_preview",
"search_context_size": "low"
}]
)
```
</TabItem>
<TabItem value="proxy" label="PROXY">
```yaml
model_list:
# Search model for /chat/completions
- model_name: gpt-5-search-api
litellm_params:
model: openai/gpt-5-search-api
api_key: os.environ/OPENAI_API_KEY
# Regular model for /responses with web_search_preview tool
- model_name: gpt-5
litellm_params:
model: openai/gpt-5
api_key: os.environ/OPENAI_API_KEY
```
</TabItem>
</Tabs>
For full details, see the [Web Search guide](../completion/web_search.md).
## OpenAI Vision Models
| Model Name | Function Call |
|-----------------------|-----------------------------------------------------------------|
| gpt-4o | `response = completion(model="gpt-4o", messages=messages)` |

View file

@ -37,6 +37,24 @@ for event in response:
print(event)
```
#### Web Search
```python showLineNumbers title="OpenAI Responses with Web Search"
import litellm
response = litellm.responses(
model="openai/gpt-5",
input="What is the capital of France?",
tools=[{
"type": "web_search_preview",
"search_context_size": "medium" # Options: "low", "medium", "high"
}]
)
print(response)
```
For full details, see the [Web Search guide](../../completion/web_search.md).
#### Image Generation with Streaming
```python showLineNumbers title="OpenAI Streaming Image Generation"
import litellm

View file

@ -41,6 +41,10 @@ class EnterpriseRouteChecks:
return get_secret_bool("DISABLE_ADMIN_ENDPOINTS") is True
# Routes that should remain accessible even when LLM API endpoints are disabled.
# These are read-only model listing routes needed by the Admin UI.
LLM_API_EXEMPT_ROUTES = ["/models", "/v1/models"]
@staticmethod
def should_call_route(route: str):
"""
@ -58,6 +62,7 @@ class EnterpriseRouteChecks:
)
elif (
RouteChecks.is_llm_api_route(route=route)
and route not in EnterpriseRouteChecks.LLM_API_EXEMPT_ROUTES
and EnterpriseRouteChecks.is_llm_api_route_disabled()
):
raise HTTPException(

View file

@ -19,6 +19,7 @@
"mcp-client-2025-11-20": "mcp-client-2025-11-20",
"mcp-client-2025-04-04": "mcp-client-2025-04-04",
"mcp-servers-2025-12-04": null,
"oauth-2025-04-20": "oauth-2025-04-20",
"output-128k-2025-02-19": "output-128k-2025-02-19",
"prompt-caching-scope-2026-01-05": "prompt-caching-scope-2026-01-05",
"skills-2025-10-02": "skills-2025-10-02",

View file

@ -12,7 +12,8 @@ import asyncio
import time
import traceback
from concurrent.futures import ThreadPoolExecutor
from typing import TYPE_CHECKING, Any, List, Optional, Union
from threading import Lock
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union
if TYPE_CHECKING:
from litellm.types.caching import RedisPipelineIncrementOperation
@ -71,6 +72,7 @@ class DualCache(BaseCache):
self.last_redis_batch_access_time = LimitedSizeOrderedDict(
max_size=default_max_redis_batch_cache_size
)
self._last_redis_batch_access_time_lock = Lock()
self.redis_batch_cache_expiry = (
default_redis_batch_cache_expiry
or litellm.default_redis_batch_cache_expiry
@ -236,22 +238,46 @@ class DualCache(BaseCache):
except Exception:
verbose_logger.error(traceback.format_exc())
def get_redis_batch_keys(
def _reserve_redis_batch_keys(
self,
current_time: float,
keys: List[str],
result: List[Any],
) -> List[str]:
sublist_keys = []
for key, value in zip(keys, result):
if value is None:
) -> Tuple[List[str], Dict[str, Optional[float]]]:
"""
Atomically choose keys to fetch from Redis and reserve their access time.
This prevents check-then-act races under concurrent async callers.
"""
sublist_keys: List[str] = []
previous_access_times: Dict[str, Optional[float]] = {}
with self._last_redis_batch_access_time_lock:
for key, value in zip(keys, result):
if value is not None:
continue
if (
key not in self.last_redis_batch_access_time
or current_time - self.last_redis_batch_access_time[key]
>= self.redis_batch_cache_expiry
):
sublist_keys.append(key)
return sublist_keys
previous_access_times[key] = self.last_redis_batch_access_time.get(
key
)
self.last_redis_batch_access_time[key] = current_time
return sublist_keys, previous_access_times
def _rollback_redis_batch_key_reservations(
self, previous_access_times: Dict[str, Optional[float]]
) -> None:
with self._last_redis_batch_access_time_lock:
for key, previous_time in previous_access_times.items():
if previous_time is None:
self.last_redis_batch_access_time.pop(key, None)
else:
self.last_redis_batch_access_time[key] = previous_time
async def async_batch_get_cache(
self,
@ -276,19 +302,23 @@ class DualCache(BaseCache):
- check the redis cache
"""
current_time = time.time()
sublist_keys = self.get_redis_batch_keys(current_time, keys, result)
sublist_keys, previous_access_times = self._reserve_redis_batch_keys(
current_time, keys, result
)
# Only hit Redis if the last access time was more than 5 seconds ago
# Only hit Redis if enough time has passed since last access.
if len(sublist_keys) > 0:
# If not found in in-memory cache, try fetching from Redis
redis_result = await self.redis_cache.async_batch_get_cache(
sublist_keys, parent_otel_span=parent_otel_span
)
# Update the last access time for ALL queried keys
# This includes keys with None values to throttle repeated Redis queries
for key in sublist_keys:
self.last_redis_batch_access_time[key] = current_time
try:
# If not found in in-memory cache, try fetching from Redis
redis_result = await self.redis_cache.async_batch_get_cache(
sublist_keys, parent_otel_span=parent_otel_span
)
except Exception:
# Do not throttle subsequent callers if the Redis read fails.
self._rollback_redis_batch_key_reservations(
previous_access_times
)
raise
# Short-circuit if redis_result is None or contains only None values
if redis_result is None or all(v is None for v in redis_result.values()):

View file

@ -1016,10 +1016,12 @@ BEDROCK_EMBEDDING_PROVIDERS_LITERAL = Literal[
BEDROCK_CONVERSE_MODELS = [
"qwen.qwen3-coder-480b-a35b-v1:0",
"qwen.qwen3-coder-next",
"qwen.qwen3-235b-a22b-2507-v1:0",
"qwen.qwen3-coder-30b-a3b-v1:0",
"qwen.qwen3-32b-v1:0",
"deepseek.v3-v1:0",
"deepseek.v3.2",
"openai.gpt-oss-20b-1:0",
"openai.gpt-oss-120b-1:0",
"anthropic.claude-haiku-4-5-20251001-v1:0",
@ -1062,6 +1064,8 @@ BEDROCK_CONVERSE_MODELS = [
"amazon.nova-pro-v1:0",
"writer.palmyra-x4-v1:0",
"writer.palmyra-x5-v1:0",
"minimax.minimax-m2.1",
"moonshotai.kimi-k2.5",
]

View file

@ -29,6 +29,7 @@ from litellm.types.utils import (
LLMResponseTypes,
StandardLoggingGuardrailInformation,
)
from fastapi.exceptions import HTTPException
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
@ -650,6 +651,23 @@ class CustomGuardrail(CustomLogger):
)
return response
@staticmethod
def _is_guardrail_intervention(e: Exception) -> bool:
"""
Returns True if the exception represents an intentional guardrail block
(this was logged previously as an API failure - guardrail_failed_to_respond).
Guardrails signal intentional blocks by raising:
- HTTPException with status 400 (content policy violation)
- ModifyResponseException (passthrough mode violation)
"""
if isinstance(e, ModifyResponseException):
return True
if isinstance(e, HTTPException) and e.status_code == 400:
return True
return False
def _process_error(
self,
e: Exception,
@ -664,6 +682,11 @@ class CustomGuardrail(CustomLogger):
This gets logged on downsteam Langfuse, DataDog, etc.
"""
guardrail_status: GuardrailStatus = (
"guardrail_intervened"
if self._is_guardrail_intervention(e)
else "guardrail_failed_to_respond"
)
# For custom_code_guardrail scenario, log as "deny" instead of full exception
# Check if this is from custom_code_guardrail by checking the class name
guardrail_response: Union[Exception, str] = e
@ -673,7 +696,7 @@ class CustomGuardrail(CustomLogger):
self.add_standard_logging_guardrail_information_to_request_data(
guardrail_json_response=guardrail_response,
request_data=request_data,
guardrail_status="guardrail_failed_to_respond",
guardrail_status=guardrail_status,
duration=duration,
start_time=start_time,
end_time=end_time,

View file

@ -2331,7 +2331,7 @@ class Logging(LiteLLMLoggingBaseClass):
result, LiteLLMBatch
):
litellm_params = self.litellm_params or {}
litellm_metadata = litellm_params.get("litellm_metadata", {})
litellm_metadata = litellm_params.get("litellm_metadata") or {}
if (
litellm_metadata.get("batch_ignore_default_logging", False) is True
): # polling job will query these frequently, don't spam db logs
@ -3127,7 +3127,7 @@ class Logging(LiteLLMLoggingBaseClass):
self, dynamic_success_callbacks: Optional[List], global_callbacks: List
) -> List:
if dynamic_success_callbacks is None:
return global_callbacks
return list(global_callbacks)
return list(set(dynamic_success_callbacks + global_callbacks))
def _remove_internal_litellm_callbacks(self, callbacks: List) -> List:

View file

@ -38,9 +38,18 @@ def optionally_handle_anthropic_oauth(
Returns:
Tuple of (updated headers, api_key)
"""
# Check Authorization header (passthrough / forwarded requests)
auth_header = headers.get("authorization", "")
if auth_header and auth_header.startswith(f"Bearer {ANTHROPIC_OAUTH_TOKEN_PREFIX}"):
api_key = auth_header.replace("Bearer ", "")
headers.pop("x-api-key", None)
headers["anthropic-beta"] = ANTHROPIC_OAUTH_BETA_HEADER
headers["anthropic-dangerous-direct-browser-access"] = "true"
return headers, api_key
# Check api_key directly (standard chat/completion flow)
if api_key and api_key.startswith(ANTHROPIC_OAUTH_TOKEN_PREFIX):
headers.pop("x-api-key", None)
headers["authorization"] = f"Bearer {api_key}"
headers["anthropic-beta"] = ANTHROPIC_OAUTH_BETA_HEADER
headers["anthropic-dangerous-direct-browser-access"] = "true"
return headers, api_key
@ -108,7 +117,9 @@ class AnthropicModelInfo(BaseLLMModelInfo):
if tools is None:
return False
for tool in tools:
if "type" in tool and tool["type"].startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value):
if "type" in tool and tool["type"].startswith(
ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value
):
return True
return False
@ -134,111 +145,126 @@ class AnthropicModelInfo(BaseLLMModelInfo):
"""
if not tools:
return False
for tool in tools:
tool_type = tool.get("type", "")
if tool_type in ["tool_search_tool_regex_20251119", "tool_search_tool_bm25_20251119"]:
if tool_type in [
"tool_search_tool_regex_20251119",
"tool_search_tool_bm25_20251119",
]:
return True
return False
def is_programmatic_tool_calling_used(self, tools: Optional[List]) -> bool:
"""
Check if programmatic tool calling is being used (tools with allowed_callers field).
Returns True if any tool has allowed_callers containing 'code_execution_20250825'.
"""
if not tools:
return False
for tool in tools:
# Check top-level allowed_callers
allowed_callers = tool.get("allowed_callers", None)
if allowed_callers and isinstance(allowed_callers, list):
if "code_execution_20250825" in allowed_callers:
return True
# Check function.allowed_callers for OpenAI format tools
function = tool.get("function", {})
if isinstance(function, dict):
function_allowed_callers = function.get("allowed_callers", None)
if function_allowed_callers and isinstance(function_allowed_callers, list):
if function_allowed_callers and isinstance(
function_allowed_callers, list
):
if "code_execution_20250825" in function_allowed_callers:
return True
return False
def is_input_examples_used(self, tools: Optional[List]) -> bool:
"""
Check if input_examples is being used in any tools.
Returns True if any tool has input_examples field.
"""
if not tools:
return False
for tool in tools:
# Check top-level input_examples
input_examples = tool.get("input_examples", None)
if input_examples and isinstance(input_examples, list) and len(input_examples) > 0:
if (
input_examples
and isinstance(input_examples, list)
and len(input_examples) > 0
):
return True
# Check function.input_examples for OpenAI format tools
function = tool.get("function", {})
if isinstance(function, dict):
function_input_examples = function.get("input_examples", None)
if function_input_examples and isinstance(function_input_examples, list) and len(function_input_examples) > 0:
if (
function_input_examples
and isinstance(function_input_examples, list)
and len(function_input_examples) > 0
):
return True
return False
def is_effort_used(self, optional_params: Optional[dict], model: Optional[str] = None) -> bool:
def is_effort_used(
self, optional_params: Optional[dict], model: Optional[str] = None
) -> bool:
"""
Check if effort parameter is being used.
Returns True if effort-related parameters are present.
"""
if not optional_params:
return False
# Check if reasoning_effort is provided for Claude Opus 4.5
if model and ("opus-4-5" in model.lower() or "opus_4_5" in model.lower()):
reasoning_effort = optional_params.get("reasoning_effort")
if reasoning_effort and isinstance(reasoning_effort, str):
return True
# Check if output_config is directly provided
output_config = optional_params.get("output_config")
if output_config and isinstance(output_config, dict):
effort = output_config.get("effort")
if effort and isinstance(effort, str):
return True
return False
def is_code_execution_tool_used(self, tools: Optional[List]) -> bool:
"""
Check if code execution tool is being used.
Returns True if any tool has type "code_execution_20250825".
"""
if not tools:
return False
for tool in tools:
tool_type = tool.get("type", "")
if tool_type == "code_execution_20250825":
return True
return False
def is_container_with_skills_used(self, optional_params: Optional[dict]) -> bool:
"""
Check if container with skills is being used.
Returns True if optional_params contains container with skills.
"""
if not optional_params:
return False
container = optional_params.get("container")
if container and isinstance(container, dict):
skills = container.get("skills")
@ -256,10 +282,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
def get_computer_tool_beta_header(self, computer_tool_version: str) -> str:
"""
Get the appropriate beta header for a given computer tool version.
Args:
computer_tool_version: The computer tool version (e.g., 'computer_20250124', 'computer_20241022')
Returns:
The corresponding beta header string
"""
@ -282,37 +308,37 @@ class AnthropicModelInfo(BaseLLMModelInfo):
) -> List[str]:
"""
Get list of common beta headers based on the features that are active.
Returns:
List of beta header strings
"""
from litellm.types.llms.anthropic import (
ANTHROPIC_EFFORT_BETA_HEADER,
)
betas = []
# Detect features
effort_used = self.is_effort_used(optional_params, model)
if effort_used:
betas.append(ANTHROPIC_EFFORT_BETA_HEADER) # effort-2025-11-24
if computer_tool_used:
beta_header = self.get_computer_tool_beta_header(computer_tool_used)
betas.append(beta_header)
# Anthropic no longer requires the prompt-caching beta header
# Prompt caching now works automatically when cache_control is used in messages
# Reference: https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
if file_id_used:
betas.append("files-api-2025-04-14")
betas.append("code-execution-2025-05-22")
if mcp_server_used:
betas.append("mcp-client-2025-04-04")
return list(set(betas))
def get_anthropic_headers(
@ -351,27 +377,35 @@ class AnthropicModelInfo(BaseLLMModelInfo):
# Tool search, programmatic tool calling, and input_examples all use the same beta header
if tool_search_used or programmatic_tool_calling_used or input_examples_used:
from litellm.types.llms.anthropic import ANTHROPIC_TOOL_SEARCH_BETA_HEADER
betas.add(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
# Effort parameter uses a separate beta header
if effort_used:
from litellm.types.llms.anthropic import ANTHROPIC_EFFORT_BETA_HEADER
betas.add(ANTHROPIC_EFFORT_BETA_HEADER)
# Code execution tool uses a separate beta header
if code_execution_tool_used:
betas.add("code-execution-2025-08-25")
# Container with skills uses a separate beta header
if container_with_skills_used:
betas.add("skills-2025-10-02")
_is_oauth = api_key and api_key.startswith(ANTHROPIC_OAUTH_TOKEN_PREFIX)
headers = {
"anthropic-version": anthropic_version or "2023-06-01",
"x-api-key": api_key,
"accept": "application/json",
"content-type": "application/json",
}
if _is_oauth:
headers["authorization"] = f"Bearer {api_key}"
headers["anthropic-dangerous-direct-browser-access"] = "true"
betas.add(ANTHROPIC_OAUTH_BETA_HEADER)
else:
headers["x-api-key"] = api_key
if user_anthropic_beta_headers is not None:
betas.update(user_anthropic_beta_headers)
@ -381,7 +415,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
# Vertex AI requires web search beta header for web search to work
if web_search_tool_used:
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES
headers["anthropic-beta"] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
headers[
"anthropic-beta"
] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
elif len(betas) > 0:
headers["anthropic-beta"] = ",".join(betas)
@ -398,7 +435,9 @@ class AnthropicModelInfo(BaseLLMModelInfo):
api_base: Optional[str] = None,
) -> Dict:
# Check for Anthropic OAuth token in headers
headers, api_key = optionally_handle_anthropic_oauth(headers=headers, api_key=api_key)
headers, api_key = optionally_handle_anthropic_oauth(
headers=headers, api_key=api_key
)
if api_key is None:
raise litellm.AuthenticationError(
message="Missing Anthropic API Key - A call is being made to anthropic but no key is set either in the environment variables or via params. Please set `ANTHROPIC_API_KEY` in your environment vars",
@ -416,11 +455,15 @@ class AnthropicModelInfo(BaseLLMModelInfo):
file_id_used = self.is_file_id_used(messages=messages)
web_search_tool_used = self.is_web_search_tool_used(tools=tools)
tool_search_used = self.is_tool_search_used(tools=tools)
programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(tools=tools)
programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(
tools=tools
)
input_examples_used = self.is_input_examples_used(tools=tools)
effort_used = self.is_effort_used(optional_params=optional_params, model=model)
code_execution_tool_used = self.is_code_execution_tool_used(tools=tools)
container_with_skills_used = self.is_container_with_skills_used(optional_params=optional_params)
container_with_skills_used = self.is_container_with_skills_used(
optional_params=optional_params
)
user_anthropic_beta_headers = self._get_user_anthropic_beta_headers(
anthropic_beta_header=headers.get("anthropic-beta")
)
@ -499,7 +542,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
def get_token_counter(self) -> Optional[BaseTokenCounter]:
"""
Factory method to create an Anthropic token counter.
Returns:
AnthropicTokenCounter instance for this provider.
"""

View file

@ -49,15 +49,15 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
# TODO: Add Anthropic `metadata` support
# "metadata",
]
@staticmethod
def _filter_billing_headers_from_system(system_param):
"""
Filter out x-anthropic-billing-header metadata from system parameter.
Args:
system_param: Can be a string or a list of system message content blocks
Returns:
Filtered system parameter (string or list), or None if all content was filtered
"""
@ -74,7 +74,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
text = content_block.get("text", "")
content_type = content_block.get("type", "")
# Skip text blocks that start with billing header
if content_type == "text" and text.startswith("x-anthropic-billing-header:"):
if content_type == "text" and text.startswith(
"x-anthropic-billing-header:"
):
continue
filtered_list.append(content_block)
else:
@ -111,11 +113,13 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
import os
# Check for Anthropic OAuth token in Authorization header
headers, api_key = optionally_handle_anthropic_oauth(headers=headers, api_key=api_key)
headers, api_key = optionally_handle_anthropic_oauth(
headers=headers, api_key=api_key
)
if api_key is None:
api_key = os.getenv("ANTHROPIC_API_KEY")
if "x-api-key" not in headers and api_key:
if "x-api-key" not in headers and "authorization" not in headers and api_key:
headers["x-api-key"] = api_key
if "anthropic-version" not in headers:
headers["anthropic-version"] = DEFAULT_ANTHROPIC_API_VERSION
@ -149,7 +153,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
message="max_tokens is required for Anthropic /v1/messages API",
status_code=400,
)
# Filter out x-anthropic-billing-header from system messages
system_param = anthropic_messages_optional_request_params.get("system")
if system_param is not None:
@ -159,7 +163,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
else:
# Remove system parameter if all content was filtered out
anthropic_messages_optional_request_params.pop("system", None)
####### get required params for all anthropic messages requests ######
verbose_logger.debug(f"TRANSFORMATION DEBUG - Messages: {messages}")
anthropic_messages_request: AnthropicMessagesRequest = AnthropicMessagesRequest(
@ -244,25 +248,29 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
edits = context_management_param.get("edits", [])
has_compact = False
has_other = False
for edit in edits:
edit_type = edit.get("type", "")
if edit_type == "compact_20260112":
has_compact = True
else:
has_other = True
# Add compact header if any compact edits exist
if has_compact:
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.COMPACT_2026_01_12.value)
# Add context management header if any other edits exist
if has_other:
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value)
beta_values.add(
ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value
)
# Check for structured outputs
if optional_params.get("output_format") is not None:
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value)
beta_values.add(
ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value
)
# Check for fast mode
if optional_params.get("speed") == "fast":

View file

@ -1206,7 +1206,28 @@ def get_async_httpx_client(
If not present, creates a new client
Caches the new client and returns it.
Note: When shared_session is provided, the cache is bypassed to ensure
the user's session (with its trace_configs, connector settings, etc.)
is used for the request.
"""
# When shared_session is provided, bypass cache and create a new handler
# that uses the user's session directly. This preserves the user's
# session configuration including trace_configs for aiohttp tracing.
if shared_session is not None:
verbose_logger.debug(
f"shared_session provided (ID: {id(shared_session)}), bypassing client cache"
)
if params is not None:
handler_params = {k: v for k, v in params.items() if k != "disable_aiohttp_transport"}
handler_params["shared_session"] = shared_session
return AsyncHTTPHandler(**handler_params)
else:
return AsyncHTTPHandler(
timeout=httpx.Timeout(timeout=600.0, connect=5.0),
shared_session=shared_session,
)
_params_key_name = ""
if params is not None:
for key, value in params.items():
@ -1233,12 +1254,10 @@ def get_async_httpx_client(
if params is not None:
# Filter out params that are only used for cache key, not for AsyncHTTPHandler.__init__
handler_params = {k: v for k, v in params.items() if k != "disable_aiohttp_transport"}
handler_params["shared_session"] = shared_session
_new_client = AsyncHTTPHandler(**handler_params)
else:
_new_client = AsyncHTTPHandler(
timeout=httpx.Timeout(timeout=600.0, connect=5.0),
shared_session=shared_session,
)
cache.set_cache(

View file

@ -772,12 +772,15 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig):
class OpenAIChatCompletionStreamingHandler(BaseModelResponseIterator):
def chunk_parser(self, chunk: dict) -> ModelResponseStream:
try:
return ModelResponseStream(
id=chunk["id"],
object="chat.completion.chunk",
created=chunk.get("created"),
model=chunk.get("model"),
choices=chunk.get("choices", []),
)
kwargs = {
"id": chunk["id"],
"object": "chat.completion.chunk",
"created": chunk.get("created"),
"model": chunk.get("model"),
"choices": chunk.get("choices", []),
}
if "usage" in chunk and chunk["usage"] is not None:
kwargs["usage"] = chunk["usage"]
return ModelResponseStream(**kwargs)
except Exception as e:
raise e

View file

@ -2,7 +2,7 @@ from typing import TYPE_CHECKING, Any, Dict, Optional, Union, cast, get_type_hin
import httpx
from openai.types.responses import ResponseReasoningItem
from pydantic import BaseModel
from pydantic import BaseModel, ValidationError
import litellm
from litellm._logging import verbose_logger
@ -249,7 +249,17 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
parsed_chunk["error"]["code"] = "unknown_error"
except Exception:
verbose_logger.debug("Failed to coalesce error.code in parsed_chunk")
return event_pydantic_model(**parsed_chunk)
try:
return event_pydantic_model(**parsed_chunk)
except ValidationError:
verbose_logger.debug(
"Pydantic validation failed for %s with chunk %s, "
"falling back to model_construct",
event_pydantic_model.__name__,
parsed_chunk,
)
return event_pydantic_model.model_construct(**parsed_chunk)
@staticmethod
def get_event_model_class(event_type: str) -> Any:

View file

@ -529,6 +529,17 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
raise e
def _pop_and_merge_extra_body(data: RequestBody, optional_params: dict) -> None:
"""Pop extra_body from optional_params and shallow-merge into data, deep-merging dict values."""
extra_body: Optional[dict] = optional_params.pop("extra_body", None)
if extra_body is not None:
for k, v in extra_body.items():
if k in data and isinstance(data[k], dict) and isinstance(v, dict):
data[k].update(v)
else:
data[k] = v
def _transform_request_body(
messages: List[AllMessageValues],
model: str,
@ -619,6 +630,7 @@ def _transform_request_body(
# Only add labels for Vertex AI endpoints (not Google GenAI/AI Studio) and only if non-empty
if labels and custom_llm_provider != LlmProviders.GEMINI:
data["labels"] = labels
_pop_and_merge_extra_body(data, optional_params)
except Exception as e:
raise e

View file

@ -480,7 +480,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
tool = {VertexToolName.COMPUTER_USE.value: computer_use_config}
# Handle OpenAI-style web_search and web_search_preview tools
# Transform them to Gemini's googleSearch tool
elif "type" in tool and tool["type"] in ("web_search", "web_search_preview"):
elif "type" in tool and tool["type"] in (
"web_search",
"web_search_preview",
):
verbose_logger.info(
f"Gemini: Transforming OpenAI-style '{tool['type']}' tool to googleSearch"
)
@ -1196,6 +1199,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
"PROHIBITED_CONTENT": "The token generation was stopped as the response was flagged for the prohibited contents.",
"SPII": "The token generation was stopped as the response was flagged for Sensitive Personally Identifiable Information (SPII) contents.",
"IMAGE_SAFETY": "The token generation was stopped as the response was flagged for image safety reasons.",
"IMAGE_PROHIBITED_CONTENT": "The token generation was stopped as the response was flagged for prohibited image content.",
}
@staticmethod
@ -1218,6 +1222,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
"SPII": "content_filter",
"MALFORMED_FUNCTION_CALL": "malformed_function_call", # openai doesn't have a way of representing this
"IMAGE_SAFETY": "content_filter",
"IMAGE_PROHIBITED_CONTENT": "content_filter",
}
def translate_exception_str(self, exception_string: str):
@ -1630,7 +1635,9 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
completion_image_tokens = response_tokens_details.image_tokens or 0
completion_audio_tokens = response_tokens_details.audio_tokens or 0
calculated_text_tokens = (
candidates_token_count - completion_image_tokens - completion_audio_tokens
candidates_token_count
- completion_image_tokens
- completion_audio_tokens
)
response_tokens_details.text_tokens = calculated_text_tokens
#########################################################
@ -2248,6 +2255,13 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
citation_metadata # older approach - maintaining to prevent regressions
)
## ADD TRAFFIC TYPE ##
traffic_type = completion_response.get("usageMetadata", {}).get(
"trafficType"
)
if traffic_type:
model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
except Exception as e:
raise VertexAIError(
message="Received={}, Error converting to valid response block={}. File an issue if litellm error - https://github.com/BerriAI/litellm/issues".format(
@ -2906,6 +2920,12 @@ class ModelResponseIterator:
PromptTokensDetailsWrapper, usage.prompt_tokens_details
).web_search_requests = web_search_requests
traffic_type = processed_chunk.get("usageMetadata", {}).get(
"trafficType"
)
if traffic_type:
model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
setattr(model_response, "usage", usage) # type: ignore
model_response._hidden_params["is_finished"] = False

View file

@ -7383,6 +7383,16 @@ def stream_chunk_builder( # noqa: PLR0915
setattr(response, "usage", usage)
# Propagate provider_specific_fields from the last chunk (contains provider
# metadata like traffic_type set during streaming)
for chunk in reversed(chunks):
hidden = getattr(chunk, "_hidden_params", None)
if hidden and "provider_specific_fields" in hidden:
response._hidden_params.setdefault(
"provider_specific_fields", {}
).update(hidden["provider_specific_fields"])
break
# Add cost to usage object if include_cost_in_streaming_usage is True
if litellm.include_cost_in_streaming_usage and logging_obj is not None:
setattr(

View file

@ -6105,6 +6105,32 @@
"output_cost_per_token": 2.4e-05,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.3e-07,
"litellm_provider": "bedrock",
@ -6116,6 +6142,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/ap-northeast-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.3e-07,
"litellm_provider": "bedrock",
@ -6128,7 +6181,7 @@
"supports_reasoning": true
},
"bedrock/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.3e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
@ -6159,6 +6212,32 @@
"mode": "chat",
"output_cost_per_token": 7.2e-07
},
"bedrock/ap-south-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-south-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-south-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.1e-07,
"litellm_provider": "bedrock",
@ -6170,6 +6249,86 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/ap-south-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-south-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0": {
"input_cost_per_token": 3.05e-06,
"litellm_provider": "bedrock",
@ -6188,6 +6347,46 @@
"mode": "chat",
"output_cost_per_token": 6.9e-07
},
"bedrock/eu-north-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-north-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-north-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
"input_cost_per_second": 0.01635,
"litellm_provider": "bedrock",
@ -6275,6 +6474,32 @@
"output_cost_per_token": 2.4e-05,
"supports_tool_choice": true
},
"bedrock/eu-central-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-central-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0": {
"input_cost_per_token": 2.86e-06,
"litellm_provider": "bedrock",
@ -6293,6 +6518,32 @@
"mode": "chat",
"output_cost_per_token": 6.5e-07
},
"bedrock/eu-west-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0": {
"input_cost_per_token": 3.45e-06,
"litellm_provider": "bedrock",
@ -6311,6 +6562,32 @@
"mode": "chat",
"output_cost_per_token": 7.8e-07
},
"bedrock/eu-west-2/minimax.minimax-m2.1": {
"input_cost_per_token": 4.7e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.86e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-2/qwen.qwen3-coder-next": {
"input_cost_per_token": 7.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.86e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2": {
"input_cost_per_token": 2e-07,
"litellm_provider": "bedrock",
@ -6341,6 +6618,32 @@
"output_cost_per_token": 9.1e-07,
"supports_tool_choice": true
},
"bedrock/eu-south-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-south-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0": {
"input_cost_per_token": 3e-06,
"litellm_provider": "bedrock",
@ -6375,6 +6678,32 @@
"mode": "chat",
"output_cost_per_token": 1.01e-06
},
"bedrock/sa-east-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/sa-east-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/sa-east-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.3e-07,
"litellm_provider": "bedrock",
@ -6386,6 +6715,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/sa-east-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/sa-east-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
@ -6522,6 +6878,32 @@
"output_cost_per_token": 7e-07,
"supports_tool_choice": true
},
"bedrock/us-east-1/deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
@ -6533,6 +6915,59 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/us-east-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
@ -6544,6 +6979,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/us-east-2/moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-gov-east-1/amazon.nova-pro-v1:0": {
"input_cost_per_token": 9.6e-07,
"litellm_provider": "bedrock",
@ -6950,6 +7412,32 @@
"output_cost_per_token": 7e-07,
"supports_tool_choice": true
},
"bedrock/us-west-2/deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
@ -6961,6 +7449,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/us-west-2/moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0": {
"cache_creation_input_token_cost": 1e-06,
"cache_read_input_token_cost": 8e-08,
@ -10874,6 +11389,19 @@
"supports_reasoning": true,
"supports_tool_choice": true
},
"deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"dolphin": {
"input_cost_per_token": 5e-07,
"litellm_provider": "nlp_cloud",
@ -21344,6 +21872,19 @@
"output_cost_per_token": 1.2e-06,
"supports_system_messages": true
},
"minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"minimax/speech-02-hd": {
"input_cost_per_character": 0.0001,
"litellm_provider": "minimax",
@ -22100,6 +22641,20 @@
"supports_reasoning": true,
"supports_system_messages": true
},
"moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"moonshot/kimi-k2-0711-preview": {
"cache_read_input_token_cost": 1.5e-07,
"input_cost_per_token": 6e-07,
@ -24381,11 +24936,8 @@
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "responses",
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"supported_endpoints": [
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
@ -25537,6 +26089,19 @@
"supports_system_messages": true,
"supports_vision": true
},
"qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"recraft/recraftv2": {
"litellm_provider": "recraft",
"mode": "image_generation",
@ -27973,6 +28538,30 @@
"supports_reasoning": true,
"supports_tool_choice": false
},
"us.deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"eu.deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"us.meta.llama3-1-405b-instruct-v1:0": {
"input_cost_per_token": 5.32e-06,
"litellm_provider": "bedrock",
@ -32083,6 +32672,23 @@
"1280x720"
]
},
"openai/sora-2-pro-high-res": {
"litellm_provider": "openai",
"mode": "video_generation",
"output_cost_per_video_per_second": 0.5,
"source": "https://platform.openai.com/docs/api-reference/videos",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_resolutions": [
"1024x1792",
"1792x1024"
]
},
"azure/sora-2": {
"litellm_provider": "azure",
"mode": "video_generation",
@ -35722,7 +36328,9 @@
"gpt-realtime-mini-2025-10-06": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_token_cost": 6e-08,
"input_cost_per_audio_token": 1e-05,
"input_cost_per_image": 8e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "openai",
"max_input_tokens": 128000,
@ -35753,7 +36361,9 @@
"gpt-realtime-mini-2025-12-15": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_token_cost": 6e-08,
"input_cost_per_audio_token": 1e-05,
"input_cost_per_image": 8e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "openai",
"max_input_tokens": 128000,
@ -35815,6 +36425,23 @@
"1280x720"
]
},
"sora-2-pro-high-res": {
"litellm_provider": "openai",
"mode": "video_generation",
"output_cost_per_video_per_second": 0.5,
"source": "https://platform.openai.com/docs/api-reference/videos",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_resolutions": [
"1024x1792",
"1792x1024"
]
},
"chatgpt-image-latest": {
"cache_read_input_image_token_cost": 2.5e-06,
"cache_read_input_token_cost": 1.25e-06,

View file

@ -71,9 +71,7 @@ try:
from mcp.shared.tool_name_validation import (
validate_tool_name, # pyright: ignore[reportAssignmentType]
)
from mcp.shared.tool_name_validation import (
SEP_986_URL,
)
from mcp.shared.tool_name_validation import SEP_986_URL
except ImportError:
from pydantic import BaseModel

View file

@ -925,10 +925,11 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
if isinstance(
api_key, str
): # if generated token, make sure it starts with sk-.
_masked_key = "{}****{}".format(api_key[:4], api_key[-4:]) if len(api_key) > 8 else "****"
assert api_key.startswith(
"sk-"
), "LiteLLM Virtual Key expected. Received={}, expected to start with 'sk-'.".format(
api_key
_masked_key
) # prevent token hashes from being used
else:
verbose_logger.warning(

View file

@ -5,14 +5,9 @@ OpenAI Moderation Guardrail Integration for LiteLLM
from typing import (
TYPE_CHECKING,
Any,
AsyncGenerator,
Dict,
List,
Literal,
Optional,
Type,
Union,
)
from fastapi import HTTPException
@ -20,7 +15,7 @@ from fastapi import HTTPException
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_guardrail import (
CustomGuardrail,
log_guardrail_information,
log_guardrail_information
)
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.custom_httpx.http_handler import (
@ -32,10 +27,8 @@ from litellm.types.utils import GenericGuardrailAPIInputs
from .base import OpenAIGuardrailBase
if TYPE_CHECKING:
from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.llms.openai import OpenAIModerationResponse
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
from litellm.types.utils import ModelResponse, ModelResponseStream
class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
@ -236,108 +229,6 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
# Moderation doesn't modify content, just blocks - return inputs unchanged
return inputs
@log_guardrail_information
async def async_post_call_streaming_iterator_hook(
self,
user_api_key_dict: "UserAPIKeyAuth",
response: Any,
request_data: Dict[str, Any],
) -> AsyncGenerator["ModelResponseStream", None]:
"""
Process streaming response chunks for OpenAI moderation.
Collects all chunks from the stream, assembles them into a complete response,
and applies moderation check. If content violates moderation policy, raises HTTPException.
"""
# Import here to avoid circular imports
from litellm.llms.base_llm.base_model_iterator import MockResponseIterator
from litellm.main import stream_chunk_builder
from litellm.types.utils import TextCompletionResponse
verbose_proxy_logger.debug("OpenAI Moderation: Running streaming response scan")
# Collect all chunks to process them together
all_chunks: List["ModelResponseStream"] = []
async for chunk in response:
all_chunks.append(chunk)
# Assemble the complete response from chunks
assembled_model_response: Optional[
Union["ModelResponse", TextCompletionResponse]
] = stream_chunk_builder(
chunks=all_chunks,
)
if isinstance(assembled_model_response, (type(None), TextCompletionResponse)):
# If we can't assemble a ModelResponse or it's a text completion,
# just yield the original chunks without moderation
verbose_proxy_logger.warning(
"OpenAI Moderation: Could not assemble ModelResponse from chunks, skipping moderation"
)
for chunk in all_chunks:
yield chunk
return
# Extract response text for moderation
response_text = self._extract_response_text(assembled_model_response)
if response_text:
verbose_proxy_logger.debug(
f"OpenAI Moderation: Streaming response text: {response_text[:100]}..." # Log first 100 chars
)
# Make moderation request - this will raise HTTPException if content is flagged
moderation_response = await self.async_make_request(
input_text=response_text,
)
# Check if content is flagged and raise exception if needed
self._check_moderation_result(moderation_response)
# If we reach here, content passed moderation - yield the original chunks
mock_response = MockResponseIterator(model_response=assembled_model_response)
# Return the reconstructed stream
async for chunk in mock_response:
yield chunk
def _extract_response_text(self, response: "ModelResponse") -> Optional[str]:
"""
Extract text content from the model response for moderation.
"""
if not hasattr(response, "choices") or not response.choices:
return None
response_texts = []
for choice in response.choices:
try:
# Try to get content from message (chat completion)
message = getattr(choice, "message", None)
if message:
content = getattr(message, "content", None)
if content and isinstance(content, str):
response_texts.append(content)
continue
# Try to get text (text completion)
text = getattr(choice, "text", None)
if text and isinstance(text, str):
response_texts.append(text)
continue
# Try to get content from delta (streaming)
delta = getattr(choice, "delta", None)
if delta:
content = getattr(delta, "content", None)
if content and isinstance(content, str):
response_texts.append(content)
continue
except (AttributeError, TypeError):
# Skip choices that don't have expected attributes
continue
return "\n".join(response_texts) if response_texts else None
@staticmethod
def get_config_model() -> Optional[Type["GuardrailConfigModel"]]:
"""

View file

@ -386,7 +386,11 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
continue
return final_results
except Exception as e:
raise e
# Sanitize exception to avoid leaking the original text (which may
# contain API keys or other secrets) in error responses.
raise Exception(
f"Presidio PII analysis failed: {type(e).__name__}"
) from e
async def anonymize_text(
self,
@ -443,9 +447,15 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
)
return redacted_text["text"]
else:
raise Exception(f"Invalid anonymizer response: {redacted_text}")
raise Exception("Invalid anonymizer response: received None")
except Exception as e:
raise e
# Sanitize exception to avoid leaking the original text (which may
# contain API keys or other secrets) in error responses.
if "Invalid anonymizer response" in str(e):
raise
raise Exception(
f"Presidio PII anonymization failed: {type(e).__name__}"
) from e
def filter_analyze_results_by_score(
self, analyze_results: Union[List[PresidioAnalyzeResponseItem], Dict]

View file

@ -21,6 +21,7 @@ from litellm.types.utils import GenericGuardrailAPIInputs
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
GUARDRAIL_TIMEOUT = 5
@ -334,3 +335,11 @@ class ZscalerAIGuard(CustomGuardrail):
user_facing_error = self._create_user_facing_error(f"{str(e)})")
# This exception will be caught by the proxy and returned to the user
raise HTTPException(status_code=500, detail=user_facing_error)
@staticmethod
def get_config_model() -> Optional[type["GuardrailConfigModel"]]:
from litellm.types.proxy.guardrails.guardrail_hooks.zscaler_ai_guard import (
ZscalerAIGuardConfigModel,
)
return ZscalerAIGuardConfigModel

View file

@ -12,6 +12,7 @@ from litellm._uuid import uuid
from litellm.integrations.custom_guardrail import CustomGuardrail
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.proxy.utils import PrismaClient
from litellm.proxy.types_utils.utils import get_instance_fn
from litellm.secret_managers.main import get_secret
from litellm.types.guardrails import (
Guardrail,
@ -489,7 +490,7 @@ class InMemoryGuardrailHandler:
config_file_path: Optional[str] = None,
) -> Optional[CustomGuardrail]:
"""
Initialize a Custom Guardrail from a python file
Initialize a Custom Guardrail from a python file or module path
This initializes it by adding it to the litellm callback manager
"""
@ -498,26 +499,12 @@ class InMemoryGuardrailHandler:
"GuardrailsAIException - Please pass the config_file_path to initialize_guardrails_v2"
)
_file_name, _class_name = guardrail_type.split(".")
verbose_proxy_logger.debug(
"Initializing custom guardrail: %s, file_name: %s, class_name: %s",
"Initializing custom guardrail: %s",
guardrail_type,
_file_name,
_class_name,
)
directory = os.path.dirname(config_file_path)
module_file_path = os.path.join(directory, _file_name) + ".py"
spec = importlib.util.spec_from_file_location(_class_name, module_file_path) # type: ignore
if not spec:
raise ImportError(
f"Could not find a module specification for {module_file_path}"
)
module = importlib.util.module_from_spec(spec) # type: ignore
spec.loader.exec_module(module) # type: ignore
_guardrail_class = getattr(module, _class_name)
_guardrail_class = get_instance_fn(guardrail_type, config_file_path=config_file_path)
mode = litellm_params.mode
if mode is None:

View file

@ -628,10 +628,11 @@ async def _common_key_generation_helper( # noqa: PLR0915
# Validate user-provided key format
if data.key is not None and not data.key.startswith("sk-"):
_masked = "{}****{}".format(data.key[:4], data.key[-4:]) if len(data.key) > 8 else "****"
raise HTTPException(
status_code=400,
detail={
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {data.key}"
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {_masked}"
},
)

View file

@ -740,7 +740,7 @@ class ProxyLogging:
self, dynamic_success_callbacks: Optional[List], global_callbacks: List
) -> List:
if dynamic_success_callbacks is None:
return global_callbacks
return list(global_callbacks)
return list(set(dynamic_success_callbacks + global_callbacks))
def _parse_pre_mcp_call_hook_response(

View file

@ -1512,6 +1512,12 @@ class LiteLLMCompletionResponsesConfig:
user=getattr(chat_completion_response, "user", None),
)
responses_api_response._hidden_params = getattr(chat_completion_response, "_hidden_params", {})
# Surface provider-specific fields (generic passthrough from any provider)
provider_fields = responses_api_response._hidden_params.get("provider_specific_fields")
if provider_fields:
responses_api_response.provider_specific_fields = provider_fields
return responses_api_response
@staticmethod

View file

@ -682,6 +682,7 @@ def responses(
_is_async=_is_async,
stream=stream,
extra_headers=extra_headers,
extra_body=extra_body,
**kwargs,
)

View file

@ -41,6 +41,53 @@ from litellm.types.utils import GenericBudgetConfigType, StandardLoggingPayload
DEFAULT_REDIS_SYNC_INTERVAL = 1
class _LiteLLMParamsDictView:
"""
Lightweight attribute view over `litellm_params` dict.
This avoids pydantic construction in request hot-path while preserving
attribute-style access used by `litellm.get_llm_provider(...)`.
"""
__slots__ = ("_params",)
def __init__(self, params: Dict[str, Any]):
self._params = params
def __getattr__(self, key: str) -> Any:
return self._params.get(key)
def __getitem__(self, key: str) -> Any:
return self._params.get(key)
def __contains__(self, key: str) -> bool:
return key in self._params
def get(self, key: str, default: Any = None) -> Any:
return self._params.get(key, default)
def keys(self):
return self._params.keys()
def values(self):
return self._params.values()
def items(self):
return self._params.items()
def __iter__(self):
return iter(self._params)
def __len__(self) -> int:
return len(self._params)
def dict(self) -> Dict[str, Any]:
return dict(self._params)
def model_dump(self) -> Dict[str, Any]:
return dict(self._params)
class RouterBudgetLimiting(CustomLogger):
def __init__(
self,
@ -98,6 +145,7 @@ class RouterBudgetLimiting(CustomLogger):
cache_keys,
provider_configs,
deployment_configs,
deployment_providers,
) = await self._async_get_cache_keys_for_router_budget_limiting(
healthy_deployments=healthy_deployments,
request_kwargs=request_kwargs,
@ -123,6 +171,7 @@ class RouterBudgetLimiting(CustomLogger):
healthy_deployments=healthy_deployments,
provider_configs=provider_configs,
deployment_configs=deployment_configs,
deployment_providers=deployment_providers,
spend_map=spend_map,
potential_deployments=potential_deployments,
request_tags=_get_tags_from_request_kwargs(
@ -145,6 +194,7 @@ class RouterBudgetLimiting(CustomLogger):
healthy_deployments: List[Dict[str, Any]],
provider_configs: Dict[str, GenericBudgetInfo],
deployment_configs: Dict[str, GenericBudgetInfo],
deployment_providers: List[Optional[str]],
spend_map: Dict[str, float],
request_tags: List[str],
) -> Tuple[List[Dict[str, Any]], str]:
@ -161,12 +211,15 @@ class RouterBudgetLimiting(CustomLogger):
"""
# Filter deployments based on both provider and deployment budgets
deployment_above_budget_info: str = ""
for deployment in healthy_deployments:
for idx, deployment in enumerate(healthy_deployments):
is_within_budget = True
# Check provider budget
if self.provider_budget_config:
provider = self._get_llm_provider_for_deployment(deployment)
if idx < len(deployment_providers):
provider = deployment_providers[idx]
else:
provider = self._get_llm_provider_for_deployment(deployment)
if provider in provider_configs:
config = provider_configs[provider]
if config.max_budget is None:
@ -230,24 +283,32 @@ class RouterBudgetLimiting(CustomLogger):
self,
healthy_deployments: List[Dict[str, Any]],
request_kwargs: Optional[Dict] = None,
) -> Tuple[List[str], Dict[str, GenericBudgetInfo], Dict[str, GenericBudgetInfo]]:
) -> Tuple[
List[str],
Dict[str, GenericBudgetInfo],
Dict[str, GenericBudgetInfo],
List[Optional[str]],
]:
"""
Returns list of cache keys to fetch from router cache for budget limiting and provider and deployment configs
Returns:
Tuple[List[str], Dict[str, GenericBudgetInfo], Dict[str, GenericBudgetInfo]]:
Tuple[List[str], Dict[str, GenericBudgetInfo], Dict[str, GenericBudgetInfo], List[Optional[str]]]:
- List of cache keys to fetch from router cache for budget limiting
- Dict of provider budget configs `provider_configs`
- Dict of deployment budget configs `deployment_configs`
- List of resolved providers aligned by deployment index `deployment_providers`
"""
cache_keys: List[str] = []
provider_configs: Dict[str, GenericBudgetInfo] = {}
deployment_configs: Dict[str, GenericBudgetInfo] = {}
deployment_providers: List[Optional[str]] = []
for deployment in healthy_deployments:
# Check provider budgets
if self.provider_budget_config:
provider = self._get_llm_provider_for_deployment(deployment)
deployment_providers.append(provider)
if provider is not None:
budget_config = self._get_budget_config_for_provider(provider)
if (
@ -280,7 +341,12 @@ class RouterBudgetLimiting(CustomLogger):
cache_keys.append(
f"tag_spend:{_tag}:{_tag_budget_config.budget_duration}"
)
return cache_keys, provider_configs, deployment_configs
return (
cache_keys,
provider_configs,
deployment_configs,
deployment_providers,
)
async def _get_or_set_budget_start_time(
self, start_time_key: str, current_time: float, ttl_seconds: int
@ -597,12 +663,23 @@ class RouterBudgetLimiting(CustomLogger):
def _get_llm_provider_for_deployment(self, deployment: Dict) -> Optional[str]:
try:
_litellm_params: LiteLLM_Params = LiteLLM_Params(
**deployment.get("litellm_params", {"model": ""})
)
deployment_litellm_params = deployment.get("litellm_params") or {}
if isinstance(deployment_litellm_params, LiteLLM_Params):
model = deployment_litellm_params.model or ""
provider_resolution_params: Any = deployment_litellm_params
elif isinstance(deployment_litellm_params, dict):
model = deployment_litellm_params.get("model") or ""
provider_resolution_params = _LiteLLMParamsDictView(
deployment_litellm_params
)
else:
model = ""
provider_resolution_params = _LiteLLMParamsDictView({})
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
model=_litellm_params.model,
litellm_params=_litellm_params,
model=str(model),
litellm_params=provider_resolution_params,
)
except Exception:
verbose_router_logger.error(

View file

@ -0,0 +1,132 @@
from typing import Optional
from pydantic import Field, model_validator
from litellm._logging import verbose_proxy_logger
from litellm.types.guardrails import GuardrailParamUITypes
from .base import GuardrailConfigModel
class ZscalerAIGuardConfigModel(GuardrailConfigModel):
api_key: Optional[str] = Field(
default=None,
description=(
"API key for Zscaler AI Guard authentication. "
"If not provided, falls back to ZSCALER_AI_GUARD_API_KEY environment variable."
),
)
api_base: Optional[str] = Field(
default=None,
description=(
"Zscaler AI Guard API endpoint. Determines policy resolution behavior:\n"
"• /execute-policy (default) - Requires explicit policy_id in configuration\n"
"• /resolve-and-execute-policy - Infers policy from user-api-key-alias header\n"
"Default: https://api.us1.zseclipse.net/v1/detection/execute-policy\n"
"Falls back to ZSCALER_AI_GUARD_URL environment variable."
),
json_schema_extra={
"examples": [
"https://api.us1.zseclipse.net/v1/detection/execute-policy",
"https://api.us1.zseclipse.net/v1/detection/resolve-and-execute-policy",
]
},
)
policy_id: Optional[int] = Field(
default=None,
description=(
"Global policy ID for Zscaler AI Guard. Required when using /execute-policy endpoint.\n\n"
"Set to 0 or leave empty when using /resolve-and-execute-policy with dynamic policy resolution.\n"
"Falls back to ZSCALER_AI_GUARD_POLICY_ID environment variable."
),
json_schema_extra={
"ui_hint": "conditional_required",
"condition": "Required when api_base ends with /execute-policy",
},
)
send_user_api_key_alias: Optional[bool] = Field(
default=False,
description=(
"Send user API key alias in request headers as 'user-api-key-alias'. "
"CRITICAL when using /resolve-and-execute-policy endpoint - the policy is inferred from this value. "
"Also useful for tracking/auditing with /execute-policy endpoint."
),
json_schema_extra={
"ui_type": GuardrailParamUITypes.BOOL,
"ui_hint": "recommended_when",
"condition": "Recommended when api_base ends with /resolve-and-execute-policy",
},
)
send_user_api_key_user_id: Optional[bool] = Field(
default=False,
description=(
"Send user API key user_id in request headers as 'user-api-key-user-id'. "
"Enables user-level tracking and analytics in Zscaler AI Guard."
),
json_schema_extra={"ui_type": GuardrailParamUITypes.BOOL},
)
send_user_api_key_team_id: Optional[bool] = Field(
default=False,
description=(
"Send user API key team_id in request headers as 'user-api-key-team-id'. "
"Enables team-level tracking and analytics in Zscaler AI Guard."
),
json_schema_extra={"ui_type": GuardrailParamUITypes.BOOL},
)
@model_validator(mode="after")
def validate_endpoint_configuration(self) -> "ZscalerAIGuardConfigModel":
"""
Validate configuration consistency between api_base and other fields.
Provides warnings but doesn't block (since env vars might provide values).
"""
import os
# Resolve actual api_base value (including env fallback)
api_base = self.api_base or os.getenv(
"ZSCALER_AI_GUARD_URL",
"https://api.us1.zseclipse.net/v1/detection/execute-policy",
)
# Resolve actual policy_id value
policy_id = self.policy_id
if policy_id is None:
env_policy = os.getenv("ZSCALER_AI_GUARD_POLICY_ID")
if env_policy:
try:
policy_id = int(env_policy)
except ValueError:
verbose_proxy_logger.warning(
f"ZSCALER_AI_GUARD_POLICY_ID env var is not a valid integer: {env_policy}"
)
# Check for configuration issues
is_resolve_policy = api_base.endswith("/resolve-and-execute-policy")
is_execute_policy = api_base.endswith("/execute-policy") and not is_resolve_policy
# Scenario A: execute-policy without policy_id
if is_execute_policy and (policy_id is None or policy_id < 1):
verbose_proxy_logger.warning(
"Using /execute-policy endpoint without a valid policy_id. "
"Ensure ZSCALER_AI_GUARD_POLICY_ID environment variable is set, "
"or provide policy_id via request/key/team metadata."
)
# Scenario B: resolve-and-execute-policy without user_api_key_alias
if is_resolve_policy and not self.send_user_api_key_alias:
verbose_proxy_logger.warning(
"Using /resolve-and-execute-policy endpoint without send_user_api_key_alias=true. "
"The endpoint requires user-api-key-alias header to resolve the policy. "
"Set send_user_api_key_alias to true or ensure the header is sent via other means."
)
return self
@staticmethod
def ui_friendly_name() -> str:
return "Zscaler AI Guard"

View file

@ -6105,6 +6105,32 @@
"output_cost_per_token": 2.4e-05,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.3e-07,
"litellm_provider": "bedrock",
@ -6116,6 +6142,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/ap-northeast-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.3e-07,
"litellm_provider": "bedrock",
@ -6128,7 +6181,7 @@
"supports_reasoning": true
},
"bedrock/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.3e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
@ -6159,6 +6212,32 @@
"mode": "chat",
"output_cost_per_token": 7.2e-07
},
"bedrock/ap-south-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-south-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-south-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.1e-07,
"litellm_provider": "bedrock",
@ -6170,6 +6249,86 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/ap-south-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-south-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-southeast-3/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0": {
"input_cost_per_token": 3.05e-06,
"litellm_provider": "bedrock",
@ -6188,6 +6347,46 @@
"mode": "chat",
"output_cost_per_token": 6.9e-07
},
"bedrock/eu-north-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-north-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-north-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
"input_cost_per_second": 0.01635,
"litellm_provider": "bedrock",
@ -6275,6 +6474,32 @@
"output_cost_per_token": 2.4e-05,
"supports_tool_choice": true
},
"bedrock/eu-central-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-central-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0": {
"input_cost_per_token": 2.86e-06,
"litellm_provider": "bedrock",
@ -6293,6 +6518,32 @@
"mode": "chat",
"output_cost_per_token": 6.5e-07
},
"bedrock/eu-west-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0": {
"input_cost_per_token": 3.45e-06,
"litellm_provider": "bedrock",
@ -6311,6 +6562,32 @@
"mode": "chat",
"output_cost_per_token": 7.8e-07
},
"bedrock/eu-west-2/minimax.minimax-m2.1": {
"input_cost_per_token": 4.7e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.86e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-2/qwen.qwen3-coder-next": {
"input_cost_per_token": 7.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.86e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2": {
"input_cost_per_token": 2e-07,
"litellm_provider": "bedrock",
@ -6341,6 +6618,32 @@
"output_cost_per_token": 9.1e-07,
"supports_tool_choice": true
},
"bedrock/eu-south-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-south-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0": {
"input_cost_per_token": 3e-06,
"litellm_provider": "bedrock",
@ -6375,6 +6678,32 @@
"mode": "chat",
"output_cost_per_token": 1.01e-06
},
"bedrock/sa-east-1/deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/sa-east-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3.6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/sa-east-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 7.3e-07,
"litellm_provider": "bedrock",
@ -6386,6 +6715,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/sa-east-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 7.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/sa-east-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.44e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
@ -6522,6 +6878,32 @@
"output_cost_per_token": 7e-07,
"supports_tool_choice": true
},
"bedrock/us-east-1/deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
@ -6533,6 +6915,59 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/us-east-1/moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
@ -6544,6 +6979,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/us-east-2/moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-2/qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-gov-east-1/amazon.nova-pro-v1:0": {
"input_cost_per_token": 9.6e-07,
"litellm_provider": "bedrock",
@ -6950,6 +7412,32 @@
"output_cost_per_token": 7e-07,
"supports_tool_choice": true
},
"bedrock/us-west-2/deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/moonshotai.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
@ -6961,6 +7449,33 @@
"supports_function_calling": true,
"supports_reasoning": true
},
"bedrock/us-west-2/moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0": {
"cache_creation_input_token_cost": 1e-06,
"cache_read_input_token_cost": 8e-08,
@ -10874,6 +11389,19 @@
"supports_reasoning": true,
"supports_tool_choice": true
},
"deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"dolphin": {
"input_cost_per_token": 5e-07,
"litellm_provider": "nlp_cloud",
@ -21344,6 +21872,19 @@
"output_cost_per_token": 1.2e-06,
"supports_system_messages": true
},
"minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 196000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"minimax/speech-02-hd": {
"input_cost_per_character": 0.0001,
"litellm_provider": "minimax",
@ -22100,6 +22641,20 @@
"supports_reasoning": true,
"supports_system_messages": true
},
"moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 3e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"moonshot/kimi-k2-0711-preview": {
"cache_read_input_token_cost": 1.5e-07,
"input_cost_per_token": 6e-07,
@ -24381,11 +24936,8 @@
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "responses",
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"supported_endpoints": [
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
@ -25537,6 +26089,19 @@
"supports_system_messages": true,
"supports_vision": true
},
"qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"recraft/recraftv2": {
"litellm_provider": "recraft",
"mode": "image_generation",
@ -27973,6 +28538,30 @@
"supports_reasoning": true,
"supports_tool_choice": false
},
"us.deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.85e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"eu.deepseek.v3.2": {
"input_cost_per_token": 7.4e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 2.22e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"us.meta.llama3-1-405b-instruct-v1:0": {
"input_cost_per_token": 5.32e-06,
"litellm_provider": "bedrock",

View file

@ -182,6 +182,76 @@ class TestEnterpriseRouteChecks:
EnterpriseRouteChecks.should_call_route("/config/update")
@patch("litellm.proxy.proxy_server.premium_user", True)
class TestEnterpriseRouteChecksModelListExemption:
"""Test that /models and /v1/models are exempt from DISABLE_LLM_API_ENDPOINTS"""
@patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled")
@patch.object(EnterpriseRouteChecks, "is_management_routes_disabled")
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_llm_api_route")
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_management_route")
def test_models_route_allowed_when_llm_api_disabled(
self,
mock_is_management_route,
mock_is_llm_api_route,
mock_is_management_disabled,
mock_is_llm_api_disabled,
):
"""Test that /models is allowed even when LLM API routes are disabled"""
mock_is_management_route.return_value = False
mock_is_llm_api_route.return_value = True
mock_is_management_disabled.return_value = False
mock_is_llm_api_disabled.return_value = True
# Should not raise exception for /models
EnterpriseRouteChecks.should_call_route("/models")
@patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled")
@patch.object(EnterpriseRouteChecks, "is_management_routes_disabled")
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_llm_api_route")
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_management_route")
def test_v1_models_route_allowed_when_llm_api_disabled(
self,
mock_is_management_route,
mock_is_llm_api_route,
mock_is_management_disabled,
mock_is_llm_api_disabled,
):
"""Test that /v1/models is allowed even when LLM API routes are disabled"""
mock_is_management_route.return_value = False
mock_is_llm_api_route.return_value = True
mock_is_management_disabled.return_value = False
mock_is_llm_api_disabled.return_value = True
# Should not raise exception for /v1/models
EnterpriseRouteChecks.should_call_route("/v1/models")
@patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled")
@patch.object(EnterpriseRouteChecks, "is_management_routes_disabled")
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_llm_api_route")
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_management_route")
def test_chat_completions_still_blocked_when_llm_api_disabled(
self,
mock_is_management_route,
mock_is_llm_api_route,
mock_is_management_disabled,
mock_is_llm_api_disabled,
):
"""Test that non-exempt LLM routes like /v1/chat/completions are still blocked"""
mock_is_management_route.return_value = False
mock_is_llm_api_route.return_value = True
mock_is_management_disabled.return_value = False
mock_is_llm_api_disabled.return_value = True
with pytest.raises(HTTPException) as exc_info:
EnterpriseRouteChecks.should_call_route("/v1/chat/completions")
assert exc_info.value.status_code == 403
assert "LLM API routes are disabled for this instance." in str(
exc_info.value.detail
)
class TestEnterpriseRouteChecksErrorMessages:
"""Test that error messages correctly identify which feature requires Enterprise license"""

View file

@ -0,0 +1,170 @@
"""
Test suite for AWS Bedrock extended beta model support
Tests model configuration, pricing, and regional availability for:
- DeepSeek V3.2
- Minimax M2.1
- Moonshot AI Kimi K2.5
- Qwen3 Coder Next
"""
import os
# Set env var to use local model cost map instead of fetching from remote
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "true"
import pytest
from litellm import get_model_info
# Model configurations: (model_name, regions, max_input, max_output)
MODEL_CONFIGS = [
(
"deepseek.v3.2",
[
"ap-northeast-1",
"ap-south-1",
"ap-southeast-3",
"eu-north-1",
"sa-east-1",
"us-east-1",
"us-east-2",
"us-west-2",
],
163840,
163840,
),
(
"minimax.minimax-m2.1",
[
"ap-northeast-1",
"ap-south-1",
"ap-southeast-3",
"eu-central-1",
"eu-north-1",
"eu-south-1",
"eu-west-1",
"eu-west-2",
"sa-east-1",
"us-east-1",
"us-east-2",
"us-west-2",
],
196000,
8192,
),
(
"moonshotai.kimi-k2.5",
[
"ap-northeast-1",
"ap-south-1",
"ap-southeast-3",
"eu-north-1",
"sa-east-1",
"us-east-1",
"us-east-2",
"us-west-2",
],
262144,
262144,
),
(
"qwen.qwen3-coder-next",
[
"ap-northeast-1",
"ap-south-1",
"ap-southeast-3",
"eu-central-1",
"eu-south-1",
"eu-west-1",
"eu-west-2",
"sa-east-1",
"us-east-1",
"us-east-2",
"us-west-2",
],
262144,
8192,
),
]
class TestBedrockNewModels:
"""Unified test suite for all new Bedrock models"""
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
def test_model_info_primary_region(
self, model_name, regions, max_input, max_output
):
"""Test model configuration in primary region (us-east-1)"""
model = f"bedrock/us-east-1/{model_name}"
model_info = get_model_info(model)
assert model_info is not None, f"Model {model_name} not found"
assert model_info["max_input_tokens"] == max_input
assert model_info["max_output_tokens"] == max_output
assert model_info["litellm_provider"] == "bedrock"
assert model_info["mode"] == "chat"
assert model_info["supports_function_calling"] is True
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
def test_pricing_configured(self, model_name, regions, max_input, max_output):
"""Verify pricing is set for all models"""
model = f"bedrock/us-east-1/{model_name}"
model_info = get_model_info(model)
assert (
model_info["input_cost_per_token"] > 0
), f"Missing input cost for {model_name}"
assert (
model_info["output_cost_per_token"] > 0
), f"Missing output cost for {model_name}"
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
def test_region_count(self, model_name, regions, max_input, max_output):
"""Verify each bedrock/{region}/{model_name} resolves via get_model_info"""
for region in regions:
model = f"bedrock/{region}/{model_name}"
model_info = get_model_info(model)
assert model_info is not None, f"Model {model_name} not found in {region}"
assert model_info["max_input_tokens"] == max_input
assert model_info["max_output_tokens"] == max_output
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
def test_sample_regional_variants(self, model_name, regions, max_input, max_output):
"""Test sample regional variants (us-east-1, eu-west-1, ap-northeast-1)"""
for region in ["us-east-1", "ap-northeast-1"]:
if region in regions:
model = f"bedrock/{region}/{model_name}"
model_info = get_model_info(model)
assert (
model_info is not None
), f"Model {model_name} not found in {region}"
assert model_info["max_input_tokens"] == max_input
assert model_info["litellm_provider"] == "bedrock"
class TestModelSpecificFeatures:
"""Model-specific capability tests"""
def test_deepseek_v3_2_context_window(self):
"""DeepSeek V3.2 has 163K context window"""
model_info = get_model_info("bedrock/us-east-1/deepseek.v3.2")
assert model_info["max_input_tokens"] == 163840
def test_minimax_m2_1_context_window(self):
"""Minimax M2.1 has 196K input, 8K output"""
model_info = get_model_info("bedrock/us-east-1/minimax.minimax-m2.1")
assert model_info["max_input_tokens"] == 196000
assert model_info["max_output_tokens"] == 8192
def test_moonshotai_kimi_k2_5_context_window(self):
"""Moonshot AI Kimi K2.5 has 256K context window"""
model_info = get_model_info("bedrock/us-east-1/moonshotai.kimi-k2.5")
assert model_info["max_input_tokens"] == 262144
assert model_info["max_output_tokens"] == 262144
def test_qwen3_coder_next_context_window(self):
"""Qwen3 Coder Next has 256K input, 8K output"""
model_info = get_model_info("bedrock/us-east-1/qwen.qwen3-coder-next")
assert model_info["max_input_tokens"] == 262144
assert model_info["max_output_tokens"] == 8192

View file

@ -293,3 +293,29 @@ def test_get_combined_callback_list():
assert "lago" in _logging.get_combined_callback_list(
dynamic_success_callbacks=["langfuse"], global_callbacks=["lago"]
)
def test_get_combined_callback_list_returns_copy_when_dynamic_is_none():
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
_logging = LiteLLMLoggingObj(
model="claude-3-opus-20240229",
messages=[{"role": "user", "content": "hi"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="123",
function_id="456",
)
global_callbacks = ["langfuse"]
combined_callbacks = _logging.get_combined_callback_list(
dynamic_success_callbacks=None, global_callbacks=global_callbacks
)
assert combined_callbacks == ["langfuse"]
assert combined_callbacks is not global_callbacks
combined_callbacks.append("new_callback")
assert global_callbacks == ["langfuse"]

View file

@ -0,0 +1,58 @@
import asyncio
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from litellm.caching.dual_cache import DualCache
from litellm.caching.redis_cache import RedisCache
@pytest.mark.asyncio
async def test_dual_cache_async_batch_get_cache_coalesces_concurrent_redis_reads():
dual_cache = DualCache(
redis_cache=MagicMock(spec=RedisCache), default_redis_batch_cache_expiry=10
)
keys = ["shared_a", "shared_b"]
start_gate = asyncio.Event()
async def _mock_async_batch_get_cache(key_list, parent_otel_span=None):
await asyncio.sleep(0.05)
return {k: None for k in key_list}
with patch.object(
dual_cache.redis_cache,
"async_batch_get_cache",
new=AsyncMock(side_effect=_mock_async_batch_get_cache),
) as mock_async_batch_get_cache:
async def worker():
await start_gate.wait()
return await dual_cache.async_batch_get_cache(keys=keys)
tasks = [asyncio.create_task(worker()) for _ in range(50)]
start_gate.set()
await asyncio.gather(*tasks)
assert mock_async_batch_get_cache.call_count == 1
@pytest.mark.asyncio
async def test_dual_cache_async_batch_get_cache_rolls_back_redis_reservation_on_error():
dual_cache = DualCache(
redis_cache=MagicMock(spec=RedisCache), default_redis_batch_cache_expiry=10
)
keys = ["shared_a", "shared_b"]
with patch.object(
dual_cache.redis_cache,
"async_batch_get_cache",
new=AsyncMock(side_effect=RuntimeError("redis unavailable")),
) as mock_async_batch_get_cache:
first_result = await dual_cache.async_batch_get_cache(keys=keys)
second_result = await dual_cache.async_batch_get_cache(keys=keys)
assert first_result is None
assert second_result is None
assert mock_async_batch_get_cache.call_count == 2
assert "shared_a" not in dual_cache.last_redis_batch_access_time
assert "shared_b" not in dual_cache.last_redis_batch_access_time

View file

@ -1,84 +1,285 @@
"""
Tests for Anthropic OAuth token handling for Claude Code Max integration.
Tests for Anthropic OAuth token handling in common_utils.
Verifies that OAuth tokens (sk-ant-oat*) are sent via Authorization: Bearer
instead of x-api-key, per Anthropic's OAuth specification.
"""
import os
import sys
# Add litellm to path
sys.path.insert(0, os.path.abspath("../../../../.."))
sys.path.insert(
0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
)
# Fake OAuth token for testing (not a real secret)
FAKE_OAUTH_TOKEN = "sk-ant-oat01-fake-token-for-testing-123456789abcdef"
FAKE_REGULAR_KEY = "sk-ant-api03-regular-key-for-testing-123456789"
def test_oauth_detection_in_common_utils():
"""Test 1: OAuth token detection in common_utils"""
from litellm.llms.anthropic.common_utils import optionally_handle_anthropic_oauth
class TestOptionallyHandleAnthropicOAuth:
"""Tests for optionally_handle_anthropic_oauth function."""
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
updated_headers, extracted_api_key = optionally_handle_anthropic_oauth(headers, None)
def test_oauth_token_in_authorization_header(self):
"""OAuth token in Authorization header should be detected and headers set correctly."""
from litellm.llms.anthropic.common_utils import (
optionally_handle_anthropic_oauth,
)
assert extracted_api_key == FAKE_OAUTH_TOKEN
assert updated_headers["anthropic-beta"] == "oauth-2025-04-20"
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
updated_headers, extracted_api_key = optionally_handle_anthropic_oauth(
headers, None
)
assert extracted_api_key == FAKE_OAUTH_TOKEN
assert updated_headers["anthropic-beta"] == "oauth-2025-04-20"
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
assert "x-api-key" not in updated_headers
def test_oauth_token_in_api_key_directly(self):
"""OAuth token passed as api_key should set Authorization: Bearer header."""
from litellm.llms.anthropic.common_utils import (
optionally_handle_anthropic_oauth,
)
headers = {}
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
headers, FAKE_OAUTH_TOKEN
)
assert returned_api_key == FAKE_OAUTH_TOKEN
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
assert updated_headers["anthropic-beta"] == "oauth-2025-04-20"
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
assert "x-api-key" not in updated_headers
def test_oauth_removes_existing_x_api_key(self):
"""When OAuth is detected, any existing x-api-key should be removed."""
from litellm.llms.anthropic.common_utils import (
optionally_handle_anthropic_oauth,
)
headers = {"x-api-key": FAKE_OAUTH_TOKEN}
updated_headers, _ = optionally_handle_anthropic_oauth(
headers, FAKE_OAUTH_TOKEN
)
assert "x-api-key" not in updated_headers
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
def test_regular_api_key_unchanged(self):
"""Regular API keys (non-OAuth) should pass through unmodified."""
from litellm.llms.anthropic.common_utils import (
optionally_handle_anthropic_oauth,
)
headers = {}
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
headers, FAKE_REGULAR_KEY
)
assert returned_api_key == FAKE_REGULAR_KEY
assert "authorization" not in updated_headers
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
assert "anthropic-beta" not in updated_headers
def test_regular_key_in_authorization_header(self):
"""Non-OAuth token in Authorization header should not trigger OAuth handling."""
from litellm.llms.anthropic.common_utils import (
optionally_handle_anthropic_oauth,
)
headers = {"authorization": f"Bearer {FAKE_REGULAR_KEY}"}
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
headers, FAKE_REGULAR_KEY
)
assert returned_api_key == FAKE_REGULAR_KEY
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
def test_none_api_key_no_error(self):
"""None api_key with empty headers should not raise errors."""
from litellm.llms.anthropic.common_utils import (
optionally_handle_anthropic_oauth,
)
headers = {}
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
headers, None
)
assert returned_api_key is None
assert "authorization" not in updated_headers
def test_oauth_integration_in_validate_environment():
"""Test 2: OAuth integration in AnthropicConfig validate_environment"""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
class TestGetAnthropicHeaders:
"""Tests for get_anthropic_headers method with OAuth support."""
config = AnthropicModelInfo()
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
def test_oauth_token_uses_authorization_bearer(self):
"""OAuth token should produce Authorization: Bearer header, not x-api-key."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
updated_headers = config.validate_environment(
headers=headers,
model="claude-3-haiku-20240307",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=None,
api_base=None,
)
config = AnthropicModelInfo()
headers = config.get_anthropic_headers(
api_key=FAKE_OAUTH_TOKEN,
computer_tool_used=False,
prompt_caching_set=False,
pdf_used=False,
is_vertex_request=False,
)
assert updated_headers["x-api-key"] == FAKE_OAUTH_TOKEN
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
assert headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
assert headers["anthropic-dangerous-direct-browser-access"] == "true"
assert "oauth-2025-04-20" in headers.get("anthropic-beta", "")
assert "x-api-key" not in headers
def test_regular_key_uses_x_api_key(self):
"""Regular API key should produce x-api-key header, not Authorization."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
config = AnthropicModelInfo()
headers = config.get_anthropic_headers(
api_key=FAKE_REGULAR_KEY,
computer_tool_used=False,
prompt_caching_set=False,
pdf_used=False,
is_vertex_request=False,
)
assert headers["x-api-key"] == FAKE_REGULAR_KEY
assert "authorization" not in headers
assert "anthropic-dangerous-direct-browser-access" not in headers
def test_oauth_includes_standard_headers(self):
"""OAuth path should still include standard Anthropic headers."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
config = AnthropicModelInfo()
headers = config.get_anthropic_headers(
api_key=FAKE_OAUTH_TOKEN,
computer_tool_used=False,
prompt_caching_set=False,
pdf_used=False,
is_vertex_request=False,
)
assert headers["anthropic-version"] == "2023-06-01"
assert headers["accept"] == "application/json"
assert headers["content-type"] == "application/json"
def test_oauth_detection_in_messages_transformation():
"""Test 3: OAuth detection in messages transformation"""
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
class TestValidateEnvironmentOAuth:
"""Tests for validate_environment with OAuth tokens."""
config = AnthropicMessagesConfig()
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
def test_oauth_via_authorization_header(self):
"""validate_environment should produce correct headers for OAuth tokens."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
updated_headers, _ = config.validate_anthropic_messages_environment(
headers=headers,
model="claude-3-haiku-20240307",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=None,
api_base=None,
)
config = AnthropicModelInfo()
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
assert updated_headers["x-api-key"] == FAKE_OAUTH_TOKEN
assert "oauth-2025-04-20" in updated_headers["anthropic-beta"]
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
updated_headers = config.validate_environment(
headers=headers,
model="claude-sonnet-4-5-20250929",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=None,
api_base=None,
)
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
assert "oauth-2025-04-20" in updated_headers.get("anthropic-beta", "")
assert "x-api-key" not in updated_headers
def test_oauth_via_api_key_param(self):
"""validate_environment with OAuth token as api_key should use Bearer auth."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
config = AnthropicModelInfo()
headers = {}
updated_headers = config.validate_environment(
headers=headers,
model="claude-sonnet-4-5-20250929",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=FAKE_OAUTH_TOKEN,
api_base=None,
)
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
assert "x-api-key" not in updated_headers
def test_regular_key_via_api_key_param(self):
"""validate_environment with regular API key should use x-api-key."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
config = AnthropicModelInfo()
headers = {}
updated_headers = config.validate_environment(
headers=headers,
model="claude-sonnet-4-5-20250929",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=FAKE_REGULAR_KEY,
api_base=None,
)
assert updated_headers["x-api-key"] == FAKE_REGULAR_KEY
assert "authorization" not in updated_headers
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
def test_regular_api_keys_still_work():
"""Test 4: Regular API keys still work (regression test)"""
from litellm.llms.anthropic.common_utils import optionally_handle_anthropic_oauth
class TestPassthroughOAuth:
"""Tests for passthrough messages endpoint with OAuth tokens."""
regular_key = "sk-ant-api03-regular-key-123"
headers = {"authorization": f"Bearer {regular_key}"}
def test_passthrough_oauth_no_x_api_key(self):
"""Passthrough endpoint should not add x-api-key for OAuth tokens."""
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
updated_headers, extracted_api_key = optionally_handle_anthropic_oauth(headers, regular_key)
config = AnthropicMessagesConfig()
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
# Regular key should be unchanged
assert extracted_api_key == regular_key
# OAuth headers should NOT be added
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
updated_headers, _ = config.validate_anthropic_messages_environment(
headers=headers,
model="claude-sonnet-4-5-20250929",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=None,
api_base=None,
)
assert "oauth-2025-04-20" in updated_headers.get("anthropic-beta", "")
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
assert "x-api-key" not in updated_headers
def test_passthrough_regular_key_uses_x_api_key(self):
"""Passthrough endpoint should still use x-api-key for regular API keys."""
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
config = AnthropicMessagesConfig()
headers = {}
updated_headers, _ = config.validate_anthropic_messages_environment(
headers=headers,
model="claude-sonnet-4-5-20250929",
messages=[{"role": "user", "content": "Hello"}],
optional_params={},
litellm_params={},
api_key=FAKE_REGULAR_KEY,
api_base=None,
)
assert updated_headers["x-api-key"] == FAKE_REGULAR_KEY
assert "authorization" not in updated_headers

View file

@ -480,6 +480,85 @@ async def test_session_reuse_integration():
await client2.close()
@pytest.mark.asyncio
async def test_shared_session_bypasses_cache():
"""
Test that when shared_session is provided, the cache is bypassed.
This is critical for aiohttp tracing support - users need their custom
ClientSession (with trace_configs) to be used, not a cached session.
Related: GitHub issue #20174
"""
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
from litellm.types.utils import LlmProviders
# First, get a cached client without shared_session
cached_client = get_async_httpx_client(
llm_provider=LlmProviders.ANTHROPIC,
shared_session=None
)
# Now create a mock shared session
mock_session = MockClientSession()
# Get a client WITH shared_session - this should NOT return the cached client
client_with_session = get_async_httpx_client(
llm_provider=LlmProviders.ANTHROPIC, # Same provider!
shared_session=mock_session # type: ignore
)
# The clients should be DIFFERENT - cache should be bypassed when shared_session is provided
assert client_with_session is not cached_client, \
"Cache should be bypassed when shared_session is provided"
# Verify the shared_session handler is using our mock session
# The transport should have our mock_session as its client
transport = client_with_session.client._transport
if hasattr(transport, 'client'):
assert transport.client is mock_session, \
"Handler should use the provided shared_session"
# Clean up
await cached_client.close()
await client_with_session.close()
@pytest.mark.asyncio
async def test_shared_session_each_call_gets_new_handler():
"""
Test that each call with shared_session creates a new handler.
This ensures user sessions (with their trace_configs, etc.) are always
used and not affected by caching.
"""
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
from litellm.types.utils import LlmProviders
# Create two different mock sessions
mock_session1 = MockClientSession()
mock_session2 = MockClientSession()
# Get clients with different sessions for the same provider
client1 = get_async_httpx_client(
llm_provider=LlmProviders.ANTHROPIC,
shared_session=mock_session1 # type: ignore
)
client2 = get_async_httpx_client(
llm_provider=LlmProviders.ANTHROPIC, # Same provider
shared_session=mock_session2 # type: ignore # Different session
)
# Should be different clients, each using their own session
assert client1 is not client2, \
"Different shared_sessions should create different handlers"
# Clean up
await client1.close()
await client2.close()
@pytest.mark.asyncio
async def test_session_validation():
"""Test that session validation works correctly"""

View file

@ -9,7 +9,10 @@ import pytest
sys.path.insert(0, os.path.abspath("../../../../.."))
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from litellm.llms.openai.chat.gpt_transformation import (
OpenAIGPTConfig,
OpenAIChatCompletionStreamingHandler,
)
class TestOpenAIGPTConfig:
@ -136,6 +139,75 @@ class TestGetOptionalParamsIntegration:
assert regular_params.get("user") == "my-end-user"
assert responses_params.get("user") == "my-end-user"
class TestOpenAIChatCompletionStreamingHandler:
"""Tests for OpenAIChatCompletionStreamingHandler.chunk_parser()"""
def test_chunk_parser_preserves_usage(self):
"""
Test that chunk_parser preserves the usage field from streaming chunks.
"""
handler = OpenAIChatCompletionStreamingHandler(
streaming_response=None, sync_stream=True
)
usage_chunk = {
"id": "gen-123",
"created": 1234567890,
"model": "openai/gpt-4o-mini",
"object": "chat.completion.chunk",
"choices": [
{
"index": 0,
"delta": {"role": "assistant", "content": ""},
"finish_reason": None,
}
],
"usage": {
"prompt_tokens": 13797,
"completion_tokens": 350,
"total_tokens": 14147,
},
}
result = handler.chunk_parser(usage_chunk)
assert result.usage is not None
assert result.usage.prompt_tokens == 13797
assert result.usage.completion_tokens == 350
assert result.usage.total_tokens == 14147
def test_chunk_parser_without_usage(self):
"""Test that chunk_parser works normally for chunks without usage."""
handler = OpenAIChatCompletionStreamingHandler(
streaming_response=None, sync_stream=True
)
chunk = {
"id": "gen-123",
"created": 1234567890,
"model": "openai/gpt-4o-mini",
"object": "chat.completion.chunk",
"choices": [
{
"index": 0,
"delta": {"role": "assistant", "content": "Hello"},
"finish_reason": None,
}
],
}
result = handler.chunk_parser(chunk)
assert result.id == "gen-123"
assert result.choices[0].delta.content == "Hello"
assert not hasattr(result, "usage") or result.usage is None
class TestPromptCacheKeyIntegration:
"""Tests for prompt_cache_key support"""
def test_prompt_cache_key_in_optional_params(self):
"""Test that 'prompt_cache_key' flows through get_optional_params for OpenAI models."""
from litellm.utils import get_optional_params

View file

@ -417,6 +417,129 @@ class TestOpenAIResponsesAPIConfig:
assert event.error.code == "unknown_error"
assert event.error.message == "Something went wrong"
def test_transform_streaming_response_missing_required_fields_response_created(
self,
):
"""Test that ResponseCreatedEvent with missing required fields (created_at,
output) does not crash but falls back to model_construct.
Reproduces https://github.com/BerriAI/litellm/issues/20570
"""
from litellm.types.llms.openai import ResponseCreatedEvent
# Minimal payload an OpenAI-compatible provider might send,
# omitting `created_at` and `output` inside the response object.
parsed_chunk = {
"type": "response.created",
"response": {
"id": "resp_q7BOLpck7clq",
"model": "gpt-oss-120b",
"status": "in_progress",
},
}
result = self.config.transform_streaming_response(
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
)
assert isinstance(result, ResponseCreatedEvent)
assert result.type == ResponsesAPIStreamEvents.RESPONSE_CREATED
assert result.response["id"] == "resp_q7BOLpck7clq"
def test_transform_streaming_response_missing_required_fields_output_text_delta(
self,
):
"""Test that OutputTextDeltaEvent with missing output_index and
content_index falls back to model_construct without crashing.
Reproduces https://github.com/BerriAI/litellm/issues/20570
"""
from litellm.types.llms.openai import OutputTextDeltaEvent
# Provider omits output_index and content_index
parsed_chunk = {
"type": "response.output_text.delta",
"item_id": "item_456",
"delta": "Hello",
}
result = self.config.transform_streaming_response(
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
)
assert isinstance(result, OutputTextDeltaEvent)
assert result.type == ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA
assert result.delta == "Hello"
assert result.item_id == "item_456"
def test_transform_streaming_response_missing_required_fields_content_part_added(
self,
):
"""Test that ContentPartAddedEvent with missing output_index and
content_index falls back to model_construct without crashing.
Reproduces https://github.com/BerriAI/litellm/issues/20570
"""
from litellm.types.llms.openai import ContentPartAddedEvent
# Provider omits output_index and content_index
parsed_chunk = {
"type": "response.content_part.added",
"item_id": "item_789",
"part": {"type": "output_text", "text": ""},
}
result = self.config.transform_streaming_response(
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
)
assert isinstance(result, ContentPartAddedEvent)
assert result.type == ResponsesAPIStreamEvents.CONTENT_PART_ADDED
assert result.item_id == "item_789"
def test_transform_streaming_response_missing_required_fields_output_item_added(
self,
):
"""Test that OutputItemAddedEvent with missing output_index falls back
to model_construct without crashing.
Reproduces https://github.com/BerriAI/litellm/issues/20570
"""
from litellm.types.llms.openai import OutputItemAddedEvent
# Provider omits output_index
parsed_chunk = {
"type": "response.output_item.added",
"item": {"type": "message", "id": "msg_001", "role": "assistant"},
}
result = self.config.transform_streaming_response(
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
)
assert isinstance(result, OutputItemAddedEvent)
assert result.type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED
def test_transform_streaming_response_valid_chunk_still_works(self):
"""Ensure that fully valid chunks still go through normal Pydantic
validation (not model_construct) and work correctly."""
parsed_chunk = {
"type": "response.output_text.delta",
"item_id": "item_123",
"output_index": 0,
"content_index": 0,
"delta": "World",
}
result = self.config.transform_streaming_response(
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
)
assert isinstance(result, OutputTextDeltaEvent)
assert result.delta == "World"
assert result.output_index == 0
assert result.content_index == 0
class TestAzureResponsesAPIConfig:
def setup_method(self):

View file

@ -3338,3 +3338,94 @@ def test_chunk_parser_handles_prompt_feedback_block_with_usage():
assert result.usage.completion_tokens == 0, f"completion_tokens should be 0, got {result.usage.completion_tokens}"
assert result.usage.total_tokens == 8175, f"total_tokens should be 8175, got {result.usage.total_tokens}"
def test_vertex_ai_traffic_type_preserved_in_hidden_params_streaming():
"""Test trafficType is preserved in _hidden_params for streaming."""
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
ModelResponseIterator,
)
chunk = {
"candidates": [{"content": {"parts": [{"text": "Hello"}]}}],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 200,
"totalTokenCount": 300,
"trafficType": "ON_DEMAND",
},
}
iterator = ModelResponseIterator(
streaming_response=[], sync_stream=True, logging_obj=MagicMock()
)
result = iterator.chunk_parser(chunk)
assert result._hidden_params["provider_specific_fields"]["traffic_type"] == "ON_DEMAND"
def test_vertex_ai_traffic_type_preserved_in_hidden_params_non_streaming():
"""Test trafficType is preserved in _hidden_params for non-streaming."""
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
VertexGeminiConfig,
)
completion_response = {
"candidates": [
{
"content": {"parts": [{"text": "Hello"}], "role": "model"},
"finishReason": "STOP",
}
],
"usageMetadata": {
"promptTokenCount": 50,
"candidatesTokenCount": 100,
"totalTokenCount": 150,
"trafficType": "PROVISIONED_THROUGHPUT",
},
}
raw_response = MagicMock()
raw_response.json.return_value = completion_response
result = VertexGeminiConfig().transform_response(
model="gemini-pro",
raw_response=raw_response,
model_response=ModelResponse(),
logging_obj=MagicMock(),
request_data={},
messages=[],
optional_params={},
litellm_params={},
encoding=None,
)
assert result._hidden_params["provider_specific_fields"]["traffic_type"] == "PROVISIONED_THROUGHPUT"
def test_vertex_ai_traffic_type_surfaced_in_responses_api():
"""Test trafficType is surfaced as provider_specific_fields in ResponsesAPIResponse."""
from litellm.responses.litellm_completion_transformation.transformation import (
LiteLLMCompletionResponsesConfig,
)
# Create a ModelResponse with provider_specific_fields in _hidden_params
from litellm.types.utils import Choices, Message
model_response = ModelResponse()
model_response._hidden_params["provider_specific_fields"] = {"traffic_type": "ON_DEMAND"}
model_response.choices = [
Choices(
message=Message(content="Hello", role="assistant"),
finish_reason="stop",
index=0,
)
]
responses_api_response = LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response(
request_input="test",
chat_completion_response=model_response,
responses_api_request={},
)
assert responses_api_response.provider_specific_fields["traffic_type"] == "ON_DEMAND"

View file

@ -936,6 +936,109 @@ def test_proxy_admin_viewer_can_access_global_spend_tags():
)
class TestModelsRouteExemptFromDisableLLMEndpoints:
"""
Test that /models and /v1/models are exempt from DISABLE_LLM_API_ENDPOINTS.
When DISABLE_LLM_API_ENDPOINTS is set, inference routes like /v1/chat/completions
should be blocked, but /models and /v1/models should remain accessible because
they are read-only model listing routes needed by the Admin UI.
Relevant issue: https://github.com/BerriAI/litellm/issues/new (UI breaks with DISABLE_LLM_ENDPOINTS)
"""
def _get_enterprise_route_checks(self):
"""Import EnterpriseRouteChecks from the local enterprise source file."""
import importlib.util
local_file = os.path.join(
os.path.dirname(__file__),
"..", "..", "..", "..", "enterprise",
"litellm_enterprise", "proxy", "auth", "route_checks.py",
)
local_file = os.path.abspath(local_file)
spec = importlib.util.spec_from_file_location(
"local_enterprise_route_checks", local_file
)
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
return mod.EnterpriseRouteChecks
@patch("litellm.proxy.proxy_server.premium_user", True)
def test_should_models_route_allowed_when_llm_api_disabled(self):
"""Test that /models is allowed even when LLM API routes are disabled"""
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with patch.object(
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
), patch.object(
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
):
# /models should NOT raise - it's exempt
EnterpriseRouteChecks.should_call_route("/models")
@patch("litellm.proxy.proxy_server.premium_user", True)
def test_should_v1_models_route_allowed_when_llm_api_disabled(self):
"""Test that /v1/models is allowed even when LLM API routes are disabled"""
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with patch.object(
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
), patch.object(
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
):
# /v1/models should NOT raise - it's exempt
EnterpriseRouteChecks.should_call_route("/v1/models")
@patch("litellm.proxy.proxy_server.premium_user", True)
def test_should_chat_completions_still_blocked_when_llm_api_disabled(self):
"""Test that non-exempt LLM routes like /v1/chat/completions are still blocked"""
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with patch.object(
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
), patch.object(
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
):
with pytest.raises(HTTPException) as exc_info:
EnterpriseRouteChecks.should_call_route("/v1/chat/completions")
assert exc_info.value.status_code == 403
assert "LLM API routes are disabled for this instance." in str(
exc_info.value.detail
)
@patch("litellm.proxy.proxy_server.premium_user", True)
def test_should_embeddings_still_blocked_when_llm_api_disabled(self):
"""Test that /v1/embeddings is still blocked when LLM API routes are disabled"""
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with patch.object(
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
), patch.object(
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
):
with pytest.raises(HTTPException) as exc_info:
EnterpriseRouteChecks.should_call_route("/v1/embeddings")
assert exc_info.value.status_code == 403
@patch("litellm.proxy.proxy_server.premium_user", True)
def test_should_models_route_allowed_when_llm_api_not_disabled(self):
"""Test that /models works normally when LLM API routes are not disabled"""
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with patch.object(
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=False
), patch.object(
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
):
# Should not raise
EnterpriseRouteChecks.should_call_route("/models")
EnterpriseRouteChecks.should_call_route("/v1/models")
def test_route_in_additional_public_routes_wildcard_match():
"""
Test that route_in_additonal_public_routes supports wildcard patterns.

View file

@ -697,3 +697,67 @@ def test_populate_request_with_path_params_does_not_overwrite_existing_values():
assert result["organization_id"] == "org-existing" # Should keep original, not "org-query-param"
# Verify other data is preserved
assert result["messages"] == [{"role": "user", "content": "Hello"}]
@pytest.mark.asyncio
async def test_request_body_with_html_script_tags():
"""
Test that JSON request bodies containing HTML tags like <script> are
parsed correctly without being blocked or modified.
Regression test for GitHub issue #20441:
https://github.com/BerriAI/litellm/issues/20441
LLM message content frequently contains HTML/code snippets.
The HTTP parsing layer must not interfere with such content.
"""
test_messages = [
{
"role": "user",
"content": "<script>alert('hello')</script>",
},
{
"role": "user",
"content": "<script> test </script>",
},
{
"role": "user",
"content": "Can you explain what <script> tags do in HTML?",
},
{
"role": "user",
"content": "Here is code: <div><script src='app.js'></script></div>",
},
{
"role": "user",
"content": "<img onerror='alert(1)' src='x'>",
},
{
"role": "user",
"content": "<iframe src='https://example.com'></iframe>",
},
]
for msg in test_messages:
test_payload = {
"model": "gpt-4o",
"messages": [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "Hello! How can I help?"},
msg,
],
}
mock_request = MagicMock()
mock_request.body = AsyncMock(return_value=orjson.dumps(test_payload))
mock_request.headers = {"content-type": "application/json"}
mock_request.scope = {}
result = await _read_request_body(mock_request)
assert result["model"] == "gpt-4o"
assert len(result["messages"]) == 3
assert result["messages"][2]["content"] == msg["content"], (
f"Message content with HTML was modified during parsing: "
f"expected={msg['content']!r}, got={result['messages'][2]['content']!r}"
)

View file

@ -986,3 +986,146 @@ class TestContentFilterGuardrail:
assert detail.get("category") == "harm_toxic_abuse"
else:
assert "harm_toxic_abuse" in str(detail)
async def test_html_tags_in_messages_not_blocked(self):
"""
Test that HTML tags like <script> in LLM message content are NOT blocked
by the content filter guardrail.
Regression test for GitHub issue #20441:
https://github.com/BerriAI/litellm/issues/20441
LLM message content is not rendered as HTML, so HTML tags should be
treated as plain text and should pass through without being blocked.
"""
# Set up a guardrail with all prebuilt patterns enabled as BLOCK
patterns = [
ContentFilterPattern(
pattern_type="prebuilt",
pattern_name="us_ssn",
action=ContentFilterAction.BLOCK,
),
ContentFilterPattern(
pattern_type="prebuilt",
pattern_name="email",
action=ContentFilterAction.BLOCK,
),
ContentFilterPattern(
pattern_type="prebuilt",
pattern_name="credit_card",
action=ContentFilterAction.BLOCK,
),
]
guardrail = ContentFilterGuardrail(
guardrail_name="test-html-tags",
patterns=patterns,
)
# Messages containing <script> and other HTML tags should NOT be blocked
html_messages = [
"<script>alert('hello')</script>",
"<script> test </script>",
"Can you explain what <script> tags do in HTML?",
"Here is some code: <div><script src='app.js'></script></div>",
"<img onerror='alert(1)' src='x'>",
"<iframe src='https://example.com'></iframe>",
"The <style> and <script> elements are important in HTML",
"<a href='javascript:void(0)'>click me</a>",
]
for message in html_messages:
# Should NOT raise HTTPException
result = await guardrail.apply_guardrail(
inputs={"texts": [message]},
request_data={},
input_type="request",
)
processed_texts = result.get("texts", [])
assert len(processed_texts) == 1
# Content should pass through unchanged (no HTML tags are patterns)
assert processed_texts[0] == message, (
f"Message containing HTML was unexpectedly modified: "
f"input={message!r}, output={processed_texts[0]!r}"
)
@pytest.mark.asyncio
async def test_script_tag_not_blocked_with_blocked_words(self):
"""
Test that <script> tags are not accidentally caught by blocked words
unless explicitly configured.
Regression test for GitHub issue #20441.
"""
blocked_words = [
BlockedWord(
keyword="confidential",
action=ContentFilterAction.BLOCK,
),
BlockedWord(
keyword="secret_project",
action=ContentFilterAction.BLOCK,
),
]
guardrail = ContentFilterGuardrail(
guardrail_name="test-script-not-blocked",
blocked_words=blocked_words,
)
# <script> should not be caught by unrelated blocked words
script_messages = [
"<script>alert('test')</script>",
"How do I use <script> tags in HTML?",
"<script src='app.js'></script>",
]
for message in script_messages:
result = await guardrail.apply_guardrail(
inputs={"texts": [message]},
request_data={},
input_type="request",
)
processed_texts = result.get("texts", [])
assert len(processed_texts) == 1
assert processed_texts[0] == message
def test_no_builtin_pattern_matches_script_tag(self):
"""
Test that NONE of the prebuilt patterns in patterns.json match
the string '<script>' or common HTML tags.
This is a safeguard to ensure that future pattern additions
do not accidentally block legitimate LLM content containing
HTML/code snippets.
Regression test for GitHub issue #20441.
"""
from litellm.proxy.guardrails.guardrail_hooks.litellm_content_filter.patterns import (
PREBUILT_PATTERNS,
get_compiled_pattern,
)
html_test_strings = [
"<script>alert('xss')</script>",
"<script> test </script>",
"<script src='app.js'></script>",
"<img onerror='alert(1)' src='x'>",
"<iframe src='https://example.com'></iframe>",
"<style>body { color: red; }</style>",
"<div onclick='alert(1)'>click</div>",
]
for pattern_name in PREBUILT_PATTERNS:
compiled = get_compiled_pattern(pattern_name)
for test_string in html_test_strings:
match = compiled.search(test_string)
if match:
# Some patterns may legitimately match substrings
# (e.g., URL pattern matching src='https://...')
# but they should not match the script/HTML tag itself
matched_text = match.group()
assert "<script" not in matched_text.lower(), (
f"Pattern '{pattern_name}' matched '<script>' in "
f"test string: {test_string!r}. "
f"LLM message content should not be blocked for HTML tags."
)

View file

@ -7,7 +7,6 @@ import sys
sys.path.insert(0, os.path.abspath("../../../../../.."))
import asyncio
from unittest.mock import MagicMock, patch
import pytest
@ -26,7 +25,7 @@ async def test_openai_moderation_guardrail_init():
guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
)
assert guardrail.guardrail_name == "test-openai-moderation"
assert guardrail.api_key == "test-key"
assert guardrail.model == "omni-moderation-latest"
@ -49,27 +48,27 @@ async def test_openai_moderation_guardrail_adds_to_litellm_callbacks():
# Clear existing callbacks for clean test
original_callbacks = litellm.callbacks.copy()
litellm.logging_callback_manager._reset_all_callbacks()
try:
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
guardrail_litellm_params = LitellmParams(
guardrail=SupportedGuardrailIntegrations.OPENAI_MODERATION,
api_key="test-key",
model="omni-moderation-latest",
mode="pre_call"
mode="pre_call",
)
guardrail = openai_initialize_guardrail(
litellm_params=guardrail_litellm_params,
guardrail=Guardrail(
guardrail_name="test-openai-moderation",
litellm_params=guardrail_litellm_params
)
litellm_params=guardrail_litellm_params,
),
)
# Check that the guardrail was added to litellm callbacks
assert guardrail in litellm.callbacks
assert len(litellm.callbacks) == 1
# Verify it's the correct guardrail
callback = litellm.callbacks[0]
assert isinstance(callback, OpenAIModerationGuardrail)
@ -85,12 +84,12 @@ async def test_openai_moderation_guardrail_adds_to_litellm_callbacks():
async def test_openai_moderation_guardrail_safe_content():
"""Test OpenAI moderation guardrail with safe content via apply_guardrail"""
from litellm.types.utils import GenericGuardrailAPIInputs
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
)
# Mock safe moderation response
mock_response = OpenAIModerationResponse(
id="modr-123",
@ -118,25 +117,29 @@ async def test_openai_moderation_guardrail_safe_content():
"harassment": [],
"self-harm": [],
"violence": [],
}
},
)
]
],
)
with patch.object(guardrail, 'async_make_request', return_value=mock_response):
with patch.object(guardrail, "async_make_request", return_value=mock_response):
# Test apply_guardrail with safe content using structured_messages
inputs = GenericGuardrailAPIInputs(
structured_messages=[
{"role": "user", "content": "Hello, how are you today?"}
]
)
result = await guardrail.apply_guardrail(
inputs=inputs,
request_data={"messages": [{"role": "user", "content": "Hello, how are you today?"}]},
input_type="request"
request_data={
"messages": [
{"role": "user", "content": "Hello, how are you today?"}
]
},
input_type="request",
)
# Should return the original inputs unchanged
assert result == inputs
@ -145,12 +148,12 @@ async def test_openai_moderation_guardrail_safe_content():
async def test_openai_moderation_guardrail_apply_guardrail():
"""Test OpenAI moderation guardrail apply_guardrail method (unified guardrail interface)"""
from litellm.types.utils import GenericGuardrailAPIInputs
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
)
# Mock safe moderation response
mock_response = OpenAIModerationResponse(
id="modr-123",
@ -178,37 +181,37 @@ async def test_openai_moderation_guardrail_apply_guardrail():
"harassment": [],
"self-harm": [],
"violence": [],
}
},
)
]
],
)
with patch.object(guardrail, 'async_make_request', return_value=mock_response):
with patch.object(guardrail, "async_make_request", return_value=mock_response):
# Test apply_guardrail with texts (embeddings-style input)
inputs = GenericGuardrailAPIInputs(
texts=["Hello, how are you?", "What is the weather?"]
)
result = await guardrail.apply_guardrail(
inputs=inputs,
request_data={},
input_type="request",
)
# Should return inputs unchanged (moderation doesn't modify, only blocks)
assert result == inputs
@pytest.mark.asyncio
@pytest.mark.asyncio
async def test_openai_moderation_guardrail_harmful_content():
"""Test OpenAI moderation guardrail with harmful content via apply_guardrail"""
from litellm.types.utils import GenericGuardrailAPIInputs
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
)
# Mock harmful moderation response
mock_response = OpenAIModerationResponse(
id="modr-123",
@ -236,40 +239,51 @@ async def test_openai_moderation_guardrail_harmful_content():
"harassment": [],
"self-harm": [],
"violence": [],
}
},
)
]
],
)
with patch.object(guardrail, 'async_make_request', return_value=mock_response):
with patch.object(guardrail, "async_make_request", return_value=mock_response):
# Test apply_guardrail with harmful content using structured_messages
inputs = GenericGuardrailAPIInputs(
structured_messages=[
{"role": "user", "content": "This is hateful content"}
]
)
# Should raise HTTPException
from fastapi import HTTPException
with pytest.raises(HTTPException) as exc_info:
await guardrail.apply_guardrail(
inputs=inputs,
request_data={"messages": [{"role": "user", "content": "This is hateful content"}]},
input_type="request"
request_data={
"messages": [
{"role": "user", "content": "This is hateful content"}
]
},
input_type="request",
)
assert exc_info.value.status_code == 400
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)
@pytest.mark.asyncio
async def test_openai_moderation_guardrail_streaming_safe_content():
"""Test OpenAI moderation guardrail with streaming safe content"""
"""Test OpenAI moderation guardrail with streaming safe content via UnifiedLLMGuardrails"""
from litellm.proxy.guardrails.guardrail_hooks.unified_guardrail.unified_guardrail import (
UnifiedLLMGuardrails,
)
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
event_hook="post_call",
)
unified_guardrail = UnifiedLLMGuardrails()
# Mock safe moderation response
mock_response = OpenAIModerationResponse(
id="modr-123",
@ -297,72 +311,85 @@ async def test_openai_moderation_guardrail_streaming_safe_content():
"harassment": [],
"self-harm": [],
"violence": [],
}
},
)
]
],
)
# Mock streaming chunks
async def mock_stream():
# Simulate streaming chunks with safe content
chunks = [
MagicMock(choices=[MagicMock(delta=MagicMock(content="Hello "))]),
MagicMock(choices=[MagicMock(delta=MagicMock(content="world"))]),
MagicMock(choices=[MagicMock(delta=MagicMock(content="!"))])
]
for chunk in chunks:
chunk1 = MagicMock()
chunk1.model = "gpt-4"
chunk1.choices = [MagicMock()]
chunk1.choices[0].delta = MagicMock()
chunk1.choices[0].delta.content = "Hello "
chunk1.choices[0].finish_reason = None
chunk2 = MagicMock()
chunk2.model = "gpt-4"
chunk2.choices = [MagicMock()]
chunk2.choices[0].delta = MagicMock()
chunk2.choices[0].delta.content = "world"
chunk2.choices[0].finish_reason = None
# Last chunk with finish_reason
chunk3 = MagicMock()
chunk3.model = "gpt-4"
chunk3.choices = [MagicMock()]
chunk3.choices[0].delta = MagicMock()
chunk3.choices[0].delta.content = "!"
chunk3.choices[0].finish_reason = "stop"
for chunk in [chunk1, chunk2, chunk3]:
yield chunk
# Mock the stream_chunk_builder to return a proper ModelResponse
# Mock for stream_chunk_builder
mock_model_response = MagicMock()
mock_model_response.choices = [
MagicMock(message=MagicMock(content="Hello world!"))
]
with patch.object(guardrail, 'async_make_request', return_value=mock_response), \
patch('litellm.main.stream_chunk_builder', return_value=mock_model_response), \
patch('litellm.llms.base_llm.base_model_iterator.MockResponseIterator') as mock_iterator:
# Mock the iterator to yield the original chunks
async def mock_yield_chunks():
chunks = [
MagicMock(choices=[MagicMock(delta=MagicMock(content="Hello "))]),
MagicMock(choices=[MagicMock(delta=MagicMock(content="world"))]),
MagicMock(choices=[MagicMock(delta=MagicMock(content="!"))])
]
for chunk in chunks:
yield chunk
mock_iterator.return_value.__aiter__ = lambda self: mock_yield_chunks()
user_api_key_dict = UserAPIKeyAuth(api_key="test")
mock_model_response.choices = [MagicMock()]
mock_model_response.choices[0].message = MagicMock()
mock_model_response.choices[0].message.content = "Hello world!"
with patch.object(guardrail, "async_make_request", return_value=mock_response), patch(
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
return_value=mock_model_response,
):
user_api_key_dict = UserAPIKeyAuth(
api_key="test", request_route="/chat/completions"
)
request_data = {
"messages": [
{"role": "user", "content": "Hello, how are you today?"}
]
"messages": [{"role": "user", "content": "Hello, how are you today?"}],
"guardrail_to_apply": guardrail,
"metadata": {"guardrails": ["test-openai-moderation"]},
}
# Test streaming hook with safe content
# Test streaming hook with safe content via UnifiedLLMGuardrails
result_chunks = []
async for chunk in guardrail.async_post_call_streaming_iterator_hook(
async for chunk in unified_guardrail.async_post_call_streaming_iterator_hook(
user_api_key_dict=user_api_key_dict,
response=mock_stream(),
request_data=request_data
request_data=request_data,
):
result_chunks.append(chunk)
# Should return all chunks without blocking
assert len(result_chunks) == 3
@pytest.mark.asyncio
async def test_openai_moderation_guardrail_streaming_harmful_content():
"""Test OpenAI moderation guardrail with streaming harmful content"""
"""Test OpenAI moderation guardrail with streaming harmful content via UnifiedLLMGuardrails"""
from litellm.proxy.guardrails.guardrail_hooks.unified_guardrail.unified_guardrail import (
UnifiedLLMGuardrails,
)
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
event_hook="post_call",
)
unified_guardrail = UnifiedLLMGuardrails()
# Mock harmful moderation response
mock_response = OpenAIModerationResponse(
id="modr-123",
@ -390,46 +417,74 @@ async def test_openai_moderation_guardrail_streaming_harmful_content():
"harassment": [],
"self-harm": [],
"violence": [],
}
},
)
]
],
)
# Mock streaming chunks with harmful content
async def mock_stream():
chunks = [
MagicMock(choices=[MagicMock(delta=MagicMock(content="This is "))]),
MagicMock(choices=[MagicMock(delta=MagicMock(content="harmful content"))])
]
for chunk in chunks:
# First chunk - no finish_reason
chunk1 = MagicMock()
chunk1.model = "gpt-4"
chunk1.choices = [MagicMock()]
chunk1.choices[0].delta = MagicMock()
chunk1.choices[0].delta.content = "This is "
chunk1.choices[0].finish_reason = None
# Last chunk - with finish_reason to signal end of stream
chunk2 = MagicMock()
chunk2.model = "gpt-4"
chunk2.choices = [MagicMock()]
chunk2.choices[0].delta = MagicMock()
chunk2.choices[0].delta.content = "harmful content"
chunk2.choices[0].finish_reason = "stop"
for chunk in [chunk1, chunk2]:
yield chunk
# Mock the stream_chunk_builder to return a ModelResponse with harmful content
mock_model_response = MagicMock()
mock_model_response.choices = [
MagicMock(message=MagicMock(content="This is harmful content"))
]
with patch.object(guardrail, 'async_make_request', return_value=mock_response), \
patch('litellm.main.stream_chunk_builder', return_value=mock_model_response):
user_api_key_dict = UserAPIKeyAuth(api_key="test")
# Mock for stream_chunk_builder - use real litellm types so isinstance checks pass
from litellm.types.utils import ModelResponse
import litellm
mock_model_response = ModelResponse(
id="mock-response",
model="gpt-4",
choices=[
litellm.Choices(
index=0,
message=litellm.Message(
role="assistant",
content="This is harmful content",
),
finish_reason="stop",
)
],
)
with patch.object(guardrail, "async_make_request", return_value=mock_response), patch(
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
return_value=mock_model_response,
):
user_api_key_dict = UserAPIKeyAuth(
api_key="test", request_route="/chat/completions"
)
request_data = {
"messages": [
{"role": "user", "content": "Generate harmful content"}
]
"messages": [{"role": "user", "content": "Generate harmful content"}],
"guardrail_to_apply": guardrail,
"metadata": {"guardrails": ["test-openai-moderation"]},
}
# Should raise HTTPException when processing streaming harmful content
from fastapi import HTTPException
with pytest.raises(HTTPException) as exc_info:
result_chunks = []
async for chunk in guardrail.async_post_call_streaming_iterator_hook(
async for chunk in unified_guardrail.async_post_call_streaming_iterator_hook(
user_api_key_dict=user_api_key_dict,
response=mock_stream(),
request_data=request_data
request_data=request_data,
):
result_chunks.append(chunk)
assert exc_info.value.status_code == 400
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)

View file

@ -0,0 +1,172 @@
import pytest
from unittest.mock import MagicMock, patch
import os
from litellm.proxy.guardrails.guardrail_hooks.openai.moderations import (
OpenAIModerationGuardrail,
)
from litellm.proxy.guardrails.guardrail_hooks.unified_guardrail.unified_guardrail import (
UnifiedLLMGuardrails,
)
from litellm.types.utils import ModelResponseStream, ModelResponse
from litellm.proxy._types import UserAPIKeyAuth
@pytest.mark.asyncio
async def test_openai_moderation_guardrail_streaming_latency():
"""
Test that the OpenAI Moderation guardrail, when run via UnifiedLLMGuardrails,
supports streaming (fast time-to-first-token) instead of buffering.
"""
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
# 1. Initialize the specific guardrail with proper event_hook
openai_guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
event_hook="post_call",
)
# 2. Initialize the Unified Guardrail system (which invokes the specific guardrail)
unified_guardrail = UnifiedLLMGuardrails()
# Mock safe moderation response
mock_mod_response = MagicMock()
mock_mod_response.results = []
# Mock streaming chunks (no artificial delay - test deterministically)
async def mock_stream():
chunks_data = ["Hello", " ", "world", "!", " Goodbye"]
for i, content in enumerate(chunks_data):
chunk = MagicMock(spec=ModelResponseStream)
chunk.model = "gpt-4"
choice = MagicMock()
choice.delta = MagicMock()
choice.delta.content = content
# Last chunk gets finish_reason
choice.finish_reason = "stop" if i == len(chunks_data) - 1 else None
chunk.choices = [choice]
yield chunk
# Mock for stream_chunk_builder to return a simple ModelResponse
mock_model_response = MagicMock(spec=ModelResponse)
mock_model_response.choices = [MagicMock()]
mock_model_response.choices[0].message = MagicMock()
mock_model_response.choices[0].message.content = "Hello world! Goodbye"
# Patch the network call in the specific guardrail
with patch.object(
openai_guardrail, "async_make_request", return_value=mock_mod_response
), patch(
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
return_value=mock_model_response,
):
user_api_key_dict = UserAPIKeyAuth(
api_key="test", request_route="/chat/completions"
)
request_data = {
"messages": [{"role": "user", "content": "hi"}],
"guardrail_to_apply": openai_guardrail,
"metadata": {
"guardrails": ["test-openai-moderation"],
"guardrail_config": {"streaming_sampling_rate": 1},
}, # Check every chunk for test
}
chunks_received = 0
first_chunk_yielded = False
# Call the hook on UnifiedLLMGuardrails
async for chunk in unified_guardrail.async_post_call_streaming_iterator_hook(
user_api_key_dict=user_api_key_dict,
response=mock_stream(),
request_data=request_data,
):
if not first_chunk_yielded:
first_chunk_yielded = True
chunks_received += 1
# Deterministic assertions (no flaky timing checks)
assert first_chunk_yielded, "Expected at least one chunk to be yielded"
assert chunks_received == 5, f"Expected 5 chunks, got {chunks_received}"
@pytest.mark.asyncio
async def test_openai_moderation_guardrail_streaming_harmful_content():
"""
Test that harmful content is caught during streaming via UnifiedLLMGuardrails
"""
from fastapi import HTTPException
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
openai_guardrail = OpenAIModerationGuardrail(
guardrail_name="test-openai-moderation",
event_hook="post_call",
)
unified_guardrail = UnifiedLLMGuardrails()
# Mock harmful moderation response
mock_mod_response = MagicMock()
mock_mod_response.results = [
MagicMock(
flagged=True, categories={"hate": True}, category_scores={"hate": 0.99}
)
]
async def mock_stream():
chunks_data = ["This ", "is ", "harmful ", "content"]
for i, content in enumerate(chunks_data):
chunk = MagicMock(spec=ModelResponseStream)
chunk.model = "gpt-4"
choice = MagicMock()
choice.delta = MagicMock()
choice.delta.content = content
# Last chunk gets finish_reason
choice.finish_reason = "stop" if i == len(chunks_data) - 1 else None
chunk.choices = [choice]
yield chunk
# Mock for stream_chunk_builder - use real litellm types so isinstance checks pass
import litellm
mock_model_response = ModelResponse(
id="mock-response",
model="gpt-4",
choices=[
litellm.Choices(
index=0,
message=litellm.Message(
role="assistant",
content="This is harmful content",
),
finish_reason="stop",
)
],
)
with patch.object(
openai_guardrail, "async_make_request", return_value=mock_mod_response
), patch(
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
return_value=mock_model_response,
):
user_api_key_dict = UserAPIKeyAuth(
api_key="test", request_route="/chat/completions"
)
request_data = {
"messages": [{"role": "user", "content": "generate hate"}],
"guardrail_to_apply": openai_guardrail,
"metadata": {
"guardrails": ["test-openai-moderation"],
"guardrail_config": {"streaming_sampling_rate": 1},
},
}
# Should raise HTTPException
with pytest.raises(HTTPException) as exc_info:
async for _ in unified_guardrail.async_post_call_streaming_iterator_hook(
user_api_key_dict=user_api_key_dict,
response=mock_stream(),
request_data=request_data,
):
pass
assert exc_info.value.status_code == 400
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)

View file

@ -1122,6 +1122,83 @@ async def test_model_armor_non_model_response():
assert not guardrail.async_handler.post.called
@pytest.mark.asyncio
async def test_model_armor_guardrail_status_intervened_vs_failed():
"""
regression test for bug where _process_error always set 'guardrail_failed_to_respond'
even for intentional blocks (error 400).
"""
mock_user_api_key_dict = UserAPIKeyAuth()
mock_cache = MagicMock(spec=DualCache)
#1: Blocked content should raise exception and show guardrail status: guardrail_intervened"
guardrail = ModelArmorGuardrail(
template_id="test-template",
project_id="test-project",
location="us-central1",
guardrail_name="model-armor-test",
)
mock_response = AsyncMock()
mock_response.status_code = 200
mock_response.json = AsyncMock(return_value={
"sanitizationResult": {
"filterMatchState": "MATCH_FOUND",
"filterResults": {
"rai": {
"raiFilterResult": {
"matchState": "MATCH_FOUND",
}
}
}
}
})
guardrail._ensure_access_token_async = AsyncMock(return_value=("token", "test-project"))
with patch.object(guardrail.async_handler, "post", AsyncMock(return_value=mock_response)):
request_data = {
"model": "gpt-4",
"messages": [{"role": "user", "content": "bad content"}],
"metadata": {"guardrails": ["model-armor-test"]},
}
with pytest.raises(HTTPException):
await guardrail.async_pre_call_hook(
user_api_key_dict=mock_user_api_key_dict,
cache=mock_cache,
data=request_data,
call_type="completion",
)
info = request_data["metadata"]["standard_logging_guardrail_information"]
assert info[0]["guardrail_status"] == "guardrail_intervened"
#2: if an API error - guardrail status should be guardrail_failed_to_respond"
guardrail2 = ModelArmorGuardrail(
template_id="test-template",
project_id="test-project",
location="us-central1",
guardrail_name="model-armor-test2",
fail_on_error=True,
)
guardrail2._ensure_access_token_async = AsyncMock(side_effect=ConnectionError("timeout"))
request_data2 = {
"model": "gpt-4",
"messages": [{"role": "user", "content": "hello"}],
"metadata": {"guardrails": ["model-armor-test2"]},
}
with pytest.raises(ConnectionError):
await guardrail2.async_pre_call_hook(
user_api_key_dict=mock_user_api_key_dict,
cache=mock_cache,
data=request_data2,
call_type="completion",
)
info2 = request_data2["metadata"]["standard_logging_guardrail_information"]
assert info2[0]["guardrail_status"] == "guardrail_failed_to_respond"
def mock_open(read_data=''):
"""Helper to create a mock file object"""
import io

View file

@ -0,0 +1,136 @@
"""
Tests that API keys are masked in error responses.
When an invalid/malformed API key is sent (e.g., with a leading space or
wrong prefix), the error response must NOT return the key in plain text.
Instead, it should show only the first 4 and last 4 characters with ****
in the middle.
"""
import pytest
class TestKeyMaskingInAuthErrors:
"""Test that user_api_key_auth masks keys in validation error messages."""
def test_assert_message_masks_key_without_sk_prefix(self):
"""
When a key doesn't start with 'sk-', the AssertionError message
should contain a masked version, not the full key.
"""
from litellm.proxy.auth.auth_utils import abbreviate_api_key
# Simulate the logic from user_api_key_auth.py
api_key = "my-secret-api-key-1234567890abcdef"
_masked_key = (
"{}****{}".format(api_key[:4], api_key[-4:])
if len(api_key) > 8
else "****"
)
# The masked key should NOT contain the full original key
assert api_key not in _masked_key
# Should show first 4 and last 4 chars
assert _masked_key == "my-s****cdef"
def test_assert_message_masks_key_with_leading_space(self):
"""
Reported case: key with leading space like ' sk-abc123...'
"""
api_key = " sk-abc123def456ghi789jkl012mno345pqr"
_masked_key = (
"{}****{}".format(api_key[:4], api_key[-4:])
if len(api_key) > 8
else "****"
)
assert api_key not in _masked_key
assert _masked_key == " sk-****5pqr"
def test_assert_message_masks_short_key(self):
"""Short keys (<=8 chars) should be fully masked."""
api_key = "short"
_masked_key = (
"{}****{}".format(api_key[:4], api_key[-4:])
if len(api_key) > 8
else "****"
)
assert _masked_key == "****"
def test_key_not_starting_with_sk_raises_masked_error(self):
"""
Verify the assert message format contains masked key, not the original.
Note: Python's AssertionError str(e) includes the expression + message,
but the *message* part (which is what gets passed to ProxyException)
should only contain the masked key.
"""
api_key = "bad-key-format-1234567890abcdefghijklmnop"
_masked_key = (
"{}****{}".format(api_key[:4], api_key[-4:])
if len(api_key) > 8
else "****"
)
# Build the same message string that user_api_key_auth.py would produce
error_message = "LiteLLM Virtual Key expected. Received={}, expected to start with 'sk-'.".format(
_masked_key
)
# The full key must NOT appear in the message
assert api_key not in error_message
# The masked version should appear
assert _masked_key in error_message
# Should still have helpful context
assert "expected to start with 'sk-'" in error_message
class TestKeyMaskingInKeyManagement:
"""Test that key_management_endpoints masks keys in validation errors."""
def test_invalid_key_format_error_is_masked(self):
"""
When creating a key that doesn't start with 'sk-', the error
should not include the full key value.
"""
key_value = "bad-prefix-1234567890abcdefghijklmnop"
_masked = (
"{}****{}".format(key_value[:4], key_value[-4:])
if len(key_value) > 8
else "****"
)
error_msg = f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {_masked}"
# Full key must not appear
assert key_value not in error_msg
# Masked version should appear
assert _masked in error_msg
assert "bad-****mnop" in error_msg
class TestPresidioErrorSanitization:
"""Test that Presidio errors don't leak request text containing keys."""
def test_analyze_text_error_does_not_leak_text(self):
"""
If Presidio analyzer fails, the error message should NOT contain
the original text that was being analyzed.
"""
# Simulate what happens: user message contains an API key,
# Presidio fails, error message should be sanitized
original_text = "Please use this key: sk-secret1234567890abcdefghijklmnop"
# The sanitized exception from our fix
sanitized_error = f"Presidio PII analysis failed: ConnectionError"
assert original_text not in sanitized_error
assert "sk-secret1234567890abcdefghijklmnop" not in sanitized_error
def test_anonymize_text_error_does_not_leak_text(self):
"""
If Presidio anonymizer fails, the error should be sanitized.
"""
sanitized_error = f"Presidio PII anonymization failed: ClientError"
assert "sk-" not in sanitized_error
assert "api_key" not in sanitized_error

View file

@ -2,6 +2,7 @@ import base64
import json
import os
import sys
from unittest.mock import MagicMock, patch
import pytest
from fastapi.testclient import TestClient
@ -352,3 +353,34 @@ class TestResponsesAPIProviderSpecificParams:
# Should not raise any exception
result = ResponsesAPIRequestUtils.get_requested_response_api_optional_param(params)
assert "temperature" in result
def test_responses_extra_body_forwarded_to_completion_transformation_handler():
"""
Regression test: extra_body must be forwarded to response_api_handler
when responses_api_provider_config is None (completion transformation path).
Before the fix, extra_body was a named parameter of responses() but was
not passed to litellm_completion_transformation_handler.response_api_handler(),
so it was silently dropped.
"""
with patch(
"litellm.responses.main.ProviderConfigManager.get_provider_responses_api_config",
return_value=None,
), patch(
"litellm.responses.main.litellm_completion_transformation_handler.response_api_handler",
) as mock_handler:
mock_handler.return_value = MagicMock()
litellm.responses(
model="openai/gpt-4o",
input="Hello",
extra_body={"custom_key": "custom_value"},
)
mock_handler.assert_called_once()
call_kwargs = mock_handler.call_args
# extra_body can be a positional or keyword arg; check both
assert call_kwargs.kwargs.get("extra_body") == {
"custom_key": "custom_value"
}

View file

@ -0,0 +1,232 @@
import pytest
import litellm
from litellm.caching.caching import DualCache
from litellm.router_strategy.budget_limiter import RouterBudgetLimiting
from litellm.types.router import LiteLLM_Params
from litellm.types.utils import BudgetConfig
@pytest.fixture
def disable_budget_sync(monkeypatch):
async def noop(*args, **kwargs):
return None
monkeypatch.setattr(
"litellm.router_strategy.budget_limiter.RouterBudgetLimiting.periodic_sync_in_memory_spend_with_redis",
noop,
)
@pytest.mark.asyncio
async def test_get_llm_provider_for_deployment_dict_does_not_require_litellm_params_instantiation(
disable_budget_sync, monkeypatch
):
class RaiseOnInit:
def __init__(self, *args, **kwargs):
raise AssertionError("LiteLLM_Params should not be instantiated in hot path")
monkeypatch.setattr(
"litellm.router_strategy.budget_limiter.LiteLLM_Params",
RaiseOnInit,
)
provider_budget = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config={},
)
deployment = {"litellm_params": {"model": "openai/gpt-4o-mini"}}
provider = provider_budget._get_llm_provider_for_deployment(deployment)
assert provider == "openai"
@pytest.mark.asyncio
async def test_get_llm_provider_for_deployment_dict_view_supports_mapping_and_attr_access(
disable_budget_sync, monkeypatch
):
observed = {}
def _future_style_get_llm_provider(
model,
custom_llm_provider=None,
api_base=None,
api_key=None,
litellm_params=None,
):
assert litellm_params is not None
observed["model_attr"] = litellm_params.model
observed["provider_get"] = litellm_params.get("custom_llm_provider")
observed["api_base_item"] = litellm_params["api_base"]
observed["has_api_key"] = "api_key" in litellm_params
observed["model_dump"] = litellm_params.model_dump()
return model, "openai", None, None
monkeypatch.setattr(
"litellm.router_strategy.budget_limiter.litellm.get_llm_provider",
_future_style_get_llm_provider,
)
provider_budget = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config={},
)
deployment = {
"litellm_params": {
"model": "openai/gpt-4o-mini",
"custom_llm_provider": "openai",
"api_base": "https://api.openai.com/v1",
}
}
provider = provider_budget._get_llm_provider_for_deployment(deployment)
assert provider == "openai"
assert observed["model_attr"] == "openai/gpt-4o-mini"
assert observed["provider_get"] == "openai"
assert observed["api_base_item"] == "https://api.openai.com/v1"
assert observed["has_api_key"] is False
assert observed["model_dump"]["model"] == "openai/gpt-4o-mini"
@pytest.mark.asyncio
async def test_async_filter_deployments_resolves_provider_once_per_deployment(
disable_budget_sync, monkeypatch
):
provider_budget = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config={
"openai": BudgetConfig(budget_duration="1d", max_budget=100.0),
},
)
healthy_deployments = [
{
"model_name": "gpt-4o-mini",
"litellm_params": {"model": "openai/gpt-4o-mini"},
"model_info": {"id": "deployment-1"},
},
{
"model_name": "gpt-4o-mini",
"litellm_params": {"model": "openai/gpt-4o-mini"},
"model_info": {"id": "deployment-2"},
},
]
provider_resolution_calls = 0
def _count_provider_calls(deployment):
nonlocal provider_resolution_calls
provider_resolution_calls += 1
return "openai"
monkeypatch.setattr(
provider_budget,
"_get_llm_provider_for_deployment",
_count_provider_calls,
)
filtered_deployments = await provider_budget.async_filter_deployments(
model="gpt-4o-mini",
healthy_deployments=healthy_deployments,
messages=[],
request_kwargs={},
parent_otel_span=None,
)
assert len(filtered_deployments) == len(healthy_deployments)
assert provider_resolution_calls == len(healthy_deployments)
@pytest.mark.asyncio
async def test_async_filter_deployments_does_not_recompute_provider_when_resolved_none(
disable_budget_sync, monkeypatch
):
provider_budget = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config={
"openai": BudgetConfig(budget_duration="1d", max_budget=100.0),
},
model_list=[
{
"model_name": "gpt-4o-mini",
"litellm_params": {
"model": "openai/gpt-4o-mini",
"max_budget": 100.0,
"budget_duration": "1d",
},
"model_info": {"id": "deployment-1"},
}
],
)
healthy_deployments = [
{
"model_name": "gpt-4o-mini",
"litellm_params": {"model": "unknown-provider/model"},
"model_info": {"id": "deployment-1"},
}
]
provider_resolution_calls = 0
def _provider_returns_none(deployment):
nonlocal provider_resolution_calls
provider_resolution_calls += 1
return None
monkeypatch.setattr(
provider_budget,
"_get_llm_provider_for_deployment",
_provider_returns_none,
)
filtered_deployments = await provider_budget.async_filter_deployments(
model="gpt-4o-mini",
healthy_deployments=healthy_deployments,
messages=[],
request_kwargs={},
parent_otel_span=None,
)
assert len(filtered_deployments) == len(healthy_deployments)
assert provider_resolution_calls == len(healthy_deployments)
def _legacy_provider_resolution(deployment):
"""
Reference implementation used before hot-path optimization.
"""
try:
_litellm_params = LiteLLM_Params(**deployment.get("litellm_params", {"model": ""}))
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
model=_litellm_params.model,
litellm_params=_litellm_params,
)
except Exception:
return None
return custom_llm_provider
@pytest.mark.parametrize(
"deployment",
[
{"litellm_params": {"model": "openai/gpt-4o-mini"}},
{"litellm_params": {"model": "gpt-4o-mini", "custom_llm_provider": "openai"}},
{"litellm_params": {"model": "unknown-provider/model"}},
],
)
@pytest.mark.asyncio
async def test_get_llm_provider_for_deployment_matches_legacy_behavior(
disable_budget_sync, deployment
):
provider_budget = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config={},
)
current_provider = provider_budget._get_llm_provider_for_deployment(deployment)
legacy_provider = _legacy_provider_resolution(deployment)
assert current_provider == legacy_provider

View file

@ -0,0 +1,5 @@
<svg width="50" height="41" viewBox="0 0 35 22" fill="none" xmlns="http://www.w3.org/2000/svg">
<path d="M34.9305 11.1952C35.4985 14.7139 32.4864 16.6962 29.4313 16.9252C27.4867 20.976 20.6324 23.7107 14.6971 20.762C12.1552 21.1535 10.764 20.5092 9.6896 19.2752C11.861 16.4425 18.4366 11.3459 26.47 13.9755C30.7569 15.3849 31.8715 12.0843 30.7961 10.8513C26.7505 6.19284 17.6134 10.3865 17.2879 10.7353C20.8729 5.16699 33.7065 3.60491 34.9305 11.1952ZM22.6298 4.80521C22.6522 4.79728 19.7125 3.74467 15.6034 5.54462C15.4367 5.4832 15.2735 5.41239 15.1146 5.33251C19.0701 2.66529 22.5359 1.53437 25.5001 1.97445C23.7042 -0.11988 15.0833 -1.65817 10.1432 3.36208C4.03683 2.14096 -0.233535 7.38224 0.00989967 12.1765C0.253334 16.9708 5.41043 19.9502 8.34533 19.1424C8.41578 19.1335 8.48704 19.1335 8.55748 19.1424C9.21348 15.9766 11.8658 8.61029 22.6298 4.80521Z" fill="#2160E1"/>
</svg>
<!--65 41, 50 32-->

After

Width:  |  Height:  |  Size: 906 B

View file

@ -104,6 +104,7 @@ export const shouldRenderContentFilterConfigSettings = (provider: string | null)
const asset_logos_folder = "../ui/assets/logos/";
export const guardrailLogoMap: Record<string, string> = {
"Zscaler AI Guard": `${asset_logos_folder}zscaler.svg`,
"Presidio PII": `${asset_logos_folder}presidio.png`,
"Bedrock Guardrail": `${asset_logos_folder}bedrock.svg`,
Lakera: `${asset_logos_folder}lakeraai.jpeg`,