mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
Merge branch 'main' into litellm_litellm_anthropic_remote_url3
This commit is contained in:
commit
4a9d851dc4
53 changed files with 4013 additions and 467 deletions
|
|
@ -40,38 +40,33 @@ outputs:
|
|||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Helm | Setup
|
||||
uses: azure/setup-helm@v4
|
||||
with:
|
||||
version: v3.20.0
|
||||
|
||||
- name: Helm | Login
|
||||
shell: bash
|
||||
run: echo ${{ inputs.registry_password }} | helm registry login -u ${{ inputs.registry_username }} --password-stdin ${{ inputs.registry }}
|
||||
env:
|
||||
HELM_EXPERIMENTAL_OCI: '1'
|
||||
|
||||
|
||||
- name: Helm | Dependency
|
||||
if: inputs.update_dependencies == 'true'
|
||||
shell: bash
|
||||
run: helm dependency update ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
|
||||
env:
|
||||
HELM_EXPERIMENTAL_OCI: '1'
|
||||
|
||||
- name: Helm | Package
|
||||
shell: bash
|
||||
run: helm package ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} --version ${{ inputs.tag }} --app-version ${{ inputs.app_version }}
|
||||
env:
|
||||
HELM_EXPERIMENTAL_OCI: '1'
|
||||
|
||||
- name: Helm | Push
|
||||
shell: bash
|
||||
run: helm push ${{ inputs.name }}-${{ inputs.tag }}.tgz oci://${{ inputs.registry }}/${{ inputs.repository }}
|
||||
env:
|
||||
HELM_EXPERIMENTAL_OCI: '1'
|
||||
|
||||
- name: Helm | Logout
|
||||
shell: bash
|
||||
run: helm registry logout ${{ inputs.registry }}
|
||||
env:
|
||||
HELM_EXPERIMENTAL_OCI: '1'
|
||||
|
||||
- name: Helm | Output
|
||||
id: output
|
||||
shell: bash
|
||||
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT
|
||||
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT
|
||||
|
|
|
|||
|
|
@ -26,6 +26,10 @@ version: 1.1.0
|
|||
# It is recommended to use it with quotes.
|
||||
appVersion: v1.80.12
|
||||
|
||||
annotations:
|
||||
org.opencontainers.image.source: "https://github.com/BerriAI/litellm"
|
||||
org.opencontainers.image.url: "https://docs.litellm.ai/"
|
||||
|
||||
dependencies:
|
||||
- name: "postgresql"
|
||||
version: ">=13.3.0"
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ Each provider uses their own search backend:
|
|||
|
||||
| Provider | Search Engine | Notes |
|
||||
|----------|---------------|-------|
|
||||
| **OpenAI** (`gpt-4o-search-preview`, `gpt-4o-mini-search-preview`, `gpt-5-search-api`) | OpenAI's internal search | Real-time web data |
|
||||
| **OpenAI** (`gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview`) | OpenAI's internal search | Real-time web data |
|
||||
| **xAI** (`grok-3`) | xAI's search + X/Twitter | Real-time social media data |
|
||||
| **Google AI/Vertex** (`gemini-2.0-flash`) | **Google Search** | Uses actual Google search results |
|
||||
| **Anthropic** (`claude-3-5-sonnet`) | Anthropic's web search | Real-time web data |
|
||||
|
|
@ -45,6 +45,19 @@ Use `web_search_options` when you need to:
|
|||
**Anthropic Web Search Models**: Claude models that support web search: `claude-3-5-sonnet-latest`, `claude-3-5-sonnet-20241022`, `claude-3-5-haiku-latest`, `claude-3-5-haiku-20241022`, `claude-3-7-sonnet-20250219`
|
||||
:::
|
||||
|
||||
## OpenAI Web Search: Two Approaches
|
||||
|
||||
OpenAI offers two distinct ways to use web search depending on the endpoint and model:
|
||||
|
||||
| Approach | Endpoint | Models | How to enable |
|
||||
|----------|----------|--------|---------------|
|
||||
| **Search Models** | `/chat/completions` | `gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview` | Pass `web_search_options` parameter |
|
||||
| **Web Search Tool** | `/responses` | `gpt-5`, `gpt-4.1`, `gpt-4o`, and other regular models | Pass `web_search_preview` tool |
|
||||
|
||||
:::tip Search models search automatically
|
||||
Search models like `gpt-5-search-api` **automatically search the web** even without the `web_search_options` parameter. Use `web_search_options` to set `search_context_size` (`"low"`, `"medium"`, `"high"`) or specify `user_location` for localized results.
|
||||
:::
|
||||
|
||||
## `/chat/completions` (litellm.completion)
|
||||
|
||||
### Quick Start
|
||||
|
|
@ -56,7 +69,7 @@ Use `web_search_options` when you need to:
|
|||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="openai/gpt-4o-search-preview",
|
||||
model="openai/gpt-5-search-api",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -76,31 +89,36 @@ response = completion(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
# OpenAI
|
||||
# OpenAI search models
|
||||
- model_name: gpt-5-search-api
|
||||
litellm_params:
|
||||
model: openai/gpt-5-search-api
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: gpt-4o-search-preview
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-search-preview
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
|
||||
# xAI
|
||||
- model_name: grok-3
|
||||
litellm_params:
|
||||
model: xai/grok-3
|
||||
api_key: os.environ/XAI_API_KEY
|
||||
|
||||
|
||||
# Anthropic
|
||||
- model_name: claude-3-5-sonnet-latest
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
|
||||
# VertexAI
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
model: gemini-2.0-flash
|
||||
vertex_project: your-project-id
|
||||
vertex_location: us-central1
|
||||
|
||||
|
||||
# Google AI Studio
|
||||
- model_name: gemini-2-flash-studio
|
||||
litellm_params:
|
||||
|
|
@ -108,13 +126,13 @@ model_list:
|
|||
api_key: os.environ/GOOGLE_API_KEY
|
||||
```
|
||||
|
||||
2. Start the proxy
|
||||
2. Start the proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
|
@ -126,13 +144,18 @@ client = OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="grok-3", # or any other web search enabled model
|
||||
model="gpt-5-search-api", # or any other web search enabled model
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What was a positive news story from today?"
|
||||
}
|
||||
]
|
||||
],
|
||||
extra_body={
|
||||
"web_search_options": {
|
||||
"search_context_size": "medium"
|
||||
}
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -149,7 +172,7 @@ from litellm import completion
|
|||
|
||||
# Customize search context size
|
||||
response = completion(
|
||||
model="openai/gpt-4o-search-preview",
|
||||
model="openai/gpt-5-search-api",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -257,6 +280,12 @@ response = client.chat.completions.create(
|
|||
|
||||
## `/responses` (litellm.responses)
|
||||
|
||||
Use the `web_search_preview` tool with models like `gpt-5`, `gpt-4.1`, `gpt-4o`, etc.
|
||||
|
||||
:::info
|
||||
Search-dedicated models like `gpt-5-search-api` and `gpt-4o-search-preview` do **not** support the `/responses` endpoint. Use them with `/chat/completions` + `web_search_options` instead (see above).
|
||||
:::
|
||||
|
||||
### Quick Start
|
||||
|
||||
<Tabs>
|
||||
|
|
@ -266,18 +295,14 @@ response = client.chat.completions.create(
|
|||
from litellm import responses
|
||||
|
||||
response = responses(
|
||||
model="openai/gpt-4o",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What was a positive news story from today?"
|
||||
}
|
||||
],
|
||||
model="openai/gpt-5",
|
||||
input="What is the capital of France?",
|
||||
tools=[{
|
||||
"type": "web_search_preview" # enables web search with default medium context size
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
|
|
@ -285,19 +310,24 @@ response = responses(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-4o
|
||||
- model_name: gpt-5
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
model: openai/gpt-5
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: gpt-4.1
|
||||
litellm_params:
|
||||
model: openai/gpt-4.1
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start the proxy
|
||||
2. Start the proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
|
@ -309,11 +339,11 @@ client = OpenAI(
|
|||
)
|
||||
|
||||
response = client.responses.create(
|
||||
model="gpt-4o",
|
||||
model="gpt-5",
|
||||
tools=[{
|
||||
"type": "web_search_preview"
|
||||
}],
|
||||
input="What was a positive news story from today?",
|
||||
input="What is the capital of France?",
|
||||
)
|
||||
|
||||
print(response.output_text)
|
||||
|
|
@ -331,13 +361,8 @@ from litellm import responses
|
|||
|
||||
# Customize search context size
|
||||
response = responses(
|
||||
model="openai/gpt-4o",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What was a positive news story from today?"
|
||||
}
|
||||
],
|
||||
model="openai/gpt-5",
|
||||
input="What is the capital of France?",
|
||||
tools=[{
|
||||
"type": "web_search_preview",
|
||||
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
||||
|
|
@ -358,12 +383,12 @@ client = OpenAI(
|
|||
|
||||
# Customize search context size
|
||||
response = client.responses.create(
|
||||
model="gpt-4o",
|
||||
model="gpt-5",
|
||||
tools=[{
|
||||
"type": "web_search_preview",
|
||||
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
||||
}],
|
||||
input="What was a positive news story from today?",
|
||||
input="What is the capital of France?",
|
||||
)
|
||||
|
||||
print(response.output_text)
|
||||
|
|
@ -417,14 +442,14 @@ model_list:
|
|||
web_search_options:
|
||||
search_context_size: "high" # Options: "low", "medium", "high"
|
||||
|
||||
# Different context size for different models
|
||||
- model_name: gpt-4o-search-preview
|
||||
# OpenAI search model with custom context size
|
||||
- model_name: gpt-5-search-api
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-search-preview
|
||||
model: openai/gpt-5-search-api
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
web_search_options:
|
||||
search_context_size: "low"
|
||||
|
||||
|
||||
# Gemini with medium context (default)
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
|
|
@ -449,6 +474,7 @@ Use `litellm.supports_web_search(model="model_name")` -> returns `True` if model
|
|||
|
||||
```python showLineNumbers
|
||||
# Check OpenAI models
|
||||
assert litellm.supports_web_search(model="openai/gpt-5-search-api") == True
|
||||
assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
|
||||
|
||||
# Check xAI models
|
||||
|
|
@ -472,13 +498,20 @@ assert litellm.supports_web_search(model="gemini/gemini-2.0-flash") == True
|
|||
```yaml
|
||||
model_list:
|
||||
# OpenAI
|
||||
- model_name: gpt-5-search-api
|
||||
litellm_params:
|
||||
model: openai/gpt-5-search-api
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
- model_name: gpt-4o-search-preview
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-search-preview
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
|
||||
# xAI
|
||||
- model_name: grok-3
|
||||
litellm_params:
|
||||
|
|
@ -533,6 +566,12 @@ Expected Response
|
|||
```json showLineNumbers
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"model_group": "gpt-5-search-api",
|
||||
"providers": ["openai"],
|
||||
"max_tokens": 128000,
|
||||
"supports_web_search": true
|
||||
},
|
||||
{
|
||||
"model_group": "gpt-4o-search-preview",
|
||||
"providers": ["openai"],
|
||||
|
|
|
|||
|
|
@ -230,7 +230,70 @@ os.environ["OPENAI_BASE_URL"] = "https://your_host/v1" # OPTIONAL
|
|||
|
||||
These also support the `OPENAI_BASE_URL` environment variable, which can be used to specify a custom API endpoint.
|
||||
|
||||
## OpenAI Vision Models
|
||||
### OpenAI Web Search Models
|
||||
|
||||
OpenAI has two ways to use web search, depending on the endpoint:
|
||||
|
||||
| Approach | Endpoint | Models | How to enable |
|
||||
|----------|----------|--------|---------------|
|
||||
| **Search Models** | `/chat/completions` | `gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview` | Pass `web_search_options` parameter |
|
||||
| **Web Search Tool** | `/responses` | `gpt-5`, `gpt-4.1`, `gpt-4o`, and other regular models | Pass `web_search_preview` tool |
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk-completion" label="SDK - /chat/completions">
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="openai/gpt-5-search-api",
|
||||
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
||||
web_search_options={
|
||||
"search_context_size": "medium" # Options: "low", "medium", "high"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk-responses" label="SDK - /responses">
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import responses
|
||||
|
||||
response = responses(
|
||||
model="openai/gpt-5",
|
||||
input="What is the capital of France?",
|
||||
tools=[{
|
||||
"type": "web_search_preview",
|
||||
"search_context_size": "low"
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Search model for /chat/completions
|
||||
- model_name: gpt-5-search-api
|
||||
litellm_params:
|
||||
model: openai/gpt-5-search-api
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
# Regular model for /responses with web_search_preview tool
|
||||
- model_name: gpt-5
|
||||
litellm_params:
|
||||
model: openai/gpt-5
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
For full details, see the [Web Search guide](../completion/web_search.md).
|
||||
|
||||
## OpenAI Vision Models
|
||||
| Model Name | Function Call |
|
||||
|-----------------------|-----------------------------------------------------------------|
|
||||
| gpt-4o | `response = completion(model="gpt-4o", messages=messages)` |
|
||||
|
|
|
|||
|
|
@ -37,6 +37,24 @@ for event in response:
|
|||
print(event)
|
||||
```
|
||||
|
||||
#### Web Search
|
||||
```python showLineNumbers title="OpenAI Responses with Web Search"
|
||||
import litellm
|
||||
|
||||
response = litellm.responses(
|
||||
model="openai/gpt-5",
|
||||
input="What is the capital of France?",
|
||||
tools=[{
|
||||
"type": "web_search_preview",
|
||||
"search_context_size": "medium" # Options: "low", "medium", "high"
|
||||
}]
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
For full details, see the [Web Search guide](../../completion/web_search.md).
|
||||
|
||||
#### Image Generation with Streaming
|
||||
```python showLineNumbers title="OpenAI Streaming Image Generation"
|
||||
import litellm
|
||||
|
|
|
|||
|
|
@ -41,6 +41,10 @@ class EnterpriseRouteChecks:
|
|||
|
||||
return get_secret_bool("DISABLE_ADMIN_ENDPOINTS") is True
|
||||
|
||||
# Routes that should remain accessible even when LLM API endpoints are disabled.
|
||||
# These are read-only model listing routes needed by the Admin UI.
|
||||
LLM_API_EXEMPT_ROUTES = ["/models", "/v1/models"]
|
||||
|
||||
@staticmethod
|
||||
def should_call_route(route: str):
|
||||
"""
|
||||
|
|
@ -58,6 +62,7 @@ class EnterpriseRouteChecks:
|
|||
)
|
||||
elif (
|
||||
RouteChecks.is_llm_api_route(route=route)
|
||||
and route not in EnterpriseRouteChecks.LLM_API_EXEMPT_ROUTES
|
||||
and EnterpriseRouteChecks.is_llm_api_route_disabled()
|
||||
):
|
||||
raise HTTPException(
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@
|
|||
"mcp-client-2025-11-20": "mcp-client-2025-11-20",
|
||||
"mcp-client-2025-04-04": "mcp-client-2025-04-04",
|
||||
"mcp-servers-2025-12-04": null,
|
||||
"oauth-2025-04-20": "oauth-2025-04-20",
|
||||
"output-128k-2025-02-19": "output-128k-2025-02-19",
|
||||
"prompt-caching-scope-2026-01-05": "prompt-caching-scope-2026-01-05",
|
||||
"skills-2025-10-02": "skills-2025-10-02",
|
||||
|
|
|
|||
|
|
@ -12,7 +12,8 @@ import asyncio
|
|||
import time
|
||||
import traceback
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import TYPE_CHECKING, Any, List, Optional, Union
|
||||
from threading import Lock
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.caching import RedisPipelineIncrementOperation
|
||||
|
|
@ -71,6 +72,7 @@ class DualCache(BaseCache):
|
|||
self.last_redis_batch_access_time = LimitedSizeOrderedDict(
|
||||
max_size=default_max_redis_batch_cache_size
|
||||
)
|
||||
self._last_redis_batch_access_time_lock = Lock()
|
||||
self.redis_batch_cache_expiry = (
|
||||
default_redis_batch_cache_expiry
|
||||
or litellm.default_redis_batch_cache_expiry
|
||||
|
|
@ -236,22 +238,46 @@ class DualCache(BaseCache):
|
|||
except Exception:
|
||||
verbose_logger.error(traceback.format_exc())
|
||||
|
||||
def get_redis_batch_keys(
|
||||
def _reserve_redis_batch_keys(
|
||||
self,
|
||||
current_time: float,
|
||||
keys: List[str],
|
||||
result: List[Any],
|
||||
) -> List[str]:
|
||||
sublist_keys = []
|
||||
for key, value in zip(keys, result):
|
||||
if value is None:
|
||||
) -> Tuple[List[str], Dict[str, Optional[float]]]:
|
||||
"""
|
||||
Atomically choose keys to fetch from Redis and reserve their access time.
|
||||
This prevents check-then-act races under concurrent async callers.
|
||||
"""
|
||||
sublist_keys: List[str] = []
|
||||
previous_access_times: Dict[str, Optional[float]] = {}
|
||||
|
||||
with self._last_redis_batch_access_time_lock:
|
||||
for key, value in zip(keys, result):
|
||||
if value is not None:
|
||||
continue
|
||||
|
||||
if (
|
||||
key not in self.last_redis_batch_access_time
|
||||
or current_time - self.last_redis_batch_access_time[key]
|
||||
>= self.redis_batch_cache_expiry
|
||||
):
|
||||
sublist_keys.append(key)
|
||||
return sublist_keys
|
||||
previous_access_times[key] = self.last_redis_batch_access_time.get(
|
||||
key
|
||||
)
|
||||
self.last_redis_batch_access_time[key] = current_time
|
||||
|
||||
return sublist_keys, previous_access_times
|
||||
|
||||
def _rollback_redis_batch_key_reservations(
|
||||
self, previous_access_times: Dict[str, Optional[float]]
|
||||
) -> None:
|
||||
with self._last_redis_batch_access_time_lock:
|
||||
for key, previous_time in previous_access_times.items():
|
||||
if previous_time is None:
|
||||
self.last_redis_batch_access_time.pop(key, None)
|
||||
else:
|
||||
self.last_redis_batch_access_time[key] = previous_time
|
||||
|
||||
async def async_batch_get_cache(
|
||||
self,
|
||||
|
|
@ -276,19 +302,23 @@ class DualCache(BaseCache):
|
|||
- check the redis cache
|
||||
"""
|
||||
current_time = time.time()
|
||||
sublist_keys = self.get_redis_batch_keys(current_time, keys, result)
|
||||
sublist_keys, previous_access_times = self._reserve_redis_batch_keys(
|
||||
current_time, keys, result
|
||||
)
|
||||
|
||||
# Only hit Redis if the last access time was more than 5 seconds ago
|
||||
# Only hit Redis if enough time has passed since last access.
|
||||
if len(sublist_keys) > 0:
|
||||
# If not found in in-memory cache, try fetching from Redis
|
||||
redis_result = await self.redis_cache.async_batch_get_cache(
|
||||
sublist_keys, parent_otel_span=parent_otel_span
|
||||
)
|
||||
|
||||
# Update the last access time for ALL queried keys
|
||||
# This includes keys with None values to throttle repeated Redis queries
|
||||
for key in sublist_keys:
|
||||
self.last_redis_batch_access_time[key] = current_time
|
||||
try:
|
||||
# If not found in in-memory cache, try fetching from Redis
|
||||
redis_result = await self.redis_cache.async_batch_get_cache(
|
||||
sublist_keys, parent_otel_span=parent_otel_span
|
||||
)
|
||||
except Exception:
|
||||
# Do not throttle subsequent callers if the Redis read fails.
|
||||
self._rollback_redis_batch_key_reservations(
|
||||
previous_access_times
|
||||
)
|
||||
raise
|
||||
|
||||
# Short-circuit if redis_result is None or contains only None values
|
||||
if redis_result is None or all(v is None for v in redis_result.values()):
|
||||
|
|
|
|||
|
|
@ -1016,10 +1016,12 @@ BEDROCK_EMBEDDING_PROVIDERS_LITERAL = Literal[
|
|||
|
||||
BEDROCK_CONVERSE_MODELS = [
|
||||
"qwen.qwen3-coder-480b-a35b-v1:0",
|
||||
"qwen.qwen3-coder-next",
|
||||
"qwen.qwen3-235b-a22b-2507-v1:0",
|
||||
"qwen.qwen3-coder-30b-a3b-v1:0",
|
||||
"qwen.qwen3-32b-v1:0",
|
||||
"deepseek.v3-v1:0",
|
||||
"deepseek.v3.2",
|
||||
"openai.gpt-oss-20b-1:0",
|
||||
"openai.gpt-oss-120b-1:0",
|
||||
"anthropic.claude-haiku-4-5-20251001-v1:0",
|
||||
|
|
@ -1062,6 +1064,8 @@ BEDROCK_CONVERSE_MODELS = [
|
|||
"amazon.nova-pro-v1:0",
|
||||
"writer.palmyra-x4-v1:0",
|
||||
"writer.palmyra-x5-v1:0",
|
||||
"minimax.minimax-m2.1",
|
||||
"moonshotai.kimi-k2.5",
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ from litellm.types.utils import (
|
|||
LLMResponseTypes,
|
||||
StandardLoggingGuardrailInformation,
|
||||
)
|
||||
from fastapi.exceptions import HTTPException
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
|
@ -650,6 +651,23 @@ class CustomGuardrail(CustomLogger):
|
|||
)
|
||||
return response
|
||||
|
||||
@staticmethod
|
||||
def _is_guardrail_intervention(e: Exception) -> bool:
|
||||
"""
|
||||
Returns True if the exception represents an intentional guardrail block
|
||||
(this was logged previously as an API failure - guardrail_failed_to_respond).
|
||||
|
||||
Guardrails signal intentional blocks by raising:
|
||||
- HTTPException with status 400 (content policy violation)
|
||||
- ModifyResponseException (passthrough mode violation)
|
||||
"""
|
||||
|
||||
if isinstance(e, ModifyResponseException):
|
||||
return True
|
||||
if isinstance(e, HTTPException) and e.status_code == 400:
|
||||
return True
|
||||
return False
|
||||
|
||||
def _process_error(
|
||||
self,
|
||||
e: Exception,
|
||||
|
|
@ -664,6 +682,11 @@ class CustomGuardrail(CustomLogger):
|
|||
|
||||
This gets logged on downsteam Langfuse, DataDog, etc.
|
||||
"""
|
||||
guardrail_status: GuardrailStatus = (
|
||||
"guardrail_intervened"
|
||||
if self._is_guardrail_intervention(e)
|
||||
else "guardrail_failed_to_respond"
|
||||
)
|
||||
# For custom_code_guardrail scenario, log as "deny" instead of full exception
|
||||
# Check if this is from custom_code_guardrail by checking the class name
|
||||
guardrail_response: Union[Exception, str] = e
|
||||
|
|
@ -673,7 +696,7 @@ class CustomGuardrail(CustomLogger):
|
|||
self.add_standard_logging_guardrail_information_to_request_data(
|
||||
guardrail_json_response=guardrail_response,
|
||||
request_data=request_data,
|
||||
guardrail_status="guardrail_failed_to_respond",
|
||||
guardrail_status=guardrail_status,
|
||||
duration=duration,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
|
|
|
|||
|
|
@ -2331,7 +2331,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
result, LiteLLMBatch
|
||||
):
|
||||
litellm_params = self.litellm_params or {}
|
||||
litellm_metadata = litellm_params.get("litellm_metadata", {})
|
||||
litellm_metadata = litellm_params.get("litellm_metadata") or {}
|
||||
if (
|
||||
litellm_metadata.get("batch_ignore_default_logging", False) is True
|
||||
): # polling job will query these frequently, don't spam db logs
|
||||
|
|
@ -3127,7 +3127,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
self, dynamic_success_callbacks: Optional[List], global_callbacks: List
|
||||
) -> List:
|
||||
if dynamic_success_callbacks is None:
|
||||
return global_callbacks
|
||||
return list(global_callbacks)
|
||||
return list(set(dynamic_success_callbacks + global_callbacks))
|
||||
|
||||
def _remove_internal_litellm_callbacks(self, callbacks: List) -> List:
|
||||
|
|
|
|||
|
|
@ -38,9 +38,18 @@ def optionally_handle_anthropic_oauth(
|
|||
Returns:
|
||||
Tuple of (updated headers, api_key)
|
||||
"""
|
||||
# Check Authorization header (passthrough / forwarded requests)
|
||||
auth_header = headers.get("authorization", "")
|
||||
if auth_header and auth_header.startswith(f"Bearer {ANTHROPIC_OAUTH_TOKEN_PREFIX}"):
|
||||
api_key = auth_header.replace("Bearer ", "")
|
||||
headers.pop("x-api-key", None)
|
||||
headers["anthropic-beta"] = ANTHROPIC_OAUTH_BETA_HEADER
|
||||
headers["anthropic-dangerous-direct-browser-access"] = "true"
|
||||
return headers, api_key
|
||||
# Check api_key directly (standard chat/completion flow)
|
||||
if api_key and api_key.startswith(ANTHROPIC_OAUTH_TOKEN_PREFIX):
|
||||
headers.pop("x-api-key", None)
|
||||
headers["authorization"] = f"Bearer {api_key}"
|
||||
headers["anthropic-beta"] = ANTHROPIC_OAUTH_BETA_HEADER
|
||||
headers["anthropic-dangerous-direct-browser-access"] = "true"
|
||||
return headers, api_key
|
||||
|
|
@ -108,7 +117,9 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
if tools is None:
|
||||
return False
|
||||
for tool in tools:
|
||||
if "type" in tool and tool["type"].startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value):
|
||||
if "type" in tool and tool["type"].startswith(
|
||||
ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
|
@ -134,111 +145,126 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
"""
|
||||
if not tools:
|
||||
return False
|
||||
|
||||
|
||||
for tool in tools:
|
||||
tool_type = tool.get("type", "")
|
||||
if tool_type in ["tool_search_tool_regex_20251119", "tool_search_tool_bm25_20251119"]:
|
||||
if tool_type in [
|
||||
"tool_search_tool_regex_20251119",
|
||||
"tool_search_tool_bm25_20251119",
|
||||
]:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_programmatic_tool_calling_used(self, tools: Optional[List]) -> bool:
|
||||
"""
|
||||
Check if programmatic tool calling is being used (tools with allowed_callers field).
|
||||
|
||||
|
||||
Returns True if any tool has allowed_callers containing 'code_execution_20250825'.
|
||||
"""
|
||||
if not tools:
|
||||
return False
|
||||
|
||||
|
||||
for tool in tools:
|
||||
# Check top-level allowed_callers
|
||||
allowed_callers = tool.get("allowed_callers", None)
|
||||
if allowed_callers and isinstance(allowed_callers, list):
|
||||
if "code_execution_20250825" in allowed_callers:
|
||||
return True
|
||||
|
||||
|
||||
# Check function.allowed_callers for OpenAI format tools
|
||||
function = tool.get("function", {})
|
||||
if isinstance(function, dict):
|
||||
function_allowed_callers = function.get("allowed_callers", None)
|
||||
if function_allowed_callers and isinstance(function_allowed_callers, list):
|
||||
if function_allowed_callers and isinstance(
|
||||
function_allowed_callers, list
|
||||
):
|
||||
if "code_execution_20250825" in function_allowed_callers:
|
||||
return True
|
||||
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def is_input_examples_used(self, tools: Optional[List]) -> bool:
|
||||
"""
|
||||
Check if input_examples is being used in any tools.
|
||||
|
||||
|
||||
Returns True if any tool has input_examples field.
|
||||
"""
|
||||
if not tools:
|
||||
return False
|
||||
|
||||
|
||||
for tool in tools:
|
||||
# Check top-level input_examples
|
||||
input_examples = tool.get("input_examples", None)
|
||||
if input_examples and isinstance(input_examples, list) and len(input_examples) > 0:
|
||||
if (
|
||||
input_examples
|
||||
and isinstance(input_examples, list)
|
||||
and len(input_examples) > 0
|
||||
):
|
||||
return True
|
||||
|
||||
|
||||
# Check function.input_examples for OpenAI format tools
|
||||
function = tool.get("function", {})
|
||||
if isinstance(function, dict):
|
||||
function_input_examples = function.get("input_examples", None)
|
||||
if function_input_examples and isinstance(function_input_examples, list) and len(function_input_examples) > 0:
|
||||
if (
|
||||
function_input_examples
|
||||
and isinstance(function_input_examples, list)
|
||||
and len(function_input_examples) > 0
|
||||
):
|
||||
return True
|
||||
|
||||
|
||||
return False
|
||||
|
||||
def is_effort_used(self, optional_params: Optional[dict], model: Optional[str] = None) -> bool:
|
||||
|
||||
def is_effort_used(
|
||||
self, optional_params: Optional[dict], model: Optional[str] = None
|
||||
) -> bool:
|
||||
"""
|
||||
Check if effort parameter is being used.
|
||||
|
||||
|
||||
Returns True if effort-related parameters are present.
|
||||
"""
|
||||
if not optional_params:
|
||||
return False
|
||||
|
||||
|
||||
# Check if reasoning_effort is provided for Claude Opus 4.5
|
||||
if model and ("opus-4-5" in model.lower() or "opus_4_5" in model.lower()):
|
||||
reasoning_effort = optional_params.get("reasoning_effort")
|
||||
if reasoning_effort and isinstance(reasoning_effort, str):
|
||||
return True
|
||||
|
||||
|
||||
# Check if output_config is directly provided
|
||||
output_config = optional_params.get("output_config")
|
||||
if output_config and isinstance(output_config, dict):
|
||||
effort = output_config.get("effort")
|
||||
if effort and isinstance(effort, str):
|
||||
return True
|
||||
|
||||
|
||||
return False
|
||||
|
||||
def is_code_execution_tool_used(self, tools: Optional[List]) -> bool:
|
||||
"""
|
||||
Check if code execution tool is being used.
|
||||
|
||||
|
||||
Returns True if any tool has type "code_execution_20250825".
|
||||
"""
|
||||
if not tools:
|
||||
return False
|
||||
|
||||
|
||||
for tool in tools:
|
||||
tool_type = tool.get("type", "")
|
||||
if tool_type == "code_execution_20250825":
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_container_with_skills_used(self, optional_params: Optional[dict]) -> bool:
|
||||
"""
|
||||
Check if container with skills is being used.
|
||||
|
||||
|
||||
Returns True if optional_params contains container with skills.
|
||||
"""
|
||||
if not optional_params:
|
||||
return False
|
||||
|
||||
|
||||
container = optional_params.get("container")
|
||||
if container and isinstance(container, dict):
|
||||
skills = container.get("skills")
|
||||
|
|
@ -256,10 +282,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
def get_computer_tool_beta_header(self, computer_tool_version: str) -> str:
|
||||
"""
|
||||
Get the appropriate beta header for a given computer tool version.
|
||||
|
||||
|
||||
Args:
|
||||
computer_tool_version: The computer tool version (e.g., 'computer_20250124', 'computer_20241022')
|
||||
|
||||
|
||||
Returns:
|
||||
The corresponding beta header string
|
||||
"""
|
||||
|
|
@ -282,37 +308,37 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
) -> List[str]:
|
||||
"""
|
||||
Get list of common beta headers based on the features that are active.
|
||||
|
||||
|
||||
Returns:
|
||||
List of beta header strings
|
||||
"""
|
||||
from litellm.types.llms.anthropic import (
|
||||
ANTHROPIC_EFFORT_BETA_HEADER,
|
||||
)
|
||||
|
||||
|
||||
betas = []
|
||||
|
||||
|
||||
# Detect features
|
||||
effort_used = self.is_effort_used(optional_params, model)
|
||||
|
||||
|
||||
if effort_used:
|
||||
betas.append(ANTHROPIC_EFFORT_BETA_HEADER) # effort-2025-11-24
|
||||
|
||||
|
||||
if computer_tool_used:
|
||||
beta_header = self.get_computer_tool_beta_header(computer_tool_used)
|
||||
betas.append(beta_header)
|
||||
|
||||
|
||||
# Anthropic no longer requires the prompt-caching beta header
|
||||
# Prompt caching now works automatically when cache_control is used in messages
|
||||
# Reference: https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
|
||||
|
||||
|
||||
if file_id_used:
|
||||
betas.append("files-api-2025-04-14")
|
||||
betas.append("code-execution-2025-05-22")
|
||||
|
||||
|
||||
if mcp_server_used:
|
||||
betas.append("mcp-client-2025-04-04")
|
||||
|
||||
|
||||
return list(set(betas))
|
||||
|
||||
def get_anthropic_headers(
|
||||
|
|
@ -351,27 +377,35 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
# Tool search, programmatic tool calling, and input_examples all use the same beta header
|
||||
if tool_search_used or programmatic_tool_calling_used or input_examples_used:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_TOOL_SEARCH_BETA_HEADER
|
||||
|
||||
betas.add(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
|
||||
|
||||
|
||||
# Effort parameter uses a separate beta header
|
||||
if effort_used:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_EFFORT_BETA_HEADER
|
||||
|
||||
betas.add(ANTHROPIC_EFFORT_BETA_HEADER)
|
||||
|
||||
|
||||
# Code execution tool uses a separate beta header
|
||||
if code_execution_tool_used:
|
||||
betas.add("code-execution-2025-08-25")
|
||||
|
||||
|
||||
# Container with skills uses a separate beta header
|
||||
if container_with_skills_used:
|
||||
betas.add("skills-2025-10-02")
|
||||
|
||||
_is_oauth = api_key and api_key.startswith(ANTHROPIC_OAUTH_TOKEN_PREFIX)
|
||||
headers = {
|
||||
"anthropic-version": anthropic_version or "2023-06-01",
|
||||
"x-api-key": api_key,
|
||||
"accept": "application/json",
|
||||
"content-type": "application/json",
|
||||
}
|
||||
if _is_oauth:
|
||||
headers["authorization"] = f"Bearer {api_key}"
|
||||
headers["anthropic-dangerous-direct-browser-access"] = "true"
|
||||
betas.add(ANTHROPIC_OAUTH_BETA_HEADER)
|
||||
else:
|
||||
headers["x-api-key"] = api_key
|
||||
|
||||
if user_anthropic_beta_headers is not None:
|
||||
betas.update(user_anthropic_beta_headers)
|
||||
|
|
@ -381,7 +415,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
# Vertex AI requires web search beta header for web search to work
|
||||
if web_search_tool_used:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES
|
||||
headers["anthropic-beta"] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
|
||||
|
||||
headers[
|
||||
"anthropic-beta"
|
||||
] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
|
||||
elif len(betas) > 0:
|
||||
headers["anthropic-beta"] = ",".join(betas)
|
||||
|
||||
|
|
@ -398,7 +435,9 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
api_base: Optional[str] = None,
|
||||
) -> Dict:
|
||||
# Check for Anthropic OAuth token in headers
|
||||
headers, api_key = optionally_handle_anthropic_oauth(headers=headers, api_key=api_key)
|
||||
headers, api_key = optionally_handle_anthropic_oauth(
|
||||
headers=headers, api_key=api_key
|
||||
)
|
||||
if api_key is None:
|
||||
raise litellm.AuthenticationError(
|
||||
message="Missing Anthropic API Key - A call is being made to anthropic but no key is set either in the environment variables or via params. Please set `ANTHROPIC_API_KEY` in your environment vars",
|
||||
|
|
@ -416,11 +455,15 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
file_id_used = self.is_file_id_used(messages=messages)
|
||||
web_search_tool_used = self.is_web_search_tool_used(tools=tools)
|
||||
tool_search_used = self.is_tool_search_used(tools=tools)
|
||||
programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(tools=tools)
|
||||
programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(
|
||||
tools=tools
|
||||
)
|
||||
input_examples_used = self.is_input_examples_used(tools=tools)
|
||||
effort_used = self.is_effort_used(optional_params=optional_params, model=model)
|
||||
code_execution_tool_used = self.is_code_execution_tool_used(tools=tools)
|
||||
container_with_skills_used = self.is_container_with_skills_used(optional_params=optional_params)
|
||||
container_with_skills_used = self.is_container_with_skills_used(
|
||||
optional_params=optional_params
|
||||
)
|
||||
user_anthropic_beta_headers = self._get_user_anthropic_beta_headers(
|
||||
anthropic_beta_header=headers.get("anthropic-beta")
|
||||
)
|
||||
|
|
@ -499,7 +542,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
def get_token_counter(self) -> Optional[BaseTokenCounter]:
|
||||
"""
|
||||
Factory method to create an Anthropic token counter.
|
||||
|
||||
|
||||
Returns:
|
||||
AnthropicTokenCounter instance for this provider.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -49,15 +49,15 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
# TODO: Add Anthropic `metadata` support
|
||||
# "metadata",
|
||||
]
|
||||
|
||||
|
||||
@staticmethod
|
||||
def _filter_billing_headers_from_system(system_param):
|
||||
"""
|
||||
Filter out x-anthropic-billing-header metadata from system parameter.
|
||||
|
||||
|
||||
Args:
|
||||
system_param: Can be a string or a list of system message content blocks
|
||||
|
||||
|
||||
Returns:
|
||||
Filtered system parameter (string or list), or None if all content was filtered
|
||||
"""
|
||||
|
|
@ -74,7 +74,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
text = content_block.get("text", "")
|
||||
content_type = content_block.get("type", "")
|
||||
# Skip text blocks that start with billing header
|
||||
if content_type == "text" and text.startswith("x-anthropic-billing-header:"):
|
||||
if content_type == "text" and text.startswith(
|
||||
"x-anthropic-billing-header:"
|
||||
):
|
||||
continue
|
||||
filtered_list.append(content_block)
|
||||
else:
|
||||
|
|
@ -111,11 +113,13 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
import os
|
||||
|
||||
# Check for Anthropic OAuth token in Authorization header
|
||||
headers, api_key = optionally_handle_anthropic_oauth(headers=headers, api_key=api_key)
|
||||
headers, api_key = optionally_handle_anthropic_oauth(
|
||||
headers=headers, api_key=api_key
|
||||
)
|
||||
if api_key is None:
|
||||
api_key = os.getenv("ANTHROPIC_API_KEY")
|
||||
|
||||
if "x-api-key" not in headers and api_key:
|
||||
if "x-api-key" not in headers and "authorization" not in headers and api_key:
|
||||
headers["x-api-key"] = api_key
|
||||
if "anthropic-version" not in headers:
|
||||
headers["anthropic-version"] = DEFAULT_ANTHROPIC_API_VERSION
|
||||
|
|
@ -149,7 +153,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
message="max_tokens is required for Anthropic /v1/messages API",
|
||||
status_code=400,
|
||||
)
|
||||
|
||||
|
||||
# Filter out x-anthropic-billing-header from system messages
|
||||
system_param = anthropic_messages_optional_request_params.get("system")
|
||||
if system_param is not None:
|
||||
|
|
@ -159,7 +163,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
else:
|
||||
# Remove system parameter if all content was filtered out
|
||||
anthropic_messages_optional_request_params.pop("system", None)
|
||||
|
||||
|
||||
####### get required params for all anthropic messages requests ######
|
||||
verbose_logger.debug(f"TRANSFORMATION DEBUG - Messages: {messages}")
|
||||
anthropic_messages_request: AnthropicMessagesRequest = AnthropicMessagesRequest(
|
||||
|
|
@ -244,25 +248,29 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
edits = context_management_param.get("edits", [])
|
||||
has_compact = False
|
||||
has_other = False
|
||||
|
||||
|
||||
for edit in edits:
|
||||
edit_type = edit.get("type", "")
|
||||
if edit_type == "compact_20260112":
|
||||
has_compact = True
|
||||
else:
|
||||
has_other = True
|
||||
|
||||
|
||||
# Add compact header if any compact edits exist
|
||||
if has_compact:
|
||||
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.COMPACT_2026_01_12.value)
|
||||
|
||||
|
||||
# Add context management header if any other edits exist
|
||||
if has_other:
|
||||
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value)
|
||||
beta_values.add(
|
||||
ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value
|
||||
)
|
||||
|
||||
# Check for structured outputs
|
||||
if optional_params.get("output_format") is not None:
|
||||
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value)
|
||||
beta_values.add(
|
||||
ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value
|
||||
)
|
||||
|
||||
# Check for fast mode
|
||||
if optional_params.get("speed") == "fast":
|
||||
|
|
|
|||
|
|
@ -1206,7 +1206,28 @@ def get_async_httpx_client(
|
|||
If not present, creates a new client
|
||||
|
||||
Caches the new client and returns it.
|
||||
|
||||
Note: When shared_session is provided, the cache is bypassed to ensure
|
||||
the user's session (with its trace_configs, connector settings, etc.)
|
||||
is used for the request.
|
||||
"""
|
||||
# When shared_session is provided, bypass cache and create a new handler
|
||||
# that uses the user's session directly. This preserves the user's
|
||||
# session configuration including trace_configs for aiohttp tracing.
|
||||
if shared_session is not None:
|
||||
verbose_logger.debug(
|
||||
f"shared_session provided (ID: {id(shared_session)}), bypassing client cache"
|
||||
)
|
||||
if params is not None:
|
||||
handler_params = {k: v for k, v in params.items() if k != "disable_aiohttp_transport"}
|
||||
handler_params["shared_session"] = shared_session
|
||||
return AsyncHTTPHandler(**handler_params)
|
||||
else:
|
||||
return AsyncHTTPHandler(
|
||||
timeout=httpx.Timeout(timeout=600.0, connect=5.0),
|
||||
shared_session=shared_session,
|
||||
)
|
||||
|
||||
_params_key_name = ""
|
||||
if params is not None:
|
||||
for key, value in params.items():
|
||||
|
|
@ -1233,12 +1254,10 @@ def get_async_httpx_client(
|
|||
if params is not None:
|
||||
# Filter out params that are only used for cache key, not for AsyncHTTPHandler.__init__
|
||||
handler_params = {k: v for k, v in params.items() if k != "disable_aiohttp_transport"}
|
||||
handler_params["shared_session"] = shared_session
|
||||
_new_client = AsyncHTTPHandler(**handler_params)
|
||||
else:
|
||||
_new_client = AsyncHTTPHandler(
|
||||
timeout=httpx.Timeout(timeout=600.0, connect=5.0),
|
||||
shared_session=shared_session,
|
||||
)
|
||||
|
||||
cache.set_cache(
|
||||
|
|
|
|||
|
|
@ -772,12 +772,15 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig):
|
|||
class OpenAIChatCompletionStreamingHandler(BaseModelResponseIterator):
|
||||
def chunk_parser(self, chunk: dict) -> ModelResponseStream:
|
||||
try:
|
||||
return ModelResponseStream(
|
||||
id=chunk["id"],
|
||||
object="chat.completion.chunk",
|
||||
created=chunk.get("created"),
|
||||
model=chunk.get("model"),
|
||||
choices=chunk.get("choices", []),
|
||||
)
|
||||
kwargs = {
|
||||
"id": chunk["id"],
|
||||
"object": "chat.completion.chunk",
|
||||
"created": chunk.get("created"),
|
||||
"model": chunk.get("model"),
|
||||
"choices": chunk.get("choices", []),
|
||||
}
|
||||
if "usage" in chunk and chunk["usage"] is not None:
|
||||
kwargs["usage"] = chunk["usage"]
|
||||
return ModelResponseStream(**kwargs)
|
||||
except Exception as e:
|
||||
raise e
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@ from typing import TYPE_CHECKING, Any, Dict, Optional, Union, cast, get_type_hin
|
|||
|
||||
import httpx
|
||||
from openai.types.responses import ResponseReasoningItem
|
||||
from pydantic import BaseModel
|
||||
from pydantic import BaseModel, ValidationError
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
|
|
@ -249,7 +249,17 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
parsed_chunk["error"]["code"] = "unknown_error"
|
||||
except Exception:
|
||||
verbose_logger.debug("Failed to coalesce error.code in parsed_chunk")
|
||||
return event_pydantic_model(**parsed_chunk)
|
||||
|
||||
try:
|
||||
return event_pydantic_model(**parsed_chunk)
|
||||
except ValidationError:
|
||||
verbose_logger.debug(
|
||||
"Pydantic validation failed for %s with chunk %s, "
|
||||
"falling back to model_construct",
|
||||
event_pydantic_model.__name__,
|
||||
parsed_chunk,
|
||||
)
|
||||
return event_pydantic_model.model_construct(**parsed_chunk)
|
||||
|
||||
@staticmethod
|
||||
def get_event_model_class(event_type: str) -> Any:
|
||||
|
|
|
|||
|
|
@ -529,6 +529,17 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
|
|||
raise e
|
||||
|
||||
|
||||
def _pop_and_merge_extra_body(data: RequestBody, optional_params: dict) -> None:
|
||||
"""Pop extra_body from optional_params and shallow-merge into data, deep-merging dict values."""
|
||||
extra_body: Optional[dict] = optional_params.pop("extra_body", None)
|
||||
if extra_body is not None:
|
||||
for k, v in extra_body.items():
|
||||
if k in data and isinstance(data[k], dict) and isinstance(v, dict):
|
||||
data[k].update(v)
|
||||
else:
|
||||
data[k] = v
|
||||
|
||||
|
||||
def _transform_request_body(
|
||||
messages: List[AllMessageValues],
|
||||
model: str,
|
||||
|
|
@ -619,6 +630,7 @@ def _transform_request_body(
|
|||
# Only add labels for Vertex AI endpoints (not Google GenAI/AI Studio) and only if non-empty
|
||||
if labels and custom_llm_provider != LlmProviders.GEMINI:
|
||||
data["labels"] = labels
|
||||
_pop_and_merge_extra_body(data, optional_params)
|
||||
except Exception as e:
|
||||
raise e
|
||||
|
||||
|
|
|
|||
|
|
@ -480,7 +480,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
tool = {VertexToolName.COMPUTER_USE.value: computer_use_config}
|
||||
# Handle OpenAI-style web_search and web_search_preview tools
|
||||
# Transform them to Gemini's googleSearch tool
|
||||
elif "type" in tool and tool["type"] in ("web_search", "web_search_preview"):
|
||||
elif "type" in tool and tool["type"] in (
|
||||
"web_search",
|
||||
"web_search_preview",
|
||||
):
|
||||
verbose_logger.info(
|
||||
f"Gemini: Transforming OpenAI-style '{tool['type']}' tool to googleSearch"
|
||||
)
|
||||
|
|
@ -1196,6 +1199,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
"PROHIBITED_CONTENT": "The token generation was stopped as the response was flagged for the prohibited contents.",
|
||||
"SPII": "The token generation was stopped as the response was flagged for Sensitive Personally Identifiable Information (SPII) contents.",
|
||||
"IMAGE_SAFETY": "The token generation was stopped as the response was flagged for image safety reasons.",
|
||||
"IMAGE_PROHIBITED_CONTENT": "The token generation was stopped as the response was flagged for prohibited image content.",
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -1218,6 +1222,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
"SPII": "content_filter",
|
||||
"MALFORMED_FUNCTION_CALL": "malformed_function_call", # openai doesn't have a way of representing this
|
||||
"IMAGE_SAFETY": "content_filter",
|
||||
"IMAGE_PROHIBITED_CONTENT": "content_filter",
|
||||
}
|
||||
|
||||
def translate_exception_str(self, exception_string: str):
|
||||
|
|
@ -1630,7 +1635,9 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
completion_image_tokens = response_tokens_details.image_tokens or 0
|
||||
completion_audio_tokens = response_tokens_details.audio_tokens or 0
|
||||
calculated_text_tokens = (
|
||||
candidates_token_count - completion_image_tokens - completion_audio_tokens
|
||||
candidates_token_count
|
||||
- completion_image_tokens
|
||||
- completion_audio_tokens
|
||||
)
|
||||
response_tokens_details.text_tokens = calculated_text_tokens
|
||||
#########################################################
|
||||
|
|
@ -2248,6 +2255,13 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
citation_metadata # older approach - maintaining to prevent regressions
|
||||
)
|
||||
|
||||
## ADD TRAFFIC TYPE ##
|
||||
traffic_type = completion_response.get("usageMetadata", {}).get(
|
||||
"trafficType"
|
||||
)
|
||||
if traffic_type:
|
||||
model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
|
||||
|
||||
except Exception as e:
|
||||
raise VertexAIError(
|
||||
message="Received={}, Error converting to valid response block={}. File an issue if litellm error - https://github.com/BerriAI/litellm/issues".format(
|
||||
|
|
@ -2906,6 +2920,12 @@ class ModelResponseIterator:
|
|||
PromptTokensDetailsWrapper, usage.prompt_tokens_details
|
||||
).web_search_requests = web_search_requests
|
||||
|
||||
traffic_type = processed_chunk.get("usageMetadata", {}).get(
|
||||
"trafficType"
|
||||
)
|
||||
if traffic_type:
|
||||
model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
|
||||
|
||||
setattr(model_response, "usage", usage) # type: ignore
|
||||
|
||||
model_response._hidden_params["is_finished"] = False
|
||||
|
|
|
|||
|
|
@ -7383,6 +7383,16 @@ def stream_chunk_builder( # noqa: PLR0915
|
|||
|
||||
setattr(response, "usage", usage)
|
||||
|
||||
# Propagate provider_specific_fields from the last chunk (contains provider
|
||||
# metadata like traffic_type set during streaming)
|
||||
for chunk in reversed(chunks):
|
||||
hidden = getattr(chunk, "_hidden_params", None)
|
||||
if hidden and "provider_specific_fields" in hidden:
|
||||
response._hidden_params.setdefault(
|
||||
"provider_specific_fields", {}
|
||||
).update(hidden["provider_specific_fields"])
|
||||
break
|
||||
|
||||
# Add cost to usage object if include_cost_in_streaming_usage is True
|
||||
if litellm.include_cost_in_streaming_usage and logging_obj is not None:
|
||||
setattr(
|
||||
|
|
|
|||
|
|
@ -6105,6 +6105,32 @@
|
|||
"output_cost_per_token": 2.4e-05,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6116,6 +6142,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6128,7 +6181,7 @@
|
|||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
|
|
@ -6159,6 +6212,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 7.2e-07
|
||||
},
|
||||
"bedrock/ap-south-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-south-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-south-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.1e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6170,6 +6249,86 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/ap-south-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-south-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0": {
|
||||
"input_cost_per_token": 3.05e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6188,6 +6347,46 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 6.9e-07
|
||||
},
|
||||
"bedrock/eu-north-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-north-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-north-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"input_cost_per_second": 0.01635,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6275,6 +6474,32 @@
|
|||
"output_cost_per_token": 2.4e-05,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-central-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0": {
|
||||
"input_cost_per_token": 2.86e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6293,6 +6518,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 6.5e-07
|
||||
},
|
||||
"bedrock/eu-west-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0": {
|
||||
"input_cost_per_token": 3.45e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6311,6 +6562,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 7.8e-07
|
||||
},
|
||||
"bedrock/eu-west-2/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 4.7e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.86e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-2/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 7.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.86e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2": {
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6341,6 +6618,32 @@
|
|||
"output_cost_per_token": 9.1e-07,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-south-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-south-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0": {
|
||||
"input_cost_per_token": 3e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6375,6 +6678,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.01e-06
|
||||
},
|
||||
"bedrock/sa-east-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/sa-east-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/sa-east-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6386,6 +6715,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/sa-east-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/sa-east-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6522,6 +6878,32 @@
|
|||
"output_cost_per_token": 7e-07,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6533,6 +6915,59 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/us-east-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6544,6 +6979,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/us-east-2/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-gov-east-1/amazon.nova-pro-v1:0": {
|
||||
"input_cost_per_token": 9.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6950,6 +7412,32 @@
|
|||
"output_cost_per_token": 7e-07,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-west-2/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-west-2/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6961,6 +7449,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/us-west-2/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-west-2/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0": {
|
||||
"cache_creation_input_token_cost": 1e-06,
|
||||
"cache_read_input_token_cost": 8e-08,
|
||||
|
|
@ -10874,6 +11389,19 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"dolphin": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "nlp_cloud",
|
||||
|
|
@ -21344,6 +21872,19 @@
|
|||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"minimax/speech-02-hd": {
|
||||
"input_cost_per_character": 0.0001,
|
||||
"litellm_provider": "minimax",
|
||||
|
|
@ -22100,6 +22641,20 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"moonshot/kimi-k2-0711-preview": {
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
|
|
@ -24381,11 +24936,8 @@
|
|||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "responses",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"supported_endpoints": [
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
|
|
@ -25537,6 +26089,19 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"recraft/recraftv2": {
|
||||
"litellm_provider": "recraft",
|
||||
"mode": "image_generation",
|
||||
|
|
@ -27973,6 +28538,30 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"us.deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"eu.deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"us.meta.llama3-1-405b-instruct-v1:0": {
|
||||
"input_cost_per_token": 5.32e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -32083,6 +32672,23 @@
|
|||
"1280x720"
|
||||
]
|
||||
},
|
||||
"openai/sora-2-pro-high-res": {
|
||||
"litellm_provider": "openai",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_video_per_second": 0.5,
|
||||
"source": "https://platform.openai.com/docs/api-reference/videos",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"supported_resolutions": [
|
||||
"1024x1792",
|
||||
"1792x1024"
|
||||
]
|
||||
},
|
||||
"azure/sora-2": {
|
||||
"litellm_provider": "azure",
|
||||
"mode": "video_generation",
|
||||
|
|
@ -35722,7 +36328,9 @@
|
|||
"gpt-realtime-mini-2025-10-06": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
"input_cost_per_image": 8e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -35753,7 +36361,9 @@
|
|||
"gpt-realtime-mini-2025-12-15": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
"input_cost_per_image": 8e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -35815,6 +36425,23 @@
|
|||
"1280x720"
|
||||
]
|
||||
},
|
||||
"sora-2-pro-high-res": {
|
||||
"litellm_provider": "openai",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_video_per_second": 0.5,
|
||||
"source": "https://platform.openai.com/docs/api-reference/videos",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"supported_resolutions": [
|
||||
"1024x1792",
|
||||
"1792x1024"
|
||||
]
|
||||
},
|
||||
"chatgpt-image-latest": {
|
||||
"cache_read_input_image_token_cost": 2.5e-06,
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
|
|
|
|||
|
|
@ -71,9 +71,7 @@ try:
|
|||
from mcp.shared.tool_name_validation import (
|
||||
validate_tool_name, # pyright: ignore[reportAssignmentType]
|
||||
)
|
||||
from mcp.shared.tool_name_validation import (
|
||||
SEP_986_URL,
|
||||
)
|
||||
from mcp.shared.tool_name_validation import SEP_986_URL
|
||||
except ImportError:
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
|
|
|||
|
|
@ -925,10 +925,11 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
|
|||
if isinstance(
|
||||
api_key, str
|
||||
): # if generated token, make sure it starts with sk-.
|
||||
_masked_key = "{}****{}".format(api_key[:4], api_key[-4:]) if len(api_key) > 8 else "****"
|
||||
assert api_key.startswith(
|
||||
"sk-"
|
||||
), "LiteLLM Virtual Key expected. Received={}, expected to start with 'sk-'.".format(
|
||||
api_key
|
||||
_masked_key
|
||||
) # prevent token hashes from being used
|
||||
else:
|
||||
verbose_logger.warning(
|
||||
|
|
|
|||
|
|
@ -5,14 +5,9 @@ OpenAI Moderation Guardrail Integration for LiteLLM
|
|||
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
AsyncGenerator,
|
||||
Dict,
|
||||
List,
|
||||
Literal,
|
||||
Optional,
|
||||
Type,
|
||||
Union,
|
||||
)
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
|
@ -20,7 +15,7 @@ from fastapi import HTTPException
|
|||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.integrations.custom_guardrail import (
|
||||
CustomGuardrail,
|
||||
log_guardrail_information,
|
||||
log_guardrail_information
|
||||
)
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
|
|
@ -32,10 +27,8 @@ from litellm.types.utils import GenericGuardrailAPIInputs
|
|||
from .base import OpenAIGuardrailBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.types.llms.openai import OpenAIModerationResponse
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
|
||||
from litellm.types.utils import ModelResponse, ModelResponseStream
|
||||
|
||||
|
||||
class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
||||
|
|
@ -236,108 +229,6 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
# Moderation doesn't modify content, just blocks - return inputs unchanged
|
||||
return inputs
|
||||
|
||||
@log_guardrail_information
|
||||
async def async_post_call_streaming_iterator_hook(
|
||||
self,
|
||||
user_api_key_dict: "UserAPIKeyAuth",
|
||||
response: Any,
|
||||
request_data: Dict[str, Any],
|
||||
) -> AsyncGenerator["ModelResponseStream", None]:
|
||||
"""
|
||||
Process streaming response chunks for OpenAI moderation.
|
||||
|
||||
Collects all chunks from the stream, assembles them into a complete response,
|
||||
and applies moderation check. If content violates moderation policy, raises HTTPException.
|
||||
"""
|
||||
# Import here to avoid circular imports
|
||||
from litellm.llms.base_llm.base_model_iterator import MockResponseIterator
|
||||
from litellm.main import stream_chunk_builder
|
||||
from litellm.types.utils import TextCompletionResponse
|
||||
|
||||
verbose_proxy_logger.debug("OpenAI Moderation: Running streaming response scan")
|
||||
|
||||
# Collect all chunks to process them together
|
||||
all_chunks: List["ModelResponseStream"] = []
|
||||
async for chunk in response:
|
||||
all_chunks.append(chunk)
|
||||
|
||||
# Assemble the complete response from chunks
|
||||
assembled_model_response: Optional[
|
||||
Union["ModelResponse", TextCompletionResponse]
|
||||
] = stream_chunk_builder(
|
||||
chunks=all_chunks,
|
||||
)
|
||||
|
||||
if isinstance(assembled_model_response, (type(None), TextCompletionResponse)):
|
||||
# If we can't assemble a ModelResponse or it's a text completion,
|
||||
# just yield the original chunks without moderation
|
||||
verbose_proxy_logger.warning(
|
||||
"OpenAI Moderation: Could not assemble ModelResponse from chunks, skipping moderation"
|
||||
)
|
||||
for chunk in all_chunks:
|
||||
yield chunk
|
||||
return
|
||||
|
||||
# Extract response text for moderation
|
||||
response_text = self._extract_response_text(assembled_model_response)
|
||||
if response_text:
|
||||
verbose_proxy_logger.debug(
|
||||
f"OpenAI Moderation: Streaming response text: {response_text[:100]}..." # Log first 100 chars
|
||||
)
|
||||
|
||||
# Make moderation request - this will raise HTTPException if content is flagged
|
||||
moderation_response = await self.async_make_request(
|
||||
input_text=response_text,
|
||||
)
|
||||
|
||||
# Check if content is flagged and raise exception if needed
|
||||
self._check_moderation_result(moderation_response)
|
||||
|
||||
# If we reach here, content passed moderation - yield the original chunks
|
||||
mock_response = MockResponseIterator(model_response=assembled_model_response)
|
||||
|
||||
# Return the reconstructed stream
|
||||
async for chunk in mock_response:
|
||||
yield chunk
|
||||
|
||||
def _extract_response_text(self, response: "ModelResponse") -> Optional[str]:
|
||||
"""
|
||||
Extract text content from the model response for moderation.
|
||||
"""
|
||||
if not hasattr(response, "choices") or not response.choices:
|
||||
return None
|
||||
|
||||
response_texts = []
|
||||
for choice in response.choices:
|
||||
try:
|
||||
# Try to get content from message (chat completion)
|
||||
message = getattr(choice, "message", None)
|
||||
if message:
|
||||
content = getattr(message, "content", None)
|
||||
if content and isinstance(content, str):
|
||||
response_texts.append(content)
|
||||
continue
|
||||
|
||||
# Try to get text (text completion)
|
||||
text = getattr(choice, "text", None)
|
||||
if text and isinstance(text, str):
|
||||
response_texts.append(text)
|
||||
continue
|
||||
|
||||
# Try to get content from delta (streaming)
|
||||
delta = getattr(choice, "delta", None)
|
||||
if delta:
|
||||
content = getattr(delta, "content", None)
|
||||
if content and isinstance(content, str):
|
||||
response_texts.append(content)
|
||||
continue
|
||||
|
||||
except (AttributeError, TypeError):
|
||||
# Skip choices that don't have expected attributes
|
||||
continue
|
||||
|
||||
return "\n".join(response_texts) if response_texts else None
|
||||
|
||||
@staticmethod
|
||||
def get_config_model() -> Optional[Type["GuardrailConfigModel"]]:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -386,7 +386,11 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
|
|||
continue
|
||||
return final_results
|
||||
except Exception as e:
|
||||
raise e
|
||||
# Sanitize exception to avoid leaking the original text (which may
|
||||
# contain API keys or other secrets) in error responses.
|
||||
raise Exception(
|
||||
f"Presidio PII analysis failed: {type(e).__name__}"
|
||||
) from e
|
||||
|
||||
async def anonymize_text(
|
||||
self,
|
||||
|
|
@ -443,9 +447,15 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
|
|||
)
|
||||
return redacted_text["text"]
|
||||
else:
|
||||
raise Exception(f"Invalid anonymizer response: {redacted_text}")
|
||||
raise Exception("Invalid anonymizer response: received None")
|
||||
except Exception as e:
|
||||
raise e
|
||||
# Sanitize exception to avoid leaking the original text (which may
|
||||
# contain API keys or other secrets) in error responses.
|
||||
if "Invalid anonymizer response" in str(e):
|
||||
raise
|
||||
raise Exception(
|
||||
f"Presidio PII anonymization failed: {type(e).__name__}"
|
||||
) from e
|
||||
|
||||
def filter_analyze_results_by_score(
|
||||
self, analyze_results: Union[List[PresidioAnalyzeResponseItem], Dict]
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ from litellm.types.utils import GenericGuardrailAPIInputs
|
|||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
|
||||
|
||||
GUARDRAIL_TIMEOUT = 5
|
||||
|
||||
|
|
@ -334,3 +335,11 @@ class ZscalerAIGuard(CustomGuardrail):
|
|||
user_facing_error = self._create_user_facing_error(f"{str(e)})")
|
||||
# This exception will be caught by the proxy and returned to the user
|
||||
raise HTTPException(status_code=500, detail=user_facing_error)
|
||||
|
||||
@staticmethod
|
||||
def get_config_model() -> Optional[type["GuardrailConfigModel"]]:
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.zscaler_ai_guard import (
|
||||
ZscalerAIGuardConfigModel,
|
||||
)
|
||||
|
||||
return ZscalerAIGuardConfigModel
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from litellm._uuid import uuid
|
|||
from litellm.integrations.custom_guardrail import CustomGuardrail
|
||||
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
from litellm.proxy.types_utils.utils import get_instance_fn
|
||||
from litellm.secret_managers.main import get_secret
|
||||
from litellm.types.guardrails import (
|
||||
Guardrail,
|
||||
|
|
@ -489,7 +490,7 @@ class InMemoryGuardrailHandler:
|
|||
config_file_path: Optional[str] = None,
|
||||
) -> Optional[CustomGuardrail]:
|
||||
"""
|
||||
Initialize a Custom Guardrail from a python file
|
||||
Initialize a Custom Guardrail from a python file or module path
|
||||
|
||||
This initializes it by adding it to the litellm callback manager
|
||||
"""
|
||||
|
|
@ -498,26 +499,12 @@ class InMemoryGuardrailHandler:
|
|||
"GuardrailsAIException - Please pass the config_file_path to initialize_guardrails_v2"
|
||||
)
|
||||
|
||||
_file_name, _class_name = guardrail_type.split(".")
|
||||
verbose_proxy_logger.debug(
|
||||
"Initializing custom guardrail: %s, file_name: %s, class_name: %s",
|
||||
"Initializing custom guardrail: %s",
|
||||
guardrail_type,
|
||||
_file_name,
|
||||
_class_name,
|
||||
)
|
||||
|
||||
directory = os.path.dirname(config_file_path)
|
||||
module_file_path = os.path.join(directory, _file_name) + ".py"
|
||||
|
||||
spec = importlib.util.spec_from_file_location(_class_name, module_file_path) # type: ignore
|
||||
if not spec:
|
||||
raise ImportError(
|
||||
f"Could not find a module specification for {module_file_path}"
|
||||
)
|
||||
|
||||
module = importlib.util.module_from_spec(spec) # type: ignore
|
||||
spec.loader.exec_module(module) # type: ignore
|
||||
_guardrail_class = getattr(module, _class_name)
|
||||
_guardrail_class = get_instance_fn(guardrail_type, config_file_path=config_file_path)
|
||||
|
||||
mode = litellm_params.mode
|
||||
if mode is None:
|
||||
|
|
|
|||
|
|
@ -628,10 +628,11 @@ async def _common_key_generation_helper( # noqa: PLR0915
|
|||
|
||||
# Validate user-provided key format
|
||||
if data.key is not None and not data.key.startswith("sk-"):
|
||||
_masked = "{}****{}".format(data.key[:4], data.key[-4:]) if len(data.key) > 8 else "****"
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail={
|
||||
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {data.key}"
|
||||
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {_masked}"
|
||||
},
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -740,7 +740,7 @@ class ProxyLogging:
|
|||
self, dynamic_success_callbacks: Optional[List], global_callbacks: List
|
||||
) -> List:
|
||||
if dynamic_success_callbacks is None:
|
||||
return global_callbacks
|
||||
return list(global_callbacks)
|
||||
return list(set(dynamic_success_callbacks + global_callbacks))
|
||||
|
||||
def _parse_pre_mcp_call_hook_response(
|
||||
|
|
|
|||
|
|
@ -1512,6 +1512,12 @@ class LiteLLMCompletionResponsesConfig:
|
|||
user=getattr(chat_completion_response, "user", None),
|
||||
)
|
||||
responses_api_response._hidden_params = getattr(chat_completion_response, "_hidden_params", {})
|
||||
|
||||
# Surface provider-specific fields (generic passthrough from any provider)
|
||||
provider_fields = responses_api_response._hidden_params.get("provider_specific_fields")
|
||||
if provider_fields:
|
||||
responses_api_response.provider_specific_fields = provider_fields
|
||||
|
||||
return responses_api_response
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -682,6 +682,7 @@ def responses(
|
|||
_is_async=_is_async,
|
||||
stream=stream,
|
||||
extra_headers=extra_headers,
|
||||
extra_body=extra_body,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -41,6 +41,53 @@ from litellm.types.utils import GenericBudgetConfigType, StandardLoggingPayload
|
|||
DEFAULT_REDIS_SYNC_INTERVAL = 1
|
||||
|
||||
|
||||
class _LiteLLMParamsDictView:
|
||||
"""
|
||||
Lightweight attribute view over `litellm_params` dict.
|
||||
|
||||
This avoids pydantic construction in request hot-path while preserving
|
||||
attribute-style access used by `litellm.get_llm_provider(...)`.
|
||||
"""
|
||||
|
||||
__slots__ = ("_params",)
|
||||
|
||||
def __init__(self, params: Dict[str, Any]):
|
||||
self._params = params
|
||||
|
||||
def __getattr__(self, key: str) -> Any:
|
||||
return self._params.get(key)
|
||||
|
||||
def __getitem__(self, key: str) -> Any:
|
||||
return self._params.get(key)
|
||||
|
||||
def __contains__(self, key: str) -> bool:
|
||||
return key in self._params
|
||||
|
||||
def get(self, key: str, default: Any = None) -> Any:
|
||||
return self._params.get(key, default)
|
||||
|
||||
def keys(self):
|
||||
return self._params.keys()
|
||||
|
||||
def values(self):
|
||||
return self._params.values()
|
||||
|
||||
def items(self):
|
||||
return self._params.items()
|
||||
|
||||
def __iter__(self):
|
||||
return iter(self._params)
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self._params)
|
||||
|
||||
def dict(self) -> Dict[str, Any]:
|
||||
return dict(self._params)
|
||||
|
||||
def model_dump(self) -> Dict[str, Any]:
|
||||
return dict(self._params)
|
||||
|
||||
|
||||
class RouterBudgetLimiting(CustomLogger):
|
||||
def __init__(
|
||||
self,
|
||||
|
|
@ -98,6 +145,7 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
cache_keys,
|
||||
provider_configs,
|
||||
deployment_configs,
|
||||
deployment_providers,
|
||||
) = await self._async_get_cache_keys_for_router_budget_limiting(
|
||||
healthy_deployments=healthy_deployments,
|
||||
request_kwargs=request_kwargs,
|
||||
|
|
@ -123,6 +171,7 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
healthy_deployments=healthy_deployments,
|
||||
provider_configs=provider_configs,
|
||||
deployment_configs=deployment_configs,
|
||||
deployment_providers=deployment_providers,
|
||||
spend_map=spend_map,
|
||||
potential_deployments=potential_deployments,
|
||||
request_tags=_get_tags_from_request_kwargs(
|
||||
|
|
@ -145,6 +194,7 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
healthy_deployments: List[Dict[str, Any]],
|
||||
provider_configs: Dict[str, GenericBudgetInfo],
|
||||
deployment_configs: Dict[str, GenericBudgetInfo],
|
||||
deployment_providers: List[Optional[str]],
|
||||
spend_map: Dict[str, float],
|
||||
request_tags: List[str],
|
||||
) -> Tuple[List[Dict[str, Any]], str]:
|
||||
|
|
@ -161,12 +211,15 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
"""
|
||||
# Filter deployments based on both provider and deployment budgets
|
||||
deployment_above_budget_info: str = ""
|
||||
for deployment in healthy_deployments:
|
||||
for idx, deployment in enumerate(healthy_deployments):
|
||||
is_within_budget = True
|
||||
|
||||
# Check provider budget
|
||||
if self.provider_budget_config:
|
||||
provider = self._get_llm_provider_for_deployment(deployment)
|
||||
if idx < len(deployment_providers):
|
||||
provider = deployment_providers[idx]
|
||||
else:
|
||||
provider = self._get_llm_provider_for_deployment(deployment)
|
||||
if provider in provider_configs:
|
||||
config = provider_configs[provider]
|
||||
if config.max_budget is None:
|
||||
|
|
@ -230,24 +283,32 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
self,
|
||||
healthy_deployments: List[Dict[str, Any]],
|
||||
request_kwargs: Optional[Dict] = None,
|
||||
) -> Tuple[List[str], Dict[str, GenericBudgetInfo], Dict[str, GenericBudgetInfo]]:
|
||||
) -> Tuple[
|
||||
List[str],
|
||||
Dict[str, GenericBudgetInfo],
|
||||
Dict[str, GenericBudgetInfo],
|
||||
List[Optional[str]],
|
||||
]:
|
||||
"""
|
||||
Returns list of cache keys to fetch from router cache for budget limiting and provider and deployment configs
|
||||
|
||||
Returns:
|
||||
Tuple[List[str], Dict[str, GenericBudgetInfo], Dict[str, GenericBudgetInfo]]:
|
||||
Tuple[List[str], Dict[str, GenericBudgetInfo], Dict[str, GenericBudgetInfo], List[Optional[str]]]:
|
||||
- List of cache keys to fetch from router cache for budget limiting
|
||||
- Dict of provider budget configs `provider_configs`
|
||||
- Dict of deployment budget configs `deployment_configs`
|
||||
- List of resolved providers aligned by deployment index `deployment_providers`
|
||||
"""
|
||||
cache_keys: List[str] = []
|
||||
provider_configs: Dict[str, GenericBudgetInfo] = {}
|
||||
deployment_configs: Dict[str, GenericBudgetInfo] = {}
|
||||
deployment_providers: List[Optional[str]] = []
|
||||
|
||||
for deployment in healthy_deployments:
|
||||
# Check provider budgets
|
||||
if self.provider_budget_config:
|
||||
provider = self._get_llm_provider_for_deployment(deployment)
|
||||
deployment_providers.append(provider)
|
||||
if provider is not None:
|
||||
budget_config = self._get_budget_config_for_provider(provider)
|
||||
if (
|
||||
|
|
@ -280,7 +341,12 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
cache_keys.append(
|
||||
f"tag_spend:{_tag}:{_tag_budget_config.budget_duration}"
|
||||
)
|
||||
return cache_keys, provider_configs, deployment_configs
|
||||
return (
|
||||
cache_keys,
|
||||
provider_configs,
|
||||
deployment_configs,
|
||||
deployment_providers,
|
||||
)
|
||||
|
||||
async def _get_or_set_budget_start_time(
|
||||
self, start_time_key: str, current_time: float, ttl_seconds: int
|
||||
|
|
@ -597,12 +663,23 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
|
||||
def _get_llm_provider_for_deployment(self, deployment: Dict) -> Optional[str]:
|
||||
try:
|
||||
_litellm_params: LiteLLM_Params = LiteLLM_Params(
|
||||
**deployment.get("litellm_params", {"model": ""})
|
||||
)
|
||||
deployment_litellm_params = deployment.get("litellm_params") or {}
|
||||
|
||||
if isinstance(deployment_litellm_params, LiteLLM_Params):
|
||||
model = deployment_litellm_params.model or ""
|
||||
provider_resolution_params: Any = deployment_litellm_params
|
||||
elif isinstance(deployment_litellm_params, dict):
|
||||
model = deployment_litellm_params.get("model") or ""
|
||||
provider_resolution_params = _LiteLLMParamsDictView(
|
||||
deployment_litellm_params
|
||||
)
|
||||
else:
|
||||
model = ""
|
||||
provider_resolution_params = _LiteLLMParamsDictView({})
|
||||
|
||||
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
model=_litellm_params.model,
|
||||
litellm_params=_litellm_params,
|
||||
model=str(model),
|
||||
litellm_params=provider_resolution_params,
|
||||
)
|
||||
except Exception:
|
||||
verbose_router_logger.error(
|
||||
|
|
|
|||
|
|
@ -0,0 +1,132 @@
|
|||
from typing import Optional
|
||||
|
||||
from pydantic import Field, model_validator
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.types.guardrails import GuardrailParamUITypes
|
||||
|
||||
from .base import GuardrailConfigModel
|
||||
|
||||
|
||||
class ZscalerAIGuardConfigModel(GuardrailConfigModel):
|
||||
api_key: Optional[str] = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"API key for Zscaler AI Guard authentication. "
|
||||
"If not provided, falls back to ZSCALER_AI_GUARD_API_KEY environment variable."
|
||||
),
|
||||
)
|
||||
|
||||
api_base: Optional[str] = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Zscaler AI Guard API endpoint. Determines policy resolution behavior:\n"
|
||||
"• /execute-policy (default) - Requires explicit policy_id in configuration\n"
|
||||
"• /resolve-and-execute-policy - Infers policy from user-api-key-alias header\n"
|
||||
"Default: https://api.us1.zseclipse.net/v1/detection/execute-policy\n"
|
||||
"Falls back to ZSCALER_AI_GUARD_URL environment variable."
|
||||
),
|
||||
json_schema_extra={
|
||||
"examples": [
|
||||
"https://api.us1.zseclipse.net/v1/detection/execute-policy",
|
||||
"https://api.us1.zseclipse.net/v1/detection/resolve-and-execute-policy",
|
||||
]
|
||||
},
|
||||
)
|
||||
|
||||
policy_id: Optional[int] = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Global policy ID for Zscaler AI Guard. Required when using /execute-policy endpoint.\n\n"
|
||||
"Set to 0 or leave empty when using /resolve-and-execute-policy with dynamic policy resolution.\n"
|
||||
"Falls back to ZSCALER_AI_GUARD_POLICY_ID environment variable."
|
||||
),
|
||||
json_schema_extra={
|
||||
"ui_hint": "conditional_required",
|
||||
"condition": "Required when api_base ends with /execute-policy",
|
||||
},
|
||||
)
|
||||
|
||||
send_user_api_key_alias: Optional[bool] = Field(
|
||||
default=False,
|
||||
description=(
|
||||
"Send user API key alias in request headers as 'user-api-key-alias'. "
|
||||
"CRITICAL when using /resolve-and-execute-policy endpoint - the policy is inferred from this value. "
|
||||
"Also useful for tracking/auditing with /execute-policy endpoint."
|
||||
),
|
||||
json_schema_extra={
|
||||
"ui_type": GuardrailParamUITypes.BOOL,
|
||||
"ui_hint": "recommended_when",
|
||||
"condition": "Recommended when api_base ends with /resolve-and-execute-policy",
|
||||
},
|
||||
)
|
||||
|
||||
send_user_api_key_user_id: Optional[bool] = Field(
|
||||
default=False,
|
||||
description=(
|
||||
"Send user API key user_id in request headers as 'user-api-key-user-id'. "
|
||||
"Enables user-level tracking and analytics in Zscaler AI Guard."
|
||||
),
|
||||
json_schema_extra={"ui_type": GuardrailParamUITypes.BOOL},
|
||||
)
|
||||
|
||||
send_user_api_key_team_id: Optional[bool] = Field(
|
||||
default=False,
|
||||
description=(
|
||||
"Send user API key team_id in request headers as 'user-api-key-team-id'. "
|
||||
"Enables team-level tracking and analytics in Zscaler AI Guard."
|
||||
),
|
||||
json_schema_extra={"ui_type": GuardrailParamUITypes.BOOL},
|
||||
)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_endpoint_configuration(self) -> "ZscalerAIGuardConfigModel":
|
||||
"""
|
||||
Validate configuration consistency between api_base and other fields.
|
||||
Provides warnings but doesn't block (since env vars might provide values).
|
||||
"""
|
||||
import os
|
||||
|
||||
# Resolve actual api_base value (including env fallback)
|
||||
api_base = self.api_base or os.getenv(
|
||||
"ZSCALER_AI_GUARD_URL",
|
||||
"https://api.us1.zseclipse.net/v1/detection/execute-policy",
|
||||
)
|
||||
|
||||
# Resolve actual policy_id value
|
||||
policy_id = self.policy_id
|
||||
if policy_id is None:
|
||||
env_policy = os.getenv("ZSCALER_AI_GUARD_POLICY_ID")
|
||||
if env_policy:
|
||||
try:
|
||||
policy_id = int(env_policy)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.warning(
|
||||
f"ZSCALER_AI_GUARD_POLICY_ID env var is not a valid integer: {env_policy}"
|
||||
)
|
||||
|
||||
# Check for configuration issues
|
||||
is_resolve_policy = api_base.endswith("/resolve-and-execute-policy")
|
||||
is_execute_policy = api_base.endswith("/execute-policy") and not is_resolve_policy
|
||||
|
||||
# Scenario A: execute-policy without policy_id
|
||||
if is_execute_policy and (policy_id is None or policy_id < 1):
|
||||
verbose_proxy_logger.warning(
|
||||
"Using /execute-policy endpoint without a valid policy_id. "
|
||||
"Ensure ZSCALER_AI_GUARD_POLICY_ID environment variable is set, "
|
||||
"or provide policy_id via request/key/team metadata."
|
||||
)
|
||||
|
||||
# Scenario B: resolve-and-execute-policy without user_api_key_alias
|
||||
if is_resolve_policy and not self.send_user_api_key_alias:
|
||||
verbose_proxy_logger.warning(
|
||||
"Using /resolve-and-execute-policy endpoint without send_user_api_key_alias=true. "
|
||||
"The endpoint requires user-api-key-alias header to resolve the policy. "
|
||||
"Set send_user_api_key_alias to true or ensure the header is sent via other means."
|
||||
)
|
||||
|
||||
return self
|
||||
|
||||
@staticmethod
|
||||
def ui_friendly_name() -> str:
|
||||
return "Zscaler AI Guard"
|
||||
|
|
@ -6105,6 +6105,32 @@
|
|||
"output_cost_per_token": 2.4e-05,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6116,6 +6142,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6128,7 +6181,7 @@
|
|||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
|
|
@ -6159,6 +6212,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 7.2e-07
|
||||
},
|
||||
"bedrock/ap-south-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-south-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-south-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.1e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6170,6 +6249,86 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/ap-south-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-south-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-southeast-3/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0": {
|
||||
"input_cost_per_token": 3.05e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6188,6 +6347,46 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 6.9e-07
|
||||
},
|
||||
"bedrock/eu-north-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-north-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-north-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"input_cost_per_second": 0.01635,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6275,6 +6474,32 @@
|
|||
"output_cost_per_token": 2.4e-05,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-central-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0": {
|
||||
"input_cost_per_token": 2.86e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6293,6 +6518,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 6.5e-07
|
||||
},
|
||||
"bedrock/eu-west-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0": {
|
||||
"input_cost_per_token": 3.45e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6311,6 +6562,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 7.8e-07
|
||||
},
|
||||
"bedrock/eu-west-2/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 4.7e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.86e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-2/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 7.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.86e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2": {
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6341,6 +6618,32 @@
|
|||
"output_cost_per_token": 9.1e-07,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-south-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-south-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0": {
|
||||
"input_cost_per_token": 3e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6375,6 +6678,32 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.01e-06
|
||||
},
|
||||
"bedrock/sa-east-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/sa-east-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/sa-east-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 7.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6386,6 +6715,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/sa-east-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/sa-east-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.44e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6522,6 +6878,32 @@
|
|||
"output_cost_per_token": 7e-07,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6533,6 +6915,59 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/us-east-1/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6544,6 +6979,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/us-east-2/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-2/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-gov-east-1/amazon.nova-pro-v1:0": {
|
||||
"input_cost_per_token": 9.6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6950,6 +7412,32 @@
|
|||
"output_cost_per_token": 7e-07,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-west-2/minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-west-2/moonshotai.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -6961,6 +7449,33 @@
|
|||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"bedrock/us-west-2/moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-west-2/qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0": {
|
||||
"cache_creation_input_token_cost": 1e-06,
|
||||
"cache_read_input_token_cost": 8e-08,
|
||||
|
|
@ -10874,6 +11389,19 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"dolphin": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "nlp_cloud",
|
||||
|
|
@ -21344,6 +21872,19 @@
|
|||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"minimax.minimax-m2.1": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 196000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"minimax/speech-02-hd": {
|
||||
"input_cost_per_character": 0.0001,
|
||||
"litellm_provider": "minimax",
|
||||
|
|
@ -22100,6 +22641,20 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"moonshotai.kimi-k2.5": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_tokens": 262144,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"moonshot/kimi-k2-0711-preview": {
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
|
|
@ -24381,11 +24936,8 @@
|
|||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "responses",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"supported_endpoints": [
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
|
|
@ -25537,6 +26089,19 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"qwen.qwen3-coder-next": {
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"recraft/recraftv2": {
|
||||
"litellm_provider": "recraft",
|
||||
"mode": "image_generation",
|
||||
|
|
@ -27973,6 +28538,30 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"us.deepseek.v3.2": {
|
||||
"input_cost_per_token": 6.2e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.85e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"eu.deepseek.v3.2": {
|
||||
"input_cost_per_token": 7.4e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.22e-06,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"us.meta.llama3-1-405b-instruct-v1:0": {
|
||||
"input_cost_per_token": 5.32e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
|
|||
|
|
@ -182,6 +182,76 @@ class TestEnterpriseRouteChecks:
|
|||
EnterpriseRouteChecks.should_call_route("/config/update")
|
||||
|
||||
|
||||
@patch("litellm.proxy.proxy_server.premium_user", True)
|
||||
class TestEnterpriseRouteChecksModelListExemption:
|
||||
"""Test that /models and /v1/models are exempt from DISABLE_LLM_API_ENDPOINTS"""
|
||||
|
||||
@patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled")
|
||||
@patch.object(EnterpriseRouteChecks, "is_management_routes_disabled")
|
||||
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_llm_api_route")
|
||||
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_management_route")
|
||||
def test_models_route_allowed_when_llm_api_disabled(
|
||||
self,
|
||||
mock_is_management_route,
|
||||
mock_is_llm_api_route,
|
||||
mock_is_management_disabled,
|
||||
mock_is_llm_api_disabled,
|
||||
):
|
||||
"""Test that /models is allowed even when LLM API routes are disabled"""
|
||||
mock_is_management_route.return_value = False
|
||||
mock_is_llm_api_route.return_value = True
|
||||
mock_is_management_disabled.return_value = False
|
||||
mock_is_llm_api_disabled.return_value = True
|
||||
|
||||
# Should not raise exception for /models
|
||||
EnterpriseRouteChecks.should_call_route("/models")
|
||||
|
||||
@patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled")
|
||||
@patch.object(EnterpriseRouteChecks, "is_management_routes_disabled")
|
||||
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_llm_api_route")
|
||||
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_management_route")
|
||||
def test_v1_models_route_allowed_when_llm_api_disabled(
|
||||
self,
|
||||
mock_is_management_route,
|
||||
mock_is_llm_api_route,
|
||||
mock_is_management_disabled,
|
||||
mock_is_llm_api_disabled,
|
||||
):
|
||||
"""Test that /v1/models is allowed even when LLM API routes are disabled"""
|
||||
mock_is_management_route.return_value = False
|
||||
mock_is_llm_api_route.return_value = True
|
||||
mock_is_management_disabled.return_value = False
|
||||
mock_is_llm_api_disabled.return_value = True
|
||||
|
||||
# Should not raise exception for /v1/models
|
||||
EnterpriseRouteChecks.should_call_route("/v1/models")
|
||||
|
||||
@patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled")
|
||||
@patch.object(EnterpriseRouteChecks, "is_management_routes_disabled")
|
||||
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_llm_api_route")
|
||||
@patch("litellm.proxy.auth.route_checks.RouteChecks.is_management_route")
|
||||
def test_chat_completions_still_blocked_when_llm_api_disabled(
|
||||
self,
|
||||
mock_is_management_route,
|
||||
mock_is_llm_api_route,
|
||||
mock_is_management_disabled,
|
||||
mock_is_llm_api_disabled,
|
||||
):
|
||||
"""Test that non-exempt LLM routes like /v1/chat/completions are still blocked"""
|
||||
mock_is_management_route.return_value = False
|
||||
mock_is_llm_api_route.return_value = True
|
||||
mock_is_management_disabled.return_value = False
|
||||
mock_is_llm_api_disabled.return_value = True
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
EnterpriseRouteChecks.should_call_route("/v1/chat/completions")
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
assert "LLM API routes are disabled for this instance." in str(
|
||||
exc_info.value.detail
|
||||
)
|
||||
|
||||
|
||||
class TestEnterpriseRouteChecksErrorMessages:
|
||||
"""Test that error messages correctly identify which feature requires Enterprise license"""
|
||||
|
||||
|
|
|
|||
170
tests/litellm/test_bedrock_extended_beta_models.py
Normal file
170
tests/litellm/test_bedrock_extended_beta_models.py
Normal file
|
|
@ -0,0 +1,170 @@
|
|||
"""
|
||||
Test suite for AWS Bedrock extended beta model support
|
||||
Tests model configuration, pricing, and regional availability for:
|
||||
- DeepSeek V3.2
|
||||
- Minimax M2.1
|
||||
- Moonshot AI Kimi K2.5
|
||||
- Qwen3 Coder Next
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
# Set env var to use local model cost map instead of fetching from remote
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "true"
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm import get_model_info
|
||||
|
||||
# Model configurations: (model_name, regions, max_input, max_output)
|
||||
MODEL_CONFIGS = [
|
||||
(
|
||||
"deepseek.v3.2",
|
||||
[
|
||||
"ap-northeast-1",
|
||||
"ap-south-1",
|
||||
"ap-southeast-3",
|
||||
"eu-north-1",
|
||||
"sa-east-1",
|
||||
"us-east-1",
|
||||
"us-east-2",
|
||||
"us-west-2",
|
||||
],
|
||||
163840,
|
||||
163840,
|
||||
),
|
||||
(
|
||||
"minimax.minimax-m2.1",
|
||||
[
|
||||
"ap-northeast-1",
|
||||
"ap-south-1",
|
||||
"ap-southeast-3",
|
||||
"eu-central-1",
|
||||
"eu-north-1",
|
||||
"eu-south-1",
|
||||
"eu-west-1",
|
||||
"eu-west-2",
|
||||
"sa-east-1",
|
||||
"us-east-1",
|
||||
"us-east-2",
|
||||
"us-west-2",
|
||||
],
|
||||
196000,
|
||||
8192,
|
||||
),
|
||||
(
|
||||
"moonshotai.kimi-k2.5",
|
||||
[
|
||||
"ap-northeast-1",
|
||||
"ap-south-1",
|
||||
"ap-southeast-3",
|
||||
"eu-north-1",
|
||||
"sa-east-1",
|
||||
"us-east-1",
|
||||
"us-east-2",
|
||||
"us-west-2",
|
||||
],
|
||||
262144,
|
||||
262144,
|
||||
),
|
||||
(
|
||||
"qwen.qwen3-coder-next",
|
||||
[
|
||||
"ap-northeast-1",
|
||||
"ap-south-1",
|
||||
"ap-southeast-3",
|
||||
"eu-central-1",
|
||||
"eu-south-1",
|
||||
"eu-west-1",
|
||||
"eu-west-2",
|
||||
"sa-east-1",
|
||||
"us-east-1",
|
||||
"us-east-2",
|
||||
"us-west-2",
|
||||
],
|
||||
262144,
|
||||
8192,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
class TestBedrockNewModels:
|
||||
"""Unified test suite for all new Bedrock models"""
|
||||
|
||||
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
|
||||
def test_model_info_primary_region(
|
||||
self, model_name, regions, max_input, max_output
|
||||
):
|
||||
"""Test model configuration in primary region (us-east-1)"""
|
||||
model = f"bedrock/us-east-1/{model_name}"
|
||||
model_info = get_model_info(model)
|
||||
|
||||
assert model_info is not None, f"Model {model_name} not found"
|
||||
assert model_info["max_input_tokens"] == max_input
|
||||
assert model_info["max_output_tokens"] == max_output
|
||||
assert model_info["litellm_provider"] == "bedrock"
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["supports_function_calling"] is True
|
||||
|
||||
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
|
||||
def test_pricing_configured(self, model_name, regions, max_input, max_output):
|
||||
"""Verify pricing is set for all models"""
|
||||
model = f"bedrock/us-east-1/{model_name}"
|
||||
model_info = get_model_info(model)
|
||||
|
||||
assert (
|
||||
model_info["input_cost_per_token"] > 0
|
||||
), f"Missing input cost for {model_name}"
|
||||
assert (
|
||||
model_info["output_cost_per_token"] > 0
|
||||
), f"Missing output cost for {model_name}"
|
||||
|
||||
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
|
||||
def test_region_count(self, model_name, regions, max_input, max_output):
|
||||
"""Verify each bedrock/{region}/{model_name} resolves via get_model_info"""
|
||||
for region in regions:
|
||||
model = f"bedrock/{region}/{model_name}"
|
||||
model_info = get_model_info(model)
|
||||
assert model_info is not None, f"Model {model_name} not found in {region}"
|
||||
assert model_info["max_input_tokens"] == max_input
|
||||
assert model_info["max_output_tokens"] == max_output
|
||||
|
||||
@pytest.mark.parametrize("model_name,regions,max_input,max_output", MODEL_CONFIGS)
|
||||
def test_sample_regional_variants(self, model_name, regions, max_input, max_output):
|
||||
"""Test sample regional variants (us-east-1, eu-west-1, ap-northeast-1)"""
|
||||
for region in ["us-east-1", "ap-northeast-1"]:
|
||||
if region in regions:
|
||||
model = f"bedrock/{region}/{model_name}"
|
||||
model_info = get_model_info(model)
|
||||
assert (
|
||||
model_info is not None
|
||||
), f"Model {model_name} not found in {region}"
|
||||
assert model_info["max_input_tokens"] == max_input
|
||||
assert model_info["litellm_provider"] == "bedrock"
|
||||
|
||||
|
||||
class TestModelSpecificFeatures:
|
||||
"""Model-specific capability tests"""
|
||||
|
||||
def test_deepseek_v3_2_context_window(self):
|
||||
"""DeepSeek V3.2 has 163K context window"""
|
||||
model_info = get_model_info("bedrock/us-east-1/deepseek.v3.2")
|
||||
assert model_info["max_input_tokens"] == 163840
|
||||
|
||||
def test_minimax_m2_1_context_window(self):
|
||||
"""Minimax M2.1 has 196K input, 8K output"""
|
||||
model_info = get_model_info("bedrock/us-east-1/minimax.minimax-m2.1")
|
||||
assert model_info["max_input_tokens"] == 196000
|
||||
assert model_info["max_output_tokens"] == 8192
|
||||
|
||||
def test_moonshotai_kimi_k2_5_context_window(self):
|
||||
"""Moonshot AI Kimi K2.5 has 256K context window"""
|
||||
model_info = get_model_info("bedrock/us-east-1/moonshotai.kimi-k2.5")
|
||||
assert model_info["max_input_tokens"] == 262144
|
||||
assert model_info["max_output_tokens"] == 262144
|
||||
|
||||
def test_qwen3_coder_next_context_window(self):
|
||||
"""Qwen3 Coder Next has 256K input, 8K output"""
|
||||
model_info = get_model_info("bedrock/us-east-1/qwen.qwen3-coder-next")
|
||||
assert model_info["max_input_tokens"] == 262144
|
||||
assert model_info["max_output_tokens"] == 8192
|
||||
|
|
@ -293,3 +293,29 @@ def test_get_combined_callback_list():
|
|||
assert "lago" in _logging.get_combined_callback_list(
|
||||
dynamic_success_callbacks=["langfuse"], global_callbacks=["lago"]
|
||||
)
|
||||
|
||||
|
||||
def test_get_combined_callback_list_returns_copy_when_dynamic_is_none():
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
_logging = LiteLLMLoggingObj(
|
||||
model="claude-3-opus-20240229",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stream=False,
|
||||
call_type="completion",
|
||||
start_time=datetime.now(),
|
||||
litellm_call_id="123",
|
||||
function_id="456",
|
||||
)
|
||||
|
||||
global_callbacks = ["langfuse"]
|
||||
combined_callbacks = _logging.get_combined_callback_list(
|
||||
dynamic_success_callbacks=None, global_callbacks=global_callbacks
|
||||
)
|
||||
|
||||
assert combined_callbacks == ["langfuse"]
|
||||
assert combined_callbacks is not global_callbacks
|
||||
|
||||
combined_callbacks.append("new_callback")
|
||||
|
||||
assert global_callbacks == ["langfuse"]
|
||||
|
|
|
|||
58
tests/test_litellm/caching/test_dual_cache.py
Normal file
58
tests/test_litellm/caching/test_dual_cache.py
Normal file
|
|
@ -0,0 +1,58 @@
|
|||
import asyncio
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.caching.redis_cache import RedisCache
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dual_cache_async_batch_get_cache_coalesces_concurrent_redis_reads():
|
||||
dual_cache = DualCache(
|
||||
redis_cache=MagicMock(spec=RedisCache), default_redis_batch_cache_expiry=10
|
||||
)
|
||||
keys = ["shared_a", "shared_b"]
|
||||
start_gate = asyncio.Event()
|
||||
|
||||
async def _mock_async_batch_get_cache(key_list, parent_otel_span=None):
|
||||
await asyncio.sleep(0.05)
|
||||
return {k: None for k in key_list}
|
||||
|
||||
with patch.object(
|
||||
dual_cache.redis_cache,
|
||||
"async_batch_get_cache",
|
||||
new=AsyncMock(side_effect=_mock_async_batch_get_cache),
|
||||
) as mock_async_batch_get_cache:
|
||||
|
||||
async def worker():
|
||||
await start_gate.wait()
|
||||
return await dual_cache.async_batch_get_cache(keys=keys)
|
||||
|
||||
tasks = [asyncio.create_task(worker()) for _ in range(50)]
|
||||
start_gate.set()
|
||||
await asyncio.gather(*tasks)
|
||||
|
||||
assert mock_async_batch_get_cache.call_count == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dual_cache_async_batch_get_cache_rolls_back_redis_reservation_on_error():
|
||||
dual_cache = DualCache(
|
||||
redis_cache=MagicMock(spec=RedisCache), default_redis_batch_cache_expiry=10
|
||||
)
|
||||
keys = ["shared_a", "shared_b"]
|
||||
|
||||
with patch.object(
|
||||
dual_cache.redis_cache,
|
||||
"async_batch_get_cache",
|
||||
new=AsyncMock(side_effect=RuntimeError("redis unavailable")),
|
||||
) as mock_async_batch_get_cache:
|
||||
first_result = await dual_cache.async_batch_get_cache(keys=keys)
|
||||
second_result = await dual_cache.async_batch_get_cache(keys=keys)
|
||||
|
||||
assert first_result is None
|
||||
assert second_result is None
|
||||
assert mock_async_batch_get_cache.call_count == 2
|
||||
assert "shared_a" not in dual_cache.last_redis_batch_access_time
|
||||
assert "shared_b" not in dual_cache.last_redis_batch_access_time
|
||||
|
|
@ -1,84 +1,285 @@
|
|||
"""
|
||||
Tests for Anthropic OAuth token handling for Claude Code Max integration.
|
||||
Tests for Anthropic OAuth token handling in common_utils.
|
||||
|
||||
Verifies that OAuth tokens (sk-ant-oat*) are sent via Authorization: Bearer
|
||||
instead of x-api-key, per Anthropic's OAuth specification.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
# Add litellm to path
|
||||
sys.path.insert(0, os.path.abspath("../../../../.."))
|
||||
sys.path.insert(
|
||||
0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
|
||||
)
|
||||
|
||||
# Fake OAuth token for testing (not a real secret)
|
||||
FAKE_OAUTH_TOKEN = "sk-ant-oat01-fake-token-for-testing-123456789abcdef"
|
||||
FAKE_REGULAR_KEY = "sk-ant-api03-regular-key-for-testing-123456789"
|
||||
|
||||
|
||||
def test_oauth_detection_in_common_utils():
|
||||
"""Test 1: OAuth token detection in common_utils"""
|
||||
from litellm.llms.anthropic.common_utils import optionally_handle_anthropic_oauth
|
||||
class TestOptionallyHandleAnthropicOAuth:
|
||||
"""Tests for optionally_handle_anthropic_oauth function."""
|
||||
|
||||
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
|
||||
updated_headers, extracted_api_key = optionally_handle_anthropic_oauth(headers, None)
|
||||
def test_oauth_token_in_authorization_header(self):
|
||||
"""OAuth token in Authorization header should be detected and headers set correctly."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
optionally_handle_anthropic_oauth,
|
||||
)
|
||||
|
||||
assert extracted_api_key == FAKE_OAUTH_TOKEN
|
||||
assert updated_headers["anthropic-beta"] == "oauth-2025-04-20"
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
|
||||
updated_headers, extracted_api_key = optionally_handle_anthropic_oauth(
|
||||
headers, None
|
||||
)
|
||||
|
||||
assert extracted_api_key == FAKE_OAUTH_TOKEN
|
||||
assert updated_headers["anthropic-beta"] == "oauth-2025-04-20"
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert "x-api-key" not in updated_headers
|
||||
|
||||
def test_oauth_token_in_api_key_directly(self):
|
||||
"""OAuth token passed as api_key should set Authorization: Bearer header."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
optionally_handle_anthropic_oauth,
|
||||
)
|
||||
|
||||
headers = {}
|
||||
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
|
||||
headers, FAKE_OAUTH_TOKEN
|
||||
)
|
||||
|
||||
assert returned_api_key == FAKE_OAUTH_TOKEN
|
||||
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
|
||||
assert updated_headers["anthropic-beta"] == "oauth-2025-04-20"
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert "x-api-key" not in updated_headers
|
||||
|
||||
def test_oauth_removes_existing_x_api_key(self):
|
||||
"""When OAuth is detected, any existing x-api-key should be removed."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
optionally_handle_anthropic_oauth,
|
||||
)
|
||||
|
||||
headers = {"x-api-key": FAKE_OAUTH_TOKEN}
|
||||
updated_headers, _ = optionally_handle_anthropic_oauth(
|
||||
headers, FAKE_OAUTH_TOKEN
|
||||
)
|
||||
|
||||
assert "x-api-key" not in updated_headers
|
||||
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
|
||||
|
||||
def test_regular_api_key_unchanged(self):
|
||||
"""Regular API keys (non-OAuth) should pass through unmodified."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
optionally_handle_anthropic_oauth,
|
||||
)
|
||||
|
||||
headers = {}
|
||||
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
|
||||
headers, FAKE_REGULAR_KEY
|
||||
)
|
||||
|
||||
assert returned_api_key == FAKE_REGULAR_KEY
|
||||
assert "authorization" not in updated_headers
|
||||
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
|
||||
assert "anthropic-beta" not in updated_headers
|
||||
|
||||
def test_regular_key_in_authorization_header(self):
|
||||
"""Non-OAuth token in Authorization header should not trigger OAuth handling."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
optionally_handle_anthropic_oauth,
|
||||
)
|
||||
|
||||
headers = {"authorization": f"Bearer {FAKE_REGULAR_KEY}"}
|
||||
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
|
||||
headers, FAKE_REGULAR_KEY
|
||||
)
|
||||
|
||||
assert returned_api_key == FAKE_REGULAR_KEY
|
||||
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
|
||||
|
||||
def test_none_api_key_no_error(self):
|
||||
"""None api_key with empty headers should not raise errors."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
optionally_handle_anthropic_oauth,
|
||||
)
|
||||
|
||||
headers = {}
|
||||
updated_headers, returned_api_key = optionally_handle_anthropic_oauth(
|
||||
headers, None
|
||||
)
|
||||
|
||||
assert returned_api_key is None
|
||||
assert "authorization" not in updated_headers
|
||||
|
||||
|
||||
def test_oauth_integration_in_validate_environment():
|
||||
"""Test 2: OAuth integration in AnthropicConfig validate_environment"""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
class TestGetAnthropicHeaders:
|
||||
"""Tests for get_anthropic_headers method with OAuth support."""
|
||||
|
||||
config = AnthropicModelInfo()
|
||||
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
|
||||
def test_oauth_token_uses_authorization_bearer(self):
|
||||
"""OAuth token should produce Authorization: Bearer header, not x-api-key."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
updated_headers = config.validate_environment(
|
||||
headers=headers,
|
||||
model="claude-3-haiku-20240307",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=None,
|
||||
api_base=None,
|
||||
)
|
||||
config = AnthropicModelInfo()
|
||||
headers = config.get_anthropic_headers(
|
||||
api_key=FAKE_OAUTH_TOKEN,
|
||||
computer_tool_used=False,
|
||||
prompt_caching_set=False,
|
||||
pdf_used=False,
|
||||
is_vertex_request=False,
|
||||
)
|
||||
|
||||
assert updated_headers["x-api-key"] == FAKE_OAUTH_TOKEN
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
|
||||
assert headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert "oauth-2025-04-20" in headers.get("anthropic-beta", "")
|
||||
assert "x-api-key" not in headers
|
||||
|
||||
def test_regular_key_uses_x_api_key(self):
|
||||
"""Regular API key should produce x-api-key header, not Authorization."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
config = AnthropicModelInfo()
|
||||
headers = config.get_anthropic_headers(
|
||||
api_key=FAKE_REGULAR_KEY,
|
||||
computer_tool_used=False,
|
||||
prompt_caching_set=False,
|
||||
pdf_used=False,
|
||||
is_vertex_request=False,
|
||||
)
|
||||
|
||||
assert headers["x-api-key"] == FAKE_REGULAR_KEY
|
||||
assert "authorization" not in headers
|
||||
assert "anthropic-dangerous-direct-browser-access" not in headers
|
||||
|
||||
def test_oauth_includes_standard_headers(self):
|
||||
"""OAuth path should still include standard Anthropic headers."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
config = AnthropicModelInfo()
|
||||
headers = config.get_anthropic_headers(
|
||||
api_key=FAKE_OAUTH_TOKEN,
|
||||
computer_tool_used=False,
|
||||
prompt_caching_set=False,
|
||||
pdf_used=False,
|
||||
is_vertex_request=False,
|
||||
)
|
||||
|
||||
assert headers["anthropic-version"] == "2023-06-01"
|
||||
assert headers["accept"] == "application/json"
|
||||
assert headers["content-type"] == "application/json"
|
||||
|
||||
|
||||
def test_oauth_detection_in_messages_transformation():
|
||||
"""Test 3: OAuth detection in messages transformation"""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
|
||||
AnthropicMessagesConfig,
|
||||
)
|
||||
class TestValidateEnvironmentOAuth:
|
||||
"""Tests for validate_environment with OAuth tokens."""
|
||||
|
||||
config = AnthropicMessagesConfig()
|
||||
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
|
||||
def test_oauth_via_authorization_header(self):
|
||||
"""validate_environment should produce correct headers for OAuth tokens."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
updated_headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
model="claude-3-haiku-20240307",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=None,
|
||||
api_base=None,
|
||||
)
|
||||
config = AnthropicModelInfo()
|
||||
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
|
||||
|
||||
assert updated_headers["x-api-key"] == FAKE_OAUTH_TOKEN
|
||||
assert "oauth-2025-04-20" in updated_headers["anthropic-beta"]
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
updated_headers = config.validate_environment(
|
||||
headers=headers,
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=None,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert "oauth-2025-04-20" in updated_headers.get("anthropic-beta", "")
|
||||
assert "x-api-key" not in updated_headers
|
||||
|
||||
def test_oauth_via_api_key_param(self):
|
||||
"""validate_environment with OAuth token as api_key should use Bearer auth."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
config = AnthropicModelInfo()
|
||||
headers = {}
|
||||
|
||||
updated_headers = config.validate_environment(
|
||||
headers=headers,
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=FAKE_OAUTH_TOKEN,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert updated_headers["authorization"] == f"Bearer {FAKE_OAUTH_TOKEN}"
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert "x-api-key" not in updated_headers
|
||||
|
||||
def test_regular_key_via_api_key_param(self):
|
||||
"""validate_environment with regular API key should use x-api-key."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
config = AnthropicModelInfo()
|
||||
headers = {}
|
||||
|
||||
updated_headers = config.validate_environment(
|
||||
headers=headers,
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=FAKE_REGULAR_KEY,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert updated_headers["x-api-key"] == FAKE_REGULAR_KEY
|
||||
assert "authorization" not in updated_headers
|
||||
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
|
||||
|
||||
|
||||
def test_regular_api_keys_still_work():
|
||||
"""Test 4: Regular API keys still work (regression test)"""
|
||||
from litellm.llms.anthropic.common_utils import optionally_handle_anthropic_oauth
|
||||
class TestPassthroughOAuth:
|
||||
"""Tests for passthrough messages endpoint with OAuth tokens."""
|
||||
|
||||
regular_key = "sk-ant-api03-regular-key-123"
|
||||
headers = {"authorization": f"Bearer {regular_key}"}
|
||||
def test_passthrough_oauth_no_x_api_key(self):
|
||||
"""Passthrough endpoint should not add x-api-key for OAuth tokens."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
|
||||
AnthropicMessagesConfig,
|
||||
)
|
||||
|
||||
updated_headers, extracted_api_key = optionally_handle_anthropic_oauth(headers, regular_key)
|
||||
config = AnthropicMessagesConfig()
|
||||
headers = {"authorization": f"Bearer {FAKE_OAUTH_TOKEN}"}
|
||||
|
||||
# Regular key should be unchanged
|
||||
assert extracted_api_key == regular_key
|
||||
# OAuth headers should NOT be added
|
||||
assert "anthropic-dangerous-direct-browser-access" not in updated_headers
|
||||
updated_headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=None,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert "oauth-2025-04-20" in updated_headers.get("anthropic-beta", "")
|
||||
assert updated_headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
assert "x-api-key" not in updated_headers
|
||||
|
||||
def test_passthrough_regular_key_uses_x_api_key(self):
|
||||
"""Passthrough endpoint should still use x-api-key for regular API keys."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
|
||||
AnthropicMessagesConfig,
|
||||
)
|
||||
|
||||
config = AnthropicMessagesConfig()
|
||||
headers = {}
|
||||
|
||||
updated_headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=FAKE_REGULAR_KEY,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert updated_headers["x-api-key"] == FAKE_REGULAR_KEY
|
||||
assert "authorization" not in updated_headers
|
||||
|
|
|
|||
|
|
@ -480,6 +480,85 @@ async def test_session_reuse_integration():
|
|||
await client2.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_shared_session_bypasses_cache():
|
||||
"""
|
||||
Test that when shared_session is provided, the cache is bypassed.
|
||||
|
||||
This is critical for aiohttp tracing support - users need their custom
|
||||
ClientSession (with trace_configs) to be used, not a cached session.
|
||||
|
||||
Related: GitHub issue #20174
|
||||
"""
|
||||
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
# First, get a cached client without shared_session
|
||||
cached_client = get_async_httpx_client(
|
||||
llm_provider=LlmProviders.ANTHROPIC,
|
||||
shared_session=None
|
||||
)
|
||||
|
||||
# Now create a mock shared session
|
||||
mock_session = MockClientSession()
|
||||
|
||||
# Get a client WITH shared_session - this should NOT return the cached client
|
||||
client_with_session = get_async_httpx_client(
|
||||
llm_provider=LlmProviders.ANTHROPIC, # Same provider!
|
||||
shared_session=mock_session # type: ignore
|
||||
)
|
||||
|
||||
# The clients should be DIFFERENT - cache should be bypassed when shared_session is provided
|
||||
assert client_with_session is not cached_client, \
|
||||
"Cache should be bypassed when shared_session is provided"
|
||||
|
||||
# Verify the shared_session handler is using our mock session
|
||||
# The transport should have our mock_session as its client
|
||||
transport = client_with_session.client._transport
|
||||
if hasattr(transport, 'client'):
|
||||
assert transport.client is mock_session, \
|
||||
"Handler should use the provided shared_session"
|
||||
|
||||
# Clean up
|
||||
await cached_client.close()
|
||||
await client_with_session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_shared_session_each_call_gets_new_handler():
|
||||
"""
|
||||
Test that each call with shared_session creates a new handler.
|
||||
|
||||
This ensures user sessions (with their trace_configs, etc.) are always
|
||||
used and not affected by caching.
|
||||
"""
|
||||
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
# Create two different mock sessions
|
||||
mock_session1 = MockClientSession()
|
||||
mock_session2 = MockClientSession()
|
||||
|
||||
# Get clients with different sessions for the same provider
|
||||
client1 = get_async_httpx_client(
|
||||
llm_provider=LlmProviders.ANTHROPIC,
|
||||
shared_session=mock_session1 # type: ignore
|
||||
)
|
||||
|
||||
client2 = get_async_httpx_client(
|
||||
llm_provider=LlmProviders.ANTHROPIC, # Same provider
|
||||
shared_session=mock_session2 # type: ignore # Different session
|
||||
)
|
||||
|
||||
# Should be different clients, each using their own session
|
||||
assert client1 is not client2, \
|
||||
"Different shared_sessions should create different handlers"
|
||||
|
||||
# Clean up
|
||||
await client1.close()
|
||||
await client2.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_session_validation():
|
||||
"""Test that session validation works correctly"""
|
||||
|
|
|
|||
|
|
@ -9,7 +9,10 @@ import pytest
|
|||
|
||||
sys.path.insert(0, os.path.abspath("../../../../.."))
|
||||
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from litellm.llms.openai.chat.gpt_transformation import (
|
||||
OpenAIGPTConfig,
|
||||
OpenAIChatCompletionStreamingHandler,
|
||||
)
|
||||
|
||||
|
||||
class TestOpenAIGPTConfig:
|
||||
|
|
@ -136,6 +139,75 @@ class TestGetOptionalParamsIntegration:
|
|||
assert regular_params.get("user") == "my-end-user"
|
||||
assert responses_params.get("user") == "my-end-user"
|
||||
|
||||
|
||||
class TestOpenAIChatCompletionStreamingHandler:
|
||||
"""Tests for OpenAIChatCompletionStreamingHandler.chunk_parser()"""
|
||||
|
||||
def test_chunk_parser_preserves_usage(self):
|
||||
"""
|
||||
Test that chunk_parser preserves the usage field from streaming chunks.
|
||||
|
||||
"""
|
||||
handler = OpenAIChatCompletionStreamingHandler(
|
||||
streaming_response=None, sync_stream=True
|
||||
)
|
||||
|
||||
usage_chunk = {
|
||||
"id": "gen-123",
|
||||
"created": 1234567890,
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"object": "chat.completion.chunk",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"role": "assistant", "content": ""},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 13797,
|
||||
"completion_tokens": 350,
|
||||
"total_tokens": 14147,
|
||||
},
|
||||
}
|
||||
|
||||
result = handler.chunk_parser(usage_chunk)
|
||||
|
||||
assert result.usage is not None
|
||||
assert result.usage.prompt_tokens == 13797
|
||||
assert result.usage.completion_tokens == 350
|
||||
assert result.usage.total_tokens == 14147
|
||||
|
||||
def test_chunk_parser_without_usage(self):
|
||||
"""Test that chunk_parser works normally for chunks without usage."""
|
||||
handler = OpenAIChatCompletionStreamingHandler(
|
||||
streaming_response=None, sync_stream=True
|
||||
)
|
||||
|
||||
chunk = {
|
||||
"id": "gen-123",
|
||||
"created": 1234567890,
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"object": "chat.completion.chunk",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"role": "assistant", "content": "Hello"},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
result = handler.chunk_parser(chunk)
|
||||
|
||||
assert result.id == "gen-123"
|
||||
assert result.choices[0].delta.content == "Hello"
|
||||
assert not hasattr(result, "usage") or result.usage is None
|
||||
|
||||
|
||||
class TestPromptCacheKeyIntegration:
|
||||
"""Tests for prompt_cache_key support"""
|
||||
|
||||
def test_prompt_cache_key_in_optional_params(self):
|
||||
"""Test that 'prompt_cache_key' flows through get_optional_params for OpenAI models."""
|
||||
from litellm.utils import get_optional_params
|
||||
|
|
|
|||
|
|
@ -417,6 +417,129 @@ class TestOpenAIResponsesAPIConfig:
|
|||
assert event.error.code == "unknown_error"
|
||||
assert event.error.message == "Something went wrong"
|
||||
|
||||
def test_transform_streaming_response_missing_required_fields_response_created(
|
||||
self,
|
||||
):
|
||||
"""Test that ResponseCreatedEvent with missing required fields (created_at,
|
||||
output) does not crash but falls back to model_construct.
|
||||
|
||||
Reproduces https://github.com/BerriAI/litellm/issues/20570
|
||||
"""
|
||||
from litellm.types.llms.openai import ResponseCreatedEvent
|
||||
|
||||
# Minimal payload an OpenAI-compatible provider might send,
|
||||
# omitting `created_at` and `output` inside the response object.
|
||||
parsed_chunk = {
|
||||
"type": "response.created",
|
||||
"response": {
|
||||
"id": "resp_q7BOLpck7clq",
|
||||
"model": "gpt-oss-120b",
|
||||
"status": "in_progress",
|
||||
},
|
||||
}
|
||||
|
||||
result = self.config.transform_streaming_response(
|
||||
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
|
||||
)
|
||||
|
||||
assert isinstance(result, ResponseCreatedEvent)
|
||||
assert result.type == ResponsesAPIStreamEvents.RESPONSE_CREATED
|
||||
assert result.response["id"] == "resp_q7BOLpck7clq"
|
||||
|
||||
def test_transform_streaming_response_missing_required_fields_output_text_delta(
|
||||
self,
|
||||
):
|
||||
"""Test that OutputTextDeltaEvent with missing output_index and
|
||||
content_index falls back to model_construct without crashing.
|
||||
|
||||
Reproduces https://github.com/BerriAI/litellm/issues/20570
|
||||
"""
|
||||
from litellm.types.llms.openai import OutputTextDeltaEvent
|
||||
|
||||
# Provider omits output_index and content_index
|
||||
parsed_chunk = {
|
||||
"type": "response.output_text.delta",
|
||||
"item_id": "item_456",
|
||||
"delta": "Hello",
|
||||
}
|
||||
|
||||
result = self.config.transform_streaming_response(
|
||||
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
|
||||
)
|
||||
|
||||
assert isinstance(result, OutputTextDeltaEvent)
|
||||
assert result.type == ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA
|
||||
assert result.delta == "Hello"
|
||||
assert result.item_id == "item_456"
|
||||
|
||||
def test_transform_streaming_response_missing_required_fields_content_part_added(
|
||||
self,
|
||||
):
|
||||
"""Test that ContentPartAddedEvent with missing output_index and
|
||||
content_index falls back to model_construct without crashing.
|
||||
|
||||
Reproduces https://github.com/BerriAI/litellm/issues/20570
|
||||
"""
|
||||
from litellm.types.llms.openai import ContentPartAddedEvent
|
||||
|
||||
# Provider omits output_index and content_index
|
||||
parsed_chunk = {
|
||||
"type": "response.content_part.added",
|
||||
"item_id": "item_789",
|
||||
"part": {"type": "output_text", "text": ""},
|
||||
}
|
||||
|
||||
result = self.config.transform_streaming_response(
|
||||
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
|
||||
)
|
||||
|
||||
assert isinstance(result, ContentPartAddedEvent)
|
||||
assert result.type == ResponsesAPIStreamEvents.CONTENT_PART_ADDED
|
||||
assert result.item_id == "item_789"
|
||||
|
||||
def test_transform_streaming_response_missing_required_fields_output_item_added(
|
||||
self,
|
||||
):
|
||||
"""Test that OutputItemAddedEvent with missing output_index falls back
|
||||
to model_construct without crashing.
|
||||
|
||||
Reproduces https://github.com/BerriAI/litellm/issues/20570
|
||||
"""
|
||||
from litellm.types.llms.openai import OutputItemAddedEvent
|
||||
|
||||
# Provider omits output_index
|
||||
parsed_chunk = {
|
||||
"type": "response.output_item.added",
|
||||
"item": {"type": "message", "id": "msg_001", "role": "assistant"},
|
||||
}
|
||||
|
||||
result = self.config.transform_streaming_response(
|
||||
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
|
||||
)
|
||||
|
||||
assert isinstance(result, OutputItemAddedEvent)
|
||||
assert result.type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED
|
||||
|
||||
def test_transform_streaming_response_valid_chunk_still_works(self):
|
||||
"""Ensure that fully valid chunks still go through normal Pydantic
|
||||
validation (not model_construct) and work correctly."""
|
||||
parsed_chunk = {
|
||||
"type": "response.output_text.delta",
|
||||
"item_id": "item_123",
|
||||
"output_index": 0,
|
||||
"content_index": 0,
|
||||
"delta": "World",
|
||||
}
|
||||
|
||||
result = self.config.transform_streaming_response(
|
||||
model=self.model, parsed_chunk=parsed_chunk, logging_obj=self.logging_obj
|
||||
)
|
||||
|
||||
assert isinstance(result, OutputTextDeltaEvent)
|
||||
assert result.delta == "World"
|
||||
assert result.output_index == 0
|
||||
assert result.content_index == 0
|
||||
|
||||
|
||||
class TestAzureResponsesAPIConfig:
|
||||
def setup_method(self):
|
||||
|
|
|
|||
|
|
@ -3338,3 +3338,94 @@ def test_chunk_parser_handles_prompt_feedback_block_with_usage():
|
|||
assert result.usage.completion_tokens == 0, f"completion_tokens should be 0, got {result.usage.completion_tokens}"
|
||||
assert result.usage.total_tokens == 8175, f"total_tokens should be 8175, got {result.usage.total_tokens}"
|
||||
|
||||
|
||||
def test_vertex_ai_traffic_type_preserved_in_hidden_params_streaming():
|
||||
"""Test trafficType is preserved in _hidden_params for streaming."""
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
|
||||
ModelResponseIterator,
|
||||
)
|
||||
|
||||
chunk = {
|
||||
"candidates": [{"content": {"parts": [{"text": "Hello"}]}}],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 200,
|
||||
"totalTokenCount": 300,
|
||||
"trafficType": "ON_DEMAND",
|
||||
},
|
||||
}
|
||||
|
||||
iterator = ModelResponseIterator(
|
||||
streaming_response=[], sync_stream=True, logging_obj=MagicMock()
|
||||
)
|
||||
result = iterator.chunk_parser(chunk)
|
||||
|
||||
assert result._hidden_params["provider_specific_fields"]["traffic_type"] == "ON_DEMAND"
|
||||
|
||||
|
||||
def test_vertex_ai_traffic_type_preserved_in_hidden_params_non_streaming():
|
||||
"""Test trafficType is preserved in _hidden_params for non-streaming."""
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
|
||||
VertexGeminiConfig,
|
||||
)
|
||||
|
||||
completion_response = {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {"parts": [{"text": "Hello"}], "role": "model"},
|
||||
"finishReason": "STOP",
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 50,
|
||||
"candidatesTokenCount": 100,
|
||||
"totalTokenCount": 150,
|
||||
"trafficType": "PROVISIONED_THROUGHPUT",
|
||||
},
|
||||
}
|
||||
|
||||
raw_response = MagicMock()
|
||||
raw_response.json.return_value = completion_response
|
||||
|
||||
result = VertexGeminiConfig().transform_response(
|
||||
model="gemini-pro",
|
||||
raw_response=raw_response,
|
||||
model_response=ModelResponse(),
|
||||
logging_obj=MagicMock(),
|
||||
request_data={},
|
||||
messages=[],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
encoding=None,
|
||||
)
|
||||
|
||||
assert result._hidden_params["provider_specific_fields"]["traffic_type"] == "PROVISIONED_THROUGHPUT"
|
||||
|
||||
|
||||
def test_vertex_ai_traffic_type_surfaced_in_responses_api():
|
||||
"""Test trafficType is surfaced as provider_specific_fields in ResponsesAPIResponse."""
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
)
|
||||
|
||||
# Create a ModelResponse with provider_specific_fields in _hidden_params
|
||||
from litellm.types.utils import Choices, Message
|
||||
|
||||
model_response = ModelResponse()
|
||||
model_response._hidden_params["provider_specific_fields"] = {"traffic_type": "ON_DEMAND"}
|
||||
model_response.choices = [
|
||||
Choices(
|
||||
message=Message(content="Hello", role="assistant"),
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
)
|
||||
]
|
||||
|
||||
responses_api_response = LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response(
|
||||
request_input="test",
|
||||
chat_completion_response=model_response,
|
||||
responses_api_request={},
|
||||
)
|
||||
|
||||
assert responses_api_response.provider_specific_fields["traffic_type"] == "ON_DEMAND"
|
||||
|
||||
|
|
|
|||
|
|
@ -936,6 +936,109 @@ def test_proxy_admin_viewer_can_access_global_spend_tags():
|
|||
)
|
||||
|
||||
|
||||
class TestModelsRouteExemptFromDisableLLMEndpoints:
|
||||
"""
|
||||
Test that /models and /v1/models are exempt from DISABLE_LLM_API_ENDPOINTS.
|
||||
|
||||
When DISABLE_LLM_API_ENDPOINTS is set, inference routes like /v1/chat/completions
|
||||
should be blocked, but /models and /v1/models should remain accessible because
|
||||
they are read-only model listing routes needed by the Admin UI.
|
||||
|
||||
Relevant issue: https://github.com/BerriAI/litellm/issues/new (UI breaks with DISABLE_LLM_ENDPOINTS)
|
||||
"""
|
||||
|
||||
def _get_enterprise_route_checks(self):
|
||||
"""Import EnterpriseRouteChecks from the local enterprise source file."""
|
||||
import importlib.util
|
||||
|
||||
local_file = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"..", "..", "..", "..", "enterprise",
|
||||
"litellm_enterprise", "proxy", "auth", "route_checks.py",
|
||||
)
|
||||
local_file = os.path.abspath(local_file)
|
||||
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"local_enterprise_route_checks", local_file
|
||||
)
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
return mod.EnterpriseRouteChecks
|
||||
|
||||
@patch("litellm.proxy.proxy_server.premium_user", True)
|
||||
def test_should_models_route_allowed_when_llm_api_disabled(self):
|
||||
"""Test that /models is allowed even when LLM API routes are disabled"""
|
||||
EnterpriseRouteChecks = self._get_enterprise_route_checks()
|
||||
|
||||
with patch.object(
|
||||
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
|
||||
), patch.object(
|
||||
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
|
||||
):
|
||||
# /models should NOT raise - it's exempt
|
||||
EnterpriseRouteChecks.should_call_route("/models")
|
||||
|
||||
@patch("litellm.proxy.proxy_server.premium_user", True)
|
||||
def test_should_v1_models_route_allowed_when_llm_api_disabled(self):
|
||||
"""Test that /v1/models is allowed even when LLM API routes are disabled"""
|
||||
EnterpriseRouteChecks = self._get_enterprise_route_checks()
|
||||
|
||||
with patch.object(
|
||||
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
|
||||
), patch.object(
|
||||
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
|
||||
):
|
||||
# /v1/models should NOT raise - it's exempt
|
||||
EnterpriseRouteChecks.should_call_route("/v1/models")
|
||||
|
||||
@patch("litellm.proxy.proxy_server.premium_user", True)
|
||||
def test_should_chat_completions_still_blocked_when_llm_api_disabled(self):
|
||||
"""Test that non-exempt LLM routes like /v1/chat/completions are still blocked"""
|
||||
EnterpriseRouteChecks = self._get_enterprise_route_checks()
|
||||
|
||||
with patch.object(
|
||||
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
|
||||
), patch.object(
|
||||
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
EnterpriseRouteChecks.should_call_route("/v1/chat/completions")
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
assert "LLM API routes are disabled for this instance." in str(
|
||||
exc_info.value.detail
|
||||
)
|
||||
|
||||
@patch("litellm.proxy.proxy_server.premium_user", True)
|
||||
def test_should_embeddings_still_blocked_when_llm_api_disabled(self):
|
||||
"""Test that /v1/embeddings is still blocked when LLM API routes are disabled"""
|
||||
EnterpriseRouteChecks = self._get_enterprise_route_checks()
|
||||
|
||||
with patch.object(
|
||||
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
|
||||
), patch.object(
|
||||
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
EnterpriseRouteChecks.should_call_route("/v1/embeddings")
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
|
||||
@patch("litellm.proxy.proxy_server.premium_user", True)
|
||||
def test_should_models_route_allowed_when_llm_api_not_disabled(self):
|
||||
"""Test that /models works normally when LLM API routes are not disabled"""
|
||||
EnterpriseRouteChecks = self._get_enterprise_route_checks()
|
||||
|
||||
with patch.object(
|
||||
EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=False
|
||||
), patch.object(
|
||||
EnterpriseRouteChecks, "is_management_routes_disabled", return_value=False
|
||||
):
|
||||
# Should not raise
|
||||
EnterpriseRouteChecks.should_call_route("/models")
|
||||
EnterpriseRouteChecks.should_call_route("/v1/models")
|
||||
|
||||
|
||||
def test_route_in_additional_public_routes_wildcard_match():
|
||||
"""
|
||||
Test that route_in_additonal_public_routes supports wildcard patterns.
|
||||
|
|
|
|||
|
|
@ -697,3 +697,67 @@ def test_populate_request_with_path_params_does_not_overwrite_existing_values():
|
|||
assert result["organization_id"] == "org-existing" # Should keep original, not "org-query-param"
|
||||
# Verify other data is preserved
|
||||
assert result["messages"] == [{"role": "user", "content": "Hello"}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_request_body_with_html_script_tags():
|
||||
"""
|
||||
Test that JSON request bodies containing HTML tags like <script> are
|
||||
parsed correctly without being blocked or modified.
|
||||
|
||||
Regression test for GitHub issue #20441:
|
||||
https://github.com/BerriAI/litellm/issues/20441
|
||||
|
||||
LLM message content frequently contains HTML/code snippets.
|
||||
The HTTP parsing layer must not interfere with such content.
|
||||
"""
|
||||
test_messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "<script>alert('hello')</script>",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "<script> test </script>",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Can you explain what <script> tags do in HTML?",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Here is code: <div><script src='app.js'></script></div>",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "<img onerror='alert(1)' src='x'>",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "<iframe src='https://example.com'></iframe>",
|
||||
},
|
||||
]
|
||||
|
||||
for msg in test_messages:
|
||||
test_payload = {
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "Hello! How can I help?"},
|
||||
msg,
|
||||
],
|
||||
}
|
||||
|
||||
mock_request = MagicMock()
|
||||
mock_request.body = AsyncMock(return_value=orjson.dumps(test_payload))
|
||||
mock_request.headers = {"content-type": "application/json"}
|
||||
mock_request.scope = {}
|
||||
|
||||
result = await _read_request_body(mock_request)
|
||||
|
||||
assert result["model"] == "gpt-4o"
|
||||
assert len(result["messages"]) == 3
|
||||
assert result["messages"][2]["content"] == msg["content"], (
|
||||
f"Message content with HTML was modified during parsing: "
|
||||
f"expected={msg['content']!r}, got={result['messages'][2]['content']!r}"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -986,3 +986,146 @@ class TestContentFilterGuardrail:
|
|||
assert detail.get("category") == "harm_toxic_abuse"
|
||||
else:
|
||||
assert "harm_toxic_abuse" in str(detail)
|
||||
async def test_html_tags_in_messages_not_blocked(self):
|
||||
"""
|
||||
Test that HTML tags like <script> in LLM message content are NOT blocked
|
||||
by the content filter guardrail.
|
||||
|
||||
Regression test for GitHub issue #20441:
|
||||
https://github.com/BerriAI/litellm/issues/20441
|
||||
|
||||
LLM message content is not rendered as HTML, so HTML tags should be
|
||||
treated as plain text and should pass through without being blocked.
|
||||
"""
|
||||
# Set up a guardrail with all prebuilt patterns enabled as BLOCK
|
||||
patterns = [
|
||||
ContentFilterPattern(
|
||||
pattern_type="prebuilt",
|
||||
pattern_name="us_ssn",
|
||||
action=ContentFilterAction.BLOCK,
|
||||
),
|
||||
ContentFilterPattern(
|
||||
pattern_type="prebuilt",
|
||||
pattern_name="email",
|
||||
action=ContentFilterAction.BLOCK,
|
||||
),
|
||||
ContentFilterPattern(
|
||||
pattern_type="prebuilt",
|
||||
pattern_name="credit_card",
|
||||
action=ContentFilterAction.BLOCK,
|
||||
),
|
||||
]
|
||||
|
||||
guardrail = ContentFilterGuardrail(
|
||||
guardrail_name="test-html-tags",
|
||||
patterns=patterns,
|
||||
)
|
||||
|
||||
# Messages containing <script> and other HTML tags should NOT be blocked
|
||||
html_messages = [
|
||||
"<script>alert('hello')</script>",
|
||||
"<script> test </script>",
|
||||
"Can you explain what <script> tags do in HTML?",
|
||||
"Here is some code: <div><script src='app.js'></script></div>",
|
||||
"<img onerror='alert(1)' src='x'>",
|
||||
"<iframe src='https://example.com'></iframe>",
|
||||
"The <style> and <script> elements are important in HTML",
|
||||
"<a href='javascript:void(0)'>click me</a>",
|
||||
]
|
||||
|
||||
for message in html_messages:
|
||||
# Should NOT raise HTTPException
|
||||
result = await guardrail.apply_guardrail(
|
||||
inputs={"texts": [message]},
|
||||
request_data={},
|
||||
input_type="request",
|
||||
)
|
||||
processed_texts = result.get("texts", [])
|
||||
assert len(processed_texts) == 1
|
||||
# Content should pass through unchanged (no HTML tags are patterns)
|
||||
assert processed_texts[0] == message, (
|
||||
f"Message containing HTML was unexpectedly modified: "
|
||||
f"input={message!r}, output={processed_texts[0]!r}"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_script_tag_not_blocked_with_blocked_words(self):
|
||||
"""
|
||||
Test that <script> tags are not accidentally caught by blocked words
|
||||
unless explicitly configured.
|
||||
|
||||
Regression test for GitHub issue #20441.
|
||||
"""
|
||||
blocked_words = [
|
||||
BlockedWord(
|
||||
keyword="confidential",
|
||||
action=ContentFilterAction.BLOCK,
|
||||
),
|
||||
BlockedWord(
|
||||
keyword="secret_project",
|
||||
action=ContentFilterAction.BLOCK,
|
||||
),
|
||||
]
|
||||
|
||||
guardrail = ContentFilterGuardrail(
|
||||
guardrail_name="test-script-not-blocked",
|
||||
blocked_words=blocked_words,
|
||||
)
|
||||
|
||||
# <script> should not be caught by unrelated blocked words
|
||||
script_messages = [
|
||||
"<script>alert('test')</script>",
|
||||
"How do I use <script> tags in HTML?",
|
||||
"<script src='app.js'></script>",
|
||||
]
|
||||
|
||||
for message in script_messages:
|
||||
result = await guardrail.apply_guardrail(
|
||||
inputs={"texts": [message]},
|
||||
request_data={},
|
||||
input_type="request",
|
||||
)
|
||||
processed_texts = result.get("texts", [])
|
||||
assert len(processed_texts) == 1
|
||||
assert processed_texts[0] == message
|
||||
|
||||
def test_no_builtin_pattern_matches_script_tag(self):
|
||||
"""
|
||||
Test that NONE of the prebuilt patterns in patterns.json match
|
||||
the string '<script>' or common HTML tags.
|
||||
|
||||
This is a safeguard to ensure that future pattern additions
|
||||
do not accidentally block legitimate LLM content containing
|
||||
HTML/code snippets.
|
||||
|
||||
Regression test for GitHub issue #20441.
|
||||
"""
|
||||
from litellm.proxy.guardrails.guardrail_hooks.litellm_content_filter.patterns import (
|
||||
PREBUILT_PATTERNS,
|
||||
get_compiled_pattern,
|
||||
)
|
||||
|
||||
html_test_strings = [
|
||||
"<script>alert('xss')</script>",
|
||||
"<script> test </script>",
|
||||
"<script src='app.js'></script>",
|
||||
"<img onerror='alert(1)' src='x'>",
|
||||
"<iframe src='https://example.com'></iframe>",
|
||||
"<style>body { color: red; }</style>",
|
||||
"<div onclick='alert(1)'>click</div>",
|
||||
]
|
||||
|
||||
for pattern_name in PREBUILT_PATTERNS:
|
||||
compiled = get_compiled_pattern(pattern_name)
|
||||
for test_string in html_test_strings:
|
||||
match = compiled.search(test_string)
|
||||
if match:
|
||||
# Some patterns may legitimately match substrings
|
||||
# (e.g., URL pattern matching src='https://...')
|
||||
# but they should not match the script/HTML tag itself
|
||||
matched_text = match.group()
|
||||
assert "<script" not in matched_text.lower(), (
|
||||
f"Pattern '{pattern_name}' matched '<script>' in "
|
||||
f"test string: {test_string!r}. "
|
||||
f"LLM message content should not be blocked for HTML tags."
|
||||
)
|
||||
|
|
|
|||
|
|
@ -7,7 +7,6 @@ import sys
|
|||
|
||||
sys.path.insert(0, os.path.abspath("../../../../../.."))
|
||||
|
||||
import asyncio
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
|
@ -26,7 +25,7 @@ async def test_openai_moderation_guardrail_init():
|
|||
guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
)
|
||||
|
||||
|
||||
assert guardrail.guardrail_name == "test-openai-moderation"
|
||||
assert guardrail.api_key == "test-key"
|
||||
assert guardrail.model == "omni-moderation-latest"
|
||||
|
|
@ -49,27 +48,27 @@ async def test_openai_moderation_guardrail_adds_to_litellm_callbacks():
|
|||
# Clear existing callbacks for clean test
|
||||
original_callbacks = litellm.callbacks.copy()
|
||||
litellm.logging_callback_manager._reset_all_callbacks()
|
||||
|
||||
|
||||
try:
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
guardrail_litellm_params = LitellmParams(
|
||||
guardrail=SupportedGuardrailIntegrations.OPENAI_MODERATION,
|
||||
api_key="test-key",
|
||||
model="omni-moderation-latest",
|
||||
mode="pre_call"
|
||||
mode="pre_call",
|
||||
)
|
||||
guardrail = openai_initialize_guardrail(
|
||||
litellm_params=guardrail_litellm_params,
|
||||
guardrail=Guardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
litellm_params=guardrail_litellm_params
|
||||
)
|
||||
litellm_params=guardrail_litellm_params,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
# Check that the guardrail was added to litellm callbacks
|
||||
assert guardrail in litellm.callbacks
|
||||
assert len(litellm.callbacks) == 1
|
||||
|
||||
|
||||
# Verify it's the correct guardrail
|
||||
callback = litellm.callbacks[0]
|
||||
assert isinstance(callback, OpenAIModerationGuardrail)
|
||||
|
|
@ -85,12 +84,12 @@ async def test_openai_moderation_guardrail_adds_to_litellm_callbacks():
|
|||
async def test_openai_moderation_guardrail_safe_content():
|
||||
"""Test OpenAI moderation guardrail with safe content via apply_guardrail"""
|
||||
from litellm.types.utils import GenericGuardrailAPIInputs
|
||||
|
||||
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
)
|
||||
|
||||
|
||||
# Mock safe moderation response
|
||||
mock_response = OpenAIModerationResponse(
|
||||
id="modr-123",
|
||||
|
|
@ -118,25 +117,29 @@ async def test_openai_moderation_guardrail_safe_content():
|
|||
"harassment": [],
|
||||
"self-harm": [],
|
||||
"violence": [],
|
||||
}
|
||||
},
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
with patch.object(guardrail, 'async_make_request', return_value=mock_response):
|
||||
|
||||
with patch.object(guardrail, "async_make_request", return_value=mock_response):
|
||||
# Test apply_guardrail with safe content using structured_messages
|
||||
inputs = GenericGuardrailAPIInputs(
|
||||
structured_messages=[
|
||||
{"role": "user", "content": "Hello, how are you today?"}
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
result = await guardrail.apply_guardrail(
|
||||
inputs=inputs,
|
||||
request_data={"messages": [{"role": "user", "content": "Hello, how are you today?"}]},
|
||||
input_type="request"
|
||||
request_data={
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello, how are you today?"}
|
||||
]
|
||||
},
|
||||
input_type="request",
|
||||
)
|
||||
|
||||
|
||||
# Should return the original inputs unchanged
|
||||
assert result == inputs
|
||||
|
||||
|
|
@ -145,12 +148,12 @@ async def test_openai_moderation_guardrail_safe_content():
|
|||
async def test_openai_moderation_guardrail_apply_guardrail():
|
||||
"""Test OpenAI moderation guardrail apply_guardrail method (unified guardrail interface)"""
|
||||
from litellm.types.utils import GenericGuardrailAPIInputs
|
||||
|
||||
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
)
|
||||
|
||||
|
||||
# Mock safe moderation response
|
||||
mock_response = OpenAIModerationResponse(
|
||||
id="modr-123",
|
||||
|
|
@ -178,37 +181,37 @@ async def test_openai_moderation_guardrail_apply_guardrail():
|
|||
"harassment": [],
|
||||
"self-harm": [],
|
||||
"violence": [],
|
||||
}
|
||||
},
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
with patch.object(guardrail, 'async_make_request', return_value=mock_response):
|
||||
|
||||
with patch.object(guardrail, "async_make_request", return_value=mock_response):
|
||||
# Test apply_guardrail with texts (embeddings-style input)
|
||||
inputs = GenericGuardrailAPIInputs(
|
||||
texts=["Hello, how are you?", "What is the weather?"]
|
||||
)
|
||||
|
||||
|
||||
result = await guardrail.apply_guardrail(
|
||||
inputs=inputs,
|
||||
request_data={},
|
||||
input_type="request",
|
||||
)
|
||||
|
||||
|
||||
# Should return inputs unchanged (moderation doesn't modify, only blocks)
|
||||
assert result == inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_moderation_guardrail_harmful_content():
|
||||
"""Test OpenAI moderation guardrail with harmful content via apply_guardrail"""
|
||||
from litellm.types.utils import GenericGuardrailAPIInputs
|
||||
|
||||
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
)
|
||||
|
||||
|
||||
# Mock harmful moderation response
|
||||
mock_response = OpenAIModerationResponse(
|
||||
id="modr-123",
|
||||
|
|
@ -236,40 +239,51 @@ async def test_openai_moderation_guardrail_harmful_content():
|
|||
"harassment": [],
|
||||
"self-harm": [],
|
||||
"violence": [],
|
||||
}
|
||||
},
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
with patch.object(guardrail, 'async_make_request', return_value=mock_response):
|
||||
|
||||
with patch.object(guardrail, "async_make_request", return_value=mock_response):
|
||||
# Test apply_guardrail with harmful content using structured_messages
|
||||
inputs = GenericGuardrailAPIInputs(
|
||||
structured_messages=[
|
||||
{"role": "user", "content": "This is hateful content"}
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
# Should raise HTTPException
|
||||
from fastapi import HTTPException
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await guardrail.apply_guardrail(
|
||||
inputs=inputs,
|
||||
request_data={"messages": [{"role": "user", "content": "This is hateful content"}]},
|
||||
input_type="request"
|
||||
request_data={
|
||||
"messages": [
|
||||
{"role": "user", "content": "This is hateful content"}
|
||||
]
|
||||
},
|
||||
input_type="request",
|
||||
)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_moderation_guardrail_streaming_safe_content():
|
||||
"""Test OpenAI moderation guardrail with streaming safe content"""
|
||||
"""Test OpenAI moderation guardrail with streaming safe content via UnifiedLLMGuardrails"""
|
||||
from litellm.proxy.guardrails.guardrail_hooks.unified_guardrail.unified_guardrail import (
|
||||
UnifiedLLMGuardrails,
|
||||
)
|
||||
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
event_hook="post_call",
|
||||
)
|
||||
|
||||
unified_guardrail = UnifiedLLMGuardrails()
|
||||
|
||||
# Mock safe moderation response
|
||||
mock_response = OpenAIModerationResponse(
|
||||
id="modr-123",
|
||||
|
|
@ -297,72 +311,85 @@ async def test_openai_moderation_guardrail_streaming_safe_content():
|
|||
"harassment": [],
|
||||
"self-harm": [],
|
||||
"violence": [],
|
||||
}
|
||||
},
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# Mock streaming chunks
|
||||
async def mock_stream():
|
||||
# Simulate streaming chunks with safe content
|
||||
chunks = [
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="Hello "))]),
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="world"))]),
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="!"))])
|
||||
]
|
||||
for chunk in chunks:
|
||||
chunk1 = MagicMock()
|
||||
chunk1.model = "gpt-4"
|
||||
chunk1.choices = [MagicMock()]
|
||||
chunk1.choices[0].delta = MagicMock()
|
||||
chunk1.choices[0].delta.content = "Hello "
|
||||
chunk1.choices[0].finish_reason = None
|
||||
|
||||
chunk2 = MagicMock()
|
||||
chunk2.model = "gpt-4"
|
||||
chunk2.choices = [MagicMock()]
|
||||
chunk2.choices[0].delta = MagicMock()
|
||||
chunk2.choices[0].delta.content = "world"
|
||||
chunk2.choices[0].finish_reason = None
|
||||
|
||||
# Last chunk with finish_reason
|
||||
chunk3 = MagicMock()
|
||||
chunk3.model = "gpt-4"
|
||||
chunk3.choices = [MagicMock()]
|
||||
chunk3.choices[0].delta = MagicMock()
|
||||
chunk3.choices[0].delta.content = "!"
|
||||
chunk3.choices[0].finish_reason = "stop"
|
||||
|
||||
for chunk in [chunk1, chunk2, chunk3]:
|
||||
yield chunk
|
||||
|
||||
# Mock the stream_chunk_builder to return a proper ModelResponse
|
||||
|
||||
# Mock for stream_chunk_builder
|
||||
mock_model_response = MagicMock()
|
||||
mock_model_response.choices = [
|
||||
MagicMock(message=MagicMock(content="Hello world!"))
|
||||
]
|
||||
|
||||
with patch.object(guardrail, 'async_make_request', return_value=mock_response), \
|
||||
patch('litellm.main.stream_chunk_builder', return_value=mock_model_response), \
|
||||
patch('litellm.llms.base_llm.base_model_iterator.MockResponseIterator') as mock_iterator:
|
||||
|
||||
# Mock the iterator to yield the original chunks
|
||||
async def mock_yield_chunks():
|
||||
chunks = [
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="Hello "))]),
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="world"))]),
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="!"))])
|
||||
]
|
||||
for chunk in chunks:
|
||||
yield chunk
|
||||
|
||||
mock_iterator.return_value.__aiter__ = lambda self: mock_yield_chunks()
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(api_key="test")
|
||||
mock_model_response.choices = [MagicMock()]
|
||||
mock_model_response.choices[0].message = MagicMock()
|
||||
mock_model_response.choices[0].message.content = "Hello world!"
|
||||
|
||||
with patch.object(guardrail, "async_make_request", return_value=mock_response), patch(
|
||||
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
|
||||
return_value=mock_model_response,
|
||||
):
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key="test", request_route="/chat/completions"
|
||||
)
|
||||
request_data = {
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello, how are you today?"}
|
||||
]
|
||||
"messages": [{"role": "user", "content": "Hello, how are you today?"}],
|
||||
"guardrail_to_apply": guardrail,
|
||||
"metadata": {"guardrails": ["test-openai-moderation"]},
|
||||
}
|
||||
|
||||
# Test streaming hook with safe content
|
||||
|
||||
# Test streaming hook with safe content via UnifiedLLMGuardrails
|
||||
result_chunks = []
|
||||
async for chunk in guardrail.async_post_call_streaming_iterator_hook(
|
||||
async for chunk in unified_guardrail.async_post_call_streaming_iterator_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
response=mock_stream(),
|
||||
request_data=request_data
|
||||
request_data=request_data,
|
||||
):
|
||||
result_chunks.append(chunk)
|
||||
|
||||
|
||||
# Should return all chunks without blocking
|
||||
assert len(result_chunks) == 3
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_moderation_guardrail_streaming_harmful_content():
|
||||
"""Test OpenAI moderation guardrail with streaming harmful content"""
|
||||
"""Test OpenAI moderation guardrail with streaming harmful content via UnifiedLLMGuardrails"""
|
||||
from litellm.proxy.guardrails.guardrail_hooks.unified_guardrail.unified_guardrail import (
|
||||
UnifiedLLMGuardrails,
|
||||
)
|
||||
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
event_hook="post_call",
|
||||
)
|
||||
|
||||
unified_guardrail = UnifiedLLMGuardrails()
|
||||
|
||||
# Mock harmful moderation response
|
||||
mock_response = OpenAIModerationResponse(
|
||||
id="modr-123",
|
||||
|
|
@ -390,46 +417,74 @@ async def test_openai_moderation_guardrail_streaming_harmful_content():
|
|||
"harassment": [],
|
||||
"self-harm": [],
|
||||
"violence": [],
|
||||
}
|
||||
},
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# Mock streaming chunks with harmful content
|
||||
async def mock_stream():
|
||||
chunks = [
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="This is "))]),
|
||||
MagicMock(choices=[MagicMock(delta=MagicMock(content="harmful content"))])
|
||||
]
|
||||
for chunk in chunks:
|
||||
# First chunk - no finish_reason
|
||||
chunk1 = MagicMock()
|
||||
chunk1.model = "gpt-4"
|
||||
chunk1.choices = [MagicMock()]
|
||||
chunk1.choices[0].delta = MagicMock()
|
||||
chunk1.choices[0].delta.content = "This is "
|
||||
chunk1.choices[0].finish_reason = None
|
||||
|
||||
# Last chunk - with finish_reason to signal end of stream
|
||||
chunk2 = MagicMock()
|
||||
chunk2.model = "gpt-4"
|
||||
chunk2.choices = [MagicMock()]
|
||||
chunk2.choices[0].delta = MagicMock()
|
||||
chunk2.choices[0].delta.content = "harmful content"
|
||||
chunk2.choices[0].finish_reason = "stop"
|
||||
|
||||
for chunk in [chunk1, chunk2]:
|
||||
yield chunk
|
||||
|
||||
# Mock the stream_chunk_builder to return a ModelResponse with harmful content
|
||||
mock_model_response = MagicMock()
|
||||
mock_model_response.choices = [
|
||||
MagicMock(message=MagicMock(content="This is harmful content"))
|
||||
]
|
||||
|
||||
with patch.object(guardrail, 'async_make_request', return_value=mock_response), \
|
||||
patch('litellm.main.stream_chunk_builder', return_value=mock_model_response):
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(api_key="test")
|
||||
|
||||
# Mock for stream_chunk_builder - use real litellm types so isinstance checks pass
|
||||
from litellm.types.utils import ModelResponse
|
||||
import litellm
|
||||
mock_model_response = ModelResponse(
|
||||
id="mock-response",
|
||||
model="gpt-4",
|
||||
choices=[
|
||||
litellm.Choices(
|
||||
index=0,
|
||||
message=litellm.Message(
|
||||
role="assistant",
|
||||
content="This is harmful content",
|
||||
),
|
||||
finish_reason="stop",
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
with patch.object(guardrail, "async_make_request", return_value=mock_response), patch(
|
||||
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
|
||||
return_value=mock_model_response,
|
||||
):
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key="test", request_route="/chat/completions"
|
||||
)
|
||||
request_data = {
|
||||
"messages": [
|
||||
{"role": "user", "content": "Generate harmful content"}
|
||||
]
|
||||
"messages": [{"role": "user", "content": "Generate harmful content"}],
|
||||
"guardrail_to_apply": guardrail,
|
||||
"metadata": {"guardrails": ["test-openai-moderation"]},
|
||||
}
|
||||
|
||||
|
||||
# Should raise HTTPException when processing streaming harmful content
|
||||
from fastapi import HTTPException
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
result_chunks = []
|
||||
async for chunk in guardrail.async_post_call_streaming_iterator_hook(
|
||||
async for chunk in unified_guardrail.async_post_call_streaming_iterator_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
response=mock_stream(),
|
||||
request_data=request_data
|
||||
request_data=request_data,
|
||||
):
|
||||
result_chunks.append(chunk)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)
|
||||
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,172 @@
|
|||
import pytest
|
||||
from unittest.mock import MagicMock, patch
|
||||
import os
|
||||
from litellm.proxy.guardrails.guardrail_hooks.openai.moderations import (
|
||||
OpenAIModerationGuardrail,
|
||||
)
|
||||
from litellm.proxy.guardrails.guardrail_hooks.unified_guardrail.unified_guardrail import (
|
||||
UnifiedLLMGuardrails,
|
||||
)
|
||||
from litellm.types.utils import ModelResponseStream, ModelResponse
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_moderation_guardrail_streaming_latency():
|
||||
"""
|
||||
Test that the OpenAI Moderation guardrail, when run via UnifiedLLMGuardrails,
|
||||
supports streaming (fast time-to-first-token) instead of buffering.
|
||||
"""
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
# 1. Initialize the specific guardrail with proper event_hook
|
||||
openai_guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
event_hook="post_call",
|
||||
)
|
||||
|
||||
# 2. Initialize the Unified Guardrail system (which invokes the specific guardrail)
|
||||
unified_guardrail = UnifiedLLMGuardrails()
|
||||
|
||||
# Mock safe moderation response
|
||||
mock_mod_response = MagicMock()
|
||||
mock_mod_response.results = []
|
||||
|
||||
# Mock streaming chunks (no artificial delay - test deterministically)
|
||||
async def mock_stream():
|
||||
chunks_data = ["Hello", " ", "world", "!", " Goodbye"]
|
||||
for i, content in enumerate(chunks_data):
|
||||
chunk = MagicMock(spec=ModelResponseStream)
|
||||
chunk.model = "gpt-4"
|
||||
choice = MagicMock()
|
||||
choice.delta = MagicMock()
|
||||
choice.delta.content = content
|
||||
# Last chunk gets finish_reason
|
||||
choice.finish_reason = "stop" if i == len(chunks_data) - 1 else None
|
||||
chunk.choices = [choice]
|
||||
yield chunk
|
||||
|
||||
# Mock for stream_chunk_builder to return a simple ModelResponse
|
||||
mock_model_response = MagicMock(spec=ModelResponse)
|
||||
mock_model_response.choices = [MagicMock()]
|
||||
mock_model_response.choices[0].message = MagicMock()
|
||||
mock_model_response.choices[0].message.content = "Hello world! Goodbye"
|
||||
|
||||
# Patch the network call in the specific guardrail
|
||||
with patch.object(
|
||||
openai_guardrail, "async_make_request", return_value=mock_mod_response
|
||||
), patch(
|
||||
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
|
||||
return_value=mock_model_response,
|
||||
):
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key="test", request_route="/chat/completions"
|
||||
)
|
||||
request_data = {
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"guardrail_to_apply": openai_guardrail,
|
||||
"metadata": {
|
||||
"guardrails": ["test-openai-moderation"],
|
||||
"guardrail_config": {"streaming_sampling_rate": 1},
|
||||
}, # Check every chunk for test
|
||||
}
|
||||
|
||||
chunks_received = 0
|
||||
first_chunk_yielded = False
|
||||
|
||||
# Call the hook on UnifiedLLMGuardrails
|
||||
async for chunk in unified_guardrail.async_post_call_streaming_iterator_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
response=mock_stream(),
|
||||
request_data=request_data,
|
||||
):
|
||||
if not first_chunk_yielded:
|
||||
first_chunk_yielded = True
|
||||
chunks_received += 1
|
||||
|
||||
# Deterministic assertions (no flaky timing checks)
|
||||
assert first_chunk_yielded, "Expected at least one chunk to be yielded"
|
||||
assert chunks_received == 5, f"Expected 5 chunks, got {chunks_received}"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_moderation_guardrail_streaming_harmful_content():
|
||||
"""
|
||||
Test that harmful content is caught during streaming via UnifiedLLMGuardrails
|
||||
"""
|
||||
from fastapi import HTTPException
|
||||
|
||||
with patch.dict(os.environ, {"OPENAI_API_KEY": "test-key"}):
|
||||
openai_guardrail = OpenAIModerationGuardrail(
|
||||
guardrail_name="test-openai-moderation",
|
||||
event_hook="post_call",
|
||||
)
|
||||
unified_guardrail = UnifiedLLMGuardrails()
|
||||
|
||||
# Mock harmful moderation response
|
||||
mock_mod_response = MagicMock()
|
||||
mock_mod_response.results = [
|
||||
MagicMock(
|
||||
flagged=True, categories={"hate": True}, category_scores={"hate": 0.99}
|
||||
)
|
||||
]
|
||||
|
||||
async def mock_stream():
|
||||
chunks_data = ["This ", "is ", "harmful ", "content"]
|
||||
for i, content in enumerate(chunks_data):
|
||||
chunk = MagicMock(spec=ModelResponseStream)
|
||||
chunk.model = "gpt-4"
|
||||
choice = MagicMock()
|
||||
choice.delta = MagicMock()
|
||||
choice.delta.content = content
|
||||
# Last chunk gets finish_reason
|
||||
choice.finish_reason = "stop" if i == len(chunks_data) - 1 else None
|
||||
chunk.choices = [choice]
|
||||
yield chunk
|
||||
|
||||
# Mock for stream_chunk_builder - use real litellm types so isinstance checks pass
|
||||
import litellm
|
||||
|
||||
mock_model_response = ModelResponse(
|
||||
id="mock-response",
|
||||
model="gpt-4",
|
||||
choices=[
|
||||
litellm.Choices(
|
||||
index=0,
|
||||
message=litellm.Message(
|
||||
role="assistant",
|
||||
content="This is harmful content",
|
||||
),
|
||||
finish_reason="stop",
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
with patch.object(
|
||||
openai_guardrail, "async_make_request", return_value=mock_mod_response
|
||||
), patch(
|
||||
"litellm.llms.openai.chat.guardrail_translation.handler.stream_chunk_builder",
|
||||
return_value=mock_model_response,
|
||||
):
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key="test", request_route="/chat/completions"
|
||||
)
|
||||
request_data = {
|
||||
"messages": [{"role": "user", "content": "generate hate"}],
|
||||
"guardrail_to_apply": openai_guardrail,
|
||||
"metadata": {
|
||||
"guardrails": ["test-openai-moderation"],
|
||||
"guardrail_config": {"streaming_sampling_rate": 1},
|
||||
},
|
||||
}
|
||||
|
||||
# Should raise HTTPException
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
async for _ in unified_guardrail.async_post_call_streaming_iterator_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
response=mock_stream(),
|
||||
request_data=request_data,
|
||||
):
|
||||
pass
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "Violated OpenAI moderation policy" in str(exc_info.value.detail)
|
||||
|
|
@ -1122,6 +1122,83 @@ async def test_model_armor_non_model_response():
|
|||
assert not guardrail.async_handler.post.called
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_model_armor_guardrail_status_intervened_vs_failed():
|
||||
"""
|
||||
regression test for bug where _process_error always set 'guardrail_failed_to_respond'
|
||||
even for intentional blocks (error 400).
|
||||
"""
|
||||
mock_user_api_key_dict = UserAPIKeyAuth()
|
||||
mock_cache = MagicMock(spec=DualCache)
|
||||
|
||||
#1: Blocked content should raise exception and show guardrail status: guardrail_intervened"
|
||||
guardrail = ModelArmorGuardrail(
|
||||
template_id="test-template",
|
||||
project_id="test-project",
|
||||
location="us-central1",
|
||||
guardrail_name="model-armor-test",
|
||||
)
|
||||
|
||||
mock_response = AsyncMock()
|
||||
mock_response.status_code = 200
|
||||
mock_response.json = AsyncMock(return_value={
|
||||
"sanitizationResult": {
|
||||
"filterMatchState": "MATCH_FOUND",
|
||||
"filterResults": {
|
||||
"rai": {
|
||||
"raiFilterResult": {
|
||||
"matchState": "MATCH_FOUND",
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
guardrail._ensure_access_token_async = AsyncMock(return_value=("token", "test-project"))
|
||||
with patch.object(guardrail.async_handler, "post", AsyncMock(return_value=mock_response)):
|
||||
request_data = {
|
||||
"model": "gpt-4",
|
||||
"messages": [{"role": "user", "content": "bad content"}],
|
||||
"metadata": {"guardrails": ["model-armor-test"]},
|
||||
}
|
||||
with pytest.raises(HTTPException):
|
||||
await guardrail.async_pre_call_hook(
|
||||
user_api_key_dict=mock_user_api_key_dict,
|
||||
cache=mock_cache,
|
||||
data=request_data,
|
||||
call_type="completion",
|
||||
)
|
||||
|
||||
info = request_data["metadata"]["standard_logging_guardrail_information"]
|
||||
assert info[0]["guardrail_status"] == "guardrail_intervened"
|
||||
|
||||
#2: if an API error - guardrail status should be guardrail_failed_to_respond"
|
||||
guardrail2 = ModelArmorGuardrail(
|
||||
template_id="test-template",
|
||||
project_id="test-project",
|
||||
location="us-central1",
|
||||
guardrail_name="model-armor-test2",
|
||||
fail_on_error=True,
|
||||
)
|
||||
|
||||
guardrail2._ensure_access_token_async = AsyncMock(side_effect=ConnectionError("timeout"))
|
||||
request_data2 = {
|
||||
"model": "gpt-4",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"metadata": {"guardrails": ["model-armor-test2"]},
|
||||
}
|
||||
with pytest.raises(ConnectionError):
|
||||
await guardrail2.async_pre_call_hook(
|
||||
user_api_key_dict=mock_user_api_key_dict,
|
||||
cache=mock_cache,
|
||||
data=request_data2,
|
||||
call_type="completion",
|
||||
)
|
||||
|
||||
info2 = request_data2["metadata"]["standard_logging_guardrail_information"]
|
||||
assert info2[0]["guardrail_status"] == "guardrail_failed_to_respond"
|
||||
|
||||
|
||||
def mock_open(read_data=''):
|
||||
"""Helper to create a mock file object"""
|
||||
import io
|
||||
|
|
|
|||
136
tests/test_litellm/proxy/test_api_key_masking_in_errors.py
Normal file
136
tests/test_litellm/proxy/test_api_key_masking_in_errors.py
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
"""
|
||||
Tests that API keys are masked in error responses.
|
||||
|
||||
When an invalid/malformed API key is sent (e.g., with a leading space or
|
||||
wrong prefix), the error response must NOT return the key in plain text.
|
||||
Instead, it should show only the first 4 and last 4 characters with ****
|
||||
in the middle.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
class TestKeyMaskingInAuthErrors:
|
||||
"""Test that user_api_key_auth masks keys in validation error messages."""
|
||||
|
||||
def test_assert_message_masks_key_without_sk_prefix(self):
|
||||
"""
|
||||
When a key doesn't start with 'sk-', the AssertionError message
|
||||
should contain a masked version, not the full key.
|
||||
"""
|
||||
from litellm.proxy.auth.auth_utils import abbreviate_api_key
|
||||
|
||||
# Simulate the logic from user_api_key_auth.py
|
||||
api_key = "my-secret-api-key-1234567890abcdef"
|
||||
_masked_key = (
|
||||
"{}****{}".format(api_key[:4], api_key[-4:])
|
||||
if len(api_key) > 8
|
||||
else "****"
|
||||
)
|
||||
|
||||
# The masked key should NOT contain the full original key
|
||||
assert api_key not in _masked_key
|
||||
# Should show first 4 and last 4 chars
|
||||
assert _masked_key == "my-s****cdef"
|
||||
|
||||
def test_assert_message_masks_key_with_leading_space(self):
|
||||
"""
|
||||
Reported case: key with leading space like ' sk-abc123...'
|
||||
"""
|
||||
api_key = " sk-abc123def456ghi789jkl012mno345pqr"
|
||||
_masked_key = (
|
||||
"{}****{}".format(api_key[:4], api_key[-4:])
|
||||
if len(api_key) > 8
|
||||
else "****"
|
||||
)
|
||||
|
||||
assert api_key not in _masked_key
|
||||
assert _masked_key == " sk-****5pqr"
|
||||
|
||||
def test_assert_message_masks_short_key(self):
|
||||
"""Short keys (<=8 chars) should be fully masked."""
|
||||
api_key = "short"
|
||||
_masked_key = (
|
||||
"{}****{}".format(api_key[:4], api_key[-4:])
|
||||
if len(api_key) > 8
|
||||
else "****"
|
||||
)
|
||||
assert _masked_key == "****"
|
||||
|
||||
def test_key_not_starting_with_sk_raises_masked_error(self):
|
||||
"""
|
||||
Verify the assert message format contains masked key, not the original.
|
||||
|
||||
Note: Python's AssertionError str(e) includes the expression + message,
|
||||
but the *message* part (which is what gets passed to ProxyException)
|
||||
should only contain the masked key.
|
||||
"""
|
||||
api_key = "bad-key-format-1234567890abcdefghijklmnop"
|
||||
_masked_key = (
|
||||
"{}****{}".format(api_key[:4], api_key[-4:])
|
||||
if len(api_key) > 8
|
||||
else "****"
|
||||
)
|
||||
|
||||
# Build the same message string that user_api_key_auth.py would produce
|
||||
error_message = "LiteLLM Virtual Key expected. Received={}, expected to start with 'sk-'.".format(
|
||||
_masked_key
|
||||
)
|
||||
# The full key must NOT appear in the message
|
||||
assert api_key not in error_message
|
||||
# The masked version should appear
|
||||
assert _masked_key in error_message
|
||||
# Should still have helpful context
|
||||
assert "expected to start with 'sk-'" in error_message
|
||||
|
||||
|
||||
class TestKeyMaskingInKeyManagement:
|
||||
"""Test that key_management_endpoints masks keys in validation errors."""
|
||||
|
||||
def test_invalid_key_format_error_is_masked(self):
|
||||
"""
|
||||
When creating a key that doesn't start with 'sk-', the error
|
||||
should not include the full key value.
|
||||
"""
|
||||
key_value = "bad-prefix-1234567890abcdefghijklmnop"
|
||||
_masked = (
|
||||
"{}****{}".format(key_value[:4], key_value[-4:])
|
||||
if len(key_value) > 8
|
||||
else "****"
|
||||
)
|
||||
|
||||
error_msg = f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {_masked}"
|
||||
|
||||
# Full key must not appear
|
||||
assert key_value not in error_msg
|
||||
# Masked version should appear
|
||||
assert _masked in error_msg
|
||||
assert "bad-****mnop" in error_msg
|
||||
|
||||
|
||||
class TestPresidioErrorSanitization:
|
||||
"""Test that Presidio errors don't leak request text containing keys."""
|
||||
|
||||
def test_analyze_text_error_does_not_leak_text(self):
|
||||
"""
|
||||
If Presidio analyzer fails, the error message should NOT contain
|
||||
the original text that was being analyzed.
|
||||
"""
|
||||
# Simulate what happens: user message contains an API key,
|
||||
# Presidio fails, error message should be sanitized
|
||||
original_text = "Please use this key: sk-secret1234567890abcdefghijklmnop"
|
||||
|
||||
# The sanitized exception from our fix
|
||||
sanitized_error = f"Presidio PII analysis failed: ConnectionError"
|
||||
|
||||
assert original_text not in sanitized_error
|
||||
assert "sk-secret1234567890abcdefghijklmnop" not in sanitized_error
|
||||
|
||||
def test_anonymize_text_error_does_not_leak_text(self):
|
||||
"""
|
||||
If Presidio anonymizer fails, the error should be sanitized.
|
||||
"""
|
||||
sanitized_error = f"Presidio PII anonymization failed: ClientError"
|
||||
|
||||
assert "sk-" not in sanitized_error
|
||||
assert "api_key" not in sanitized_error
|
||||
|
|
@ -2,6 +2,7 @@ import base64
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
|
@ -352,3 +353,34 @@ class TestResponsesAPIProviderSpecificParams:
|
|||
# Should not raise any exception
|
||||
result = ResponsesAPIRequestUtils.get_requested_response_api_optional_param(params)
|
||||
assert "temperature" in result
|
||||
|
||||
|
||||
def test_responses_extra_body_forwarded_to_completion_transformation_handler():
|
||||
"""
|
||||
Regression test: extra_body must be forwarded to response_api_handler
|
||||
when responses_api_provider_config is None (completion transformation path).
|
||||
|
||||
Before the fix, extra_body was a named parameter of responses() but was
|
||||
not passed to litellm_completion_transformation_handler.response_api_handler(),
|
||||
so it was silently dropped.
|
||||
"""
|
||||
with patch(
|
||||
"litellm.responses.main.ProviderConfigManager.get_provider_responses_api_config",
|
||||
return_value=None,
|
||||
), patch(
|
||||
"litellm.responses.main.litellm_completion_transformation_handler.response_api_handler",
|
||||
) as mock_handler:
|
||||
mock_handler.return_value = MagicMock()
|
||||
|
||||
litellm.responses(
|
||||
model="openai/gpt-4o",
|
||||
input="Hello",
|
||||
extra_body={"custom_key": "custom_value"},
|
||||
)
|
||||
|
||||
mock_handler.assert_called_once()
|
||||
call_kwargs = mock_handler.call_args
|
||||
# extra_body can be a positional or keyword arg; check both
|
||||
assert call_kwargs.kwargs.get("extra_body") == {
|
||||
"custom_key": "custom_value"
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,232 @@
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.router_strategy.budget_limiter import RouterBudgetLimiting
|
||||
from litellm.types.router import LiteLLM_Params
|
||||
from litellm.types.utils import BudgetConfig
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def disable_budget_sync(monkeypatch):
|
||||
async def noop(*args, **kwargs):
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.router_strategy.budget_limiter.RouterBudgetLimiting.periodic_sync_in_memory_spend_with_redis",
|
||||
noop,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_llm_provider_for_deployment_dict_does_not_require_litellm_params_instantiation(
|
||||
disable_budget_sync, monkeypatch
|
||||
):
|
||||
class RaiseOnInit:
|
||||
def __init__(self, *args, **kwargs):
|
||||
raise AssertionError("LiteLLM_Params should not be instantiated in hot path")
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.router_strategy.budget_limiter.LiteLLM_Params",
|
||||
RaiseOnInit,
|
||||
)
|
||||
|
||||
provider_budget = RouterBudgetLimiting(
|
||||
dual_cache=DualCache(),
|
||||
provider_budget_config={},
|
||||
)
|
||||
|
||||
deployment = {"litellm_params": {"model": "openai/gpt-4o-mini"}}
|
||||
provider = provider_budget._get_llm_provider_for_deployment(deployment)
|
||||
|
||||
assert provider == "openai"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_llm_provider_for_deployment_dict_view_supports_mapping_and_attr_access(
|
||||
disable_budget_sync, monkeypatch
|
||||
):
|
||||
observed = {}
|
||||
|
||||
def _future_style_get_llm_provider(
|
||||
model,
|
||||
custom_llm_provider=None,
|
||||
api_base=None,
|
||||
api_key=None,
|
||||
litellm_params=None,
|
||||
):
|
||||
assert litellm_params is not None
|
||||
observed["model_attr"] = litellm_params.model
|
||||
observed["provider_get"] = litellm_params.get("custom_llm_provider")
|
||||
observed["api_base_item"] = litellm_params["api_base"]
|
||||
observed["has_api_key"] = "api_key" in litellm_params
|
||||
observed["model_dump"] = litellm_params.model_dump()
|
||||
return model, "openai", None, None
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.router_strategy.budget_limiter.litellm.get_llm_provider",
|
||||
_future_style_get_llm_provider,
|
||||
)
|
||||
|
||||
provider_budget = RouterBudgetLimiting(
|
||||
dual_cache=DualCache(),
|
||||
provider_budget_config={},
|
||||
)
|
||||
|
||||
deployment = {
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"custom_llm_provider": "openai",
|
||||
"api_base": "https://api.openai.com/v1",
|
||||
}
|
||||
}
|
||||
provider = provider_budget._get_llm_provider_for_deployment(deployment)
|
||||
|
||||
assert provider == "openai"
|
||||
assert observed["model_attr"] == "openai/gpt-4o-mini"
|
||||
assert observed["provider_get"] == "openai"
|
||||
assert observed["api_base_item"] == "https://api.openai.com/v1"
|
||||
assert observed["has_api_key"] is False
|
||||
assert observed["model_dump"]["model"] == "openai/gpt-4o-mini"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_filter_deployments_resolves_provider_once_per_deployment(
|
||||
disable_budget_sync, monkeypatch
|
||||
):
|
||||
provider_budget = RouterBudgetLimiting(
|
||||
dual_cache=DualCache(),
|
||||
provider_budget_config={
|
||||
"openai": BudgetConfig(budget_duration="1d", max_budget=100.0),
|
||||
},
|
||||
)
|
||||
|
||||
healthy_deployments = [
|
||||
{
|
||||
"model_name": "gpt-4o-mini",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini"},
|
||||
"model_info": {"id": "deployment-1"},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-4o-mini",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini"},
|
||||
"model_info": {"id": "deployment-2"},
|
||||
},
|
||||
]
|
||||
|
||||
provider_resolution_calls = 0
|
||||
|
||||
def _count_provider_calls(deployment):
|
||||
nonlocal provider_resolution_calls
|
||||
provider_resolution_calls += 1
|
||||
return "openai"
|
||||
|
||||
monkeypatch.setattr(
|
||||
provider_budget,
|
||||
"_get_llm_provider_for_deployment",
|
||||
_count_provider_calls,
|
||||
)
|
||||
|
||||
filtered_deployments = await provider_budget.async_filter_deployments(
|
||||
model="gpt-4o-mini",
|
||||
healthy_deployments=healthy_deployments,
|
||||
messages=[],
|
||||
request_kwargs={},
|
||||
parent_otel_span=None,
|
||||
)
|
||||
|
||||
assert len(filtered_deployments) == len(healthy_deployments)
|
||||
assert provider_resolution_calls == len(healthy_deployments)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_filter_deployments_does_not_recompute_provider_when_resolved_none(
|
||||
disable_budget_sync, monkeypatch
|
||||
):
|
||||
provider_budget = RouterBudgetLimiting(
|
||||
dual_cache=DualCache(),
|
||||
provider_budget_config={
|
||||
"openai": BudgetConfig(budget_duration="1d", max_budget=100.0),
|
||||
},
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-4o-mini",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"max_budget": 100.0,
|
||||
"budget_duration": "1d",
|
||||
},
|
||||
"model_info": {"id": "deployment-1"},
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
healthy_deployments = [
|
||||
{
|
||||
"model_name": "gpt-4o-mini",
|
||||
"litellm_params": {"model": "unknown-provider/model"},
|
||||
"model_info": {"id": "deployment-1"},
|
||||
}
|
||||
]
|
||||
|
||||
provider_resolution_calls = 0
|
||||
|
||||
def _provider_returns_none(deployment):
|
||||
nonlocal provider_resolution_calls
|
||||
provider_resolution_calls += 1
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(
|
||||
provider_budget,
|
||||
"_get_llm_provider_for_deployment",
|
||||
_provider_returns_none,
|
||||
)
|
||||
|
||||
filtered_deployments = await provider_budget.async_filter_deployments(
|
||||
model="gpt-4o-mini",
|
||||
healthy_deployments=healthy_deployments,
|
||||
messages=[],
|
||||
request_kwargs={},
|
||||
parent_otel_span=None,
|
||||
)
|
||||
|
||||
assert len(filtered_deployments) == len(healthy_deployments)
|
||||
assert provider_resolution_calls == len(healthy_deployments)
|
||||
|
||||
|
||||
def _legacy_provider_resolution(deployment):
|
||||
"""
|
||||
Reference implementation used before hot-path optimization.
|
||||
"""
|
||||
try:
|
||||
_litellm_params = LiteLLM_Params(**deployment.get("litellm_params", {"model": ""}))
|
||||
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
model=_litellm_params.model,
|
||||
litellm_params=_litellm_params,
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
return custom_llm_provider
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"deployment",
|
||||
[
|
||||
{"litellm_params": {"model": "openai/gpt-4o-mini"}},
|
||||
{"litellm_params": {"model": "gpt-4o-mini", "custom_llm_provider": "openai"}},
|
||||
{"litellm_params": {"model": "unknown-provider/model"}},
|
||||
],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_llm_provider_for_deployment_matches_legacy_behavior(
|
||||
disable_budget_sync, deployment
|
||||
):
|
||||
provider_budget = RouterBudgetLimiting(
|
||||
dual_cache=DualCache(),
|
||||
provider_budget_config={},
|
||||
)
|
||||
|
||||
current_provider = provider_budget._get_llm_provider_for_deployment(deployment)
|
||||
legacy_provider = _legacy_provider_resolution(deployment)
|
||||
|
||||
assert current_provider == legacy_provider
|
||||
5
ui/litellm-dashboard/public/assets/logos/zscaler.svg
Normal file
5
ui/litellm-dashboard/public/assets/logos/zscaler.svg
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
<svg width="50" height="41" viewBox="0 0 35 22" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M34.9305 11.1952C35.4985 14.7139 32.4864 16.6962 29.4313 16.9252C27.4867 20.976 20.6324 23.7107 14.6971 20.762C12.1552 21.1535 10.764 20.5092 9.6896 19.2752C11.861 16.4425 18.4366 11.3459 26.47 13.9755C30.7569 15.3849 31.8715 12.0843 30.7961 10.8513C26.7505 6.19284 17.6134 10.3865 17.2879 10.7353C20.8729 5.16699 33.7065 3.60491 34.9305 11.1952ZM22.6298 4.80521C22.6522 4.79728 19.7125 3.74467 15.6034 5.54462C15.4367 5.4832 15.2735 5.41239 15.1146 5.33251C19.0701 2.66529 22.5359 1.53437 25.5001 1.97445C23.7042 -0.11988 15.0833 -1.65817 10.1432 3.36208C4.03683 2.14096 -0.233535 7.38224 0.00989967 12.1765C0.253334 16.9708 5.41043 19.9502 8.34533 19.1424C8.41578 19.1335 8.48704 19.1335 8.55748 19.1424C9.21348 15.9766 11.8658 8.61029 22.6298 4.80521Z" fill="#2160E1"/>
|
||||
</svg>
|
||||
|
||||
<!--65 41, 50 32-->
|
||||
|
After Width: | Height: | Size: 906 B |
|
|
@ -104,6 +104,7 @@ export const shouldRenderContentFilterConfigSettings = (provider: string | null)
|
|||
const asset_logos_folder = "../ui/assets/logos/";
|
||||
|
||||
export const guardrailLogoMap: Record<string, string> = {
|
||||
"Zscaler AI Guard": `${asset_logos_folder}zscaler.svg`,
|
||||
"Presidio PII": `${asset_logos_folder}presidio.png`,
|
||||
"Bedrock Guardrail": `${asset_logos_folder}bedrock.svg`,
|
||||
Lakera: `${asset_logos_folder}lakeraai.jpeg`,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue