Merge branch 'BerriAI:main' into LangfuseUsageDetails

This commit is contained in:
Fabrício Ceschin 2025-09-11 10:15:22 -04:00 • committed by GitHub
commit 5cb5268e43
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
128 changed files with 6375 additions and 6644 deletions

View file

@ -535,11 +535,9 @@ jobs:
- litellm_router_coverage.xml
- litellm_router_coverage
litellm_security_tests:
docker:
- image: cimg/python:3.11
auth:
username: ${DOCKERHUB_USERNAME}
password: ${DOCKERHUB_PASSWORD}
machine:
image: ubuntu-2204:2023.10.1
resource_class: xlarge
working_directory: ~/project
steps:
- checkout
@ -548,32 +546,67 @@ jobs:
name: Show git commit hash
command: |
echo "Git commit hash: $CIRCLE_SHA1"
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
sudo apt-get update
sudo apt-get install -y docker-ce docker-ce-cli containerd.io
- run:
name: Install Python 3.9
command: |
curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh
bash miniconda.sh -b -p $HOME/miniconda
export PATH="$HOME/miniconda/bin:$PATH"
conda init bash
source ~/.bashrc
conda create -n myenv python=3.9 -y
conda activate myenv
python --version
- run:
name: Install Dependencies
command: |
pip install "pytest==7.3.1"
pip install "pytest-asyncio==0.21.1"
pip install aiohttp
python -m pip install --upgrade pip
python -m pip install -r requirements.txt
pip install "pytest==7.3.1"
pip install "pytest-retry==1.6.3"
pip install "pytest-mock==3.12.0"
pip install "pytest-asyncio==0.21.1"
pip install mypy
pip install "google-generativeai==0.3.2"
pip install "google-cloud-aiplatform==1.43.0"
pip install pyarrow
pip install "boto3==1.36.0"
pip install "aioboto3==13.4.0"
pip install langchain
pip install "langfuse>=2.0.0"
pip install "logfire==0.29.0"
pip install numpydoc
pip install prisma
pip install fastapi
pip install jsonschema
pip install "httpx==0.24.1"
pip install "gunicorn==21.2.0"
pip install "anyio==3.7.1"
pip install "aiodynamo==23.10.1"
pip install "asyncio==3.4.3"
pip install "PyGithub==1.59.1"
pip install "openai==1.100.1"
pip install "pytest-cov==5.0.0"
pip install "apscheduler"
- run:
name: Install Trivy
name: Install dockerize
command: |
sudo apt-get update
sudo apt-get install wget apt-transport-https gnupg lsb-release
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
sudo apt-get update
sudo apt-get install trivy
wget https://github.com/jwilder/dockerize/releases/download/v0.6.1/dockerize-linux-amd64-v0.6.1.tar.gz
sudo tar -C /usr/local/bin -xzvf dockerize-linux-amd64-v0.6.1.tar.gz
rm dockerize-linux-amd64-v0.6.1.tar.gz
- run:
name: Run Trivy scan on LiteLLM Docs
name: Run Security Scans
command: |
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
- run:
name: Run Trivy scan on LiteLLM UI
command: |
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
chmod +x ci_cd/security_scans.sh
./ci_cd/security_scans.sh
- run:
name: Run prisma ./docker/entrypoint.sh
command: |
@ -1424,6 +1457,7 @@ jobs:
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
- run: python ./tests/code_coverage_tests/check_licenses.py
- run: python ./tests/code_coverage_tests/router_code_coverage.py
- run: python ./tests/code_coverage_tests/info_log_check.py
- run: python ./tests/code_coverage_tests/test_ban_set_verbose.py
- run: python ./tests/code_coverage_tests/code_qa_check_tests.py
- run: python ./tests/code_coverage_tests/test_proxy_types_import.py
@ -1593,23 +1627,6 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
- run:
name: Install Grype
command: |
curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin
- run:
name: Build and Scan Docker Images
command: |
# Build and scan Dockerfile.database
echo "Building and scanning Dockerfile.database..."
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
grype litellm-database:latest --fail-on critical
# Build and scan main Dockerfile
echo "Building and scanning main Dockerfile..."
docker build -t litellm:latest .
grype litellm:latest --fail-on critical
- run:
name: Build Docker image
command: docker build -t my-app:latest -f ./docker/Dockerfile.database .

View file

@ -25,7 +25,7 @@
<a href="https://discord.gg/wuPM9dRgDw">
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Discord&color=blue&logo=Discord&style=flat-square" alt="Discord">
</a>
<a href="https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3">
<a href="https://www.litellm.ai/support">
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Slack&color=black&logo=Slack&style=flat-square" alt="Slack">
</a>
</h4>
@ -408,7 +408,7 @@ All these checks must pass before your PR can be merged.
- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
- [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
- [Community Slack 💭](https://www.litellm.ai/support)
- Our numbers 📞 +1 (770) 8783-106 / ‭+1 (412) 618-6238‬
- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai

Binary file not shown.

105
ci_cd/security_scans.sh Executable file
View file

@ -0,0 +1,105 @@
#!/bin/bash
# Security Scans Script for LiteLLM
# This script runs comprehensive security scans including Trivy and Grype
set -e
echo "Starting security scans for LiteLLM..."
# Function to install Trivy and required tools
install_trivy() {
echo "Installing Trivy and required tools..."
sudo apt-get update
sudo apt-get install -y wget apt-transport-https gnupg lsb-release jq curl
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
sudo apt-get update
sudo apt-get install trivy
echo "Trivy and required tools installed successfully"
}
# Function to install Grype
install_grype() {
echo "Installing Grype..."
curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin
echo "Grype installed successfully"
}
# Function to run Trivy scans
run_trivy_scans() {
echo "Running Trivy scans..."
echo "Scanning LiteLLM Docs..."
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
echo "Scanning LiteLLM UI..."
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
echo "Trivy scans completed successfully"
}
# Function to build and scan Docker images with Grype
run_grype_scans() {
echo "Running Grype scans..."
# Temporarily add wheel files to .dockerignore for security scans
echo "Temporarily modifying .dockerignore to exclude problematic wheel files..."
cp .dockerignore .dockerignore.backup 2>/dev/null || touch .dockerignore.backup
echo "/*.whl" >> .dockerignore
# Build and scan Dockerfile.database
echo "Building and scanning Dockerfile.database..."
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
grype litellm-database:latest --fail-on critical
# Build and scan main Dockerfile
echo "Building and scanning main Dockerfile..."
docker build -t litellm:latest .
grype litellm:latest --fail-on critical
# Restore original .dockerignore
echo "Restoring original .dockerignore..."
mv .dockerignore.backup .dockerignore
# Scan the locally built LiteLLM image for vulnerabilities with CVSS >= 4.0
echo "Scanning locally built LiteLLM image for high-severity vulnerabilities..."
echo "Using locally built image: litellm:latest"
# Run grype scan and check for vulnerabilities with CVSS >= 4.0
echo "Checking for vulnerabilities with CVSS score >= 4.0..."
HIGH_SEVERITY_COUNT=$(grype litellm:latest -o json | jq -r '.matches[] | select(.vulnerability.cvss[]?.metrics.baseScore >= 4.0) | .vulnerability.id' | wc -l)
if [ "$HIGH_SEVERITY_COUNT" -gt 0 ]; then
echo "ERROR: Found $HIGH_SEVERITY_COUNT vulnerabilities with CVSS score >= 4.0 in litellm:latest"
echo "Detailed vulnerability report:"
grype litellm:latest -o json | jq -r '
["Package", "Version", "Vulnerability ID", "CVSS Score", "Severity", "Fix Version", "Description"],
(.matches[] | select(.vulnerability.cvss[]?.metrics.baseScore >= 4.0) |
[.artifact.name, .artifact.version, .vulnerability.id, .vulnerability.cvss[0].metrics.baseScore, .vulnerability.severity, (.vulnerability.fix.versions[0] // "No fix available"), .vulnerability.description]) |
@tsv' | column -t -s $'\t'
exit 1
else
echo "No high-severity vulnerabilities (CVSS >= 4.0) found in litellm:latest"
fi
echo "Grype scans completed successfully"
}
# Main execution
main() {
echo "Installing security scanning tools..."
install_trivy
install_grype
echo "Running filesystem vulnerability scans..."
run_trivy_scans
echo "Running Docker image vulnerability scans..."
run_grype_scans
echo "All security scans completed successfully!"
}
# Execute main function
main "$@"

View file

@ -0,0 +1,9 @@
# Security Scans
## Scans that run:
- Trivy scan on `./docs/` (HIGH/CRITICAL/MEDIUM)
- Trivy scan on `./ui/` (HIGH/CRITICAL/MEDIUM)
- Grype scan on `Dockerfile.database` (fails on CRITICAL)
- Grype scan on main `Dockerfile` (fails on CRITICAL)
- Grype CVSS ≥ 4.0 scan on main `Dockerfile` (fails any vulnerabilities with CVSS ≥ 4.0)

View file

@ -0,0 +1,36 @@
"""
Use LiteLLM Proxy MCP Gateway to call MCP tools.
When using LiteLLM Proxy, you can use the same MCP tools across all your LLM providers.
"""
import openai
client = openai.OpenAI(
api_key="sk-1234", # paste your litellm proxy api key here
base_url="http://localhost:4000" # paste your litellm proxy base url here
)
print("Making API request to Responses API with MCP tools")
response = client.responses.create(
model="gpt-5",
input=[
{
"role": "user",
"content": "give me TLDR of what BerriAI/litellm repo is about",
"type": "message"
}
],
tools=[
{
"type": "mcp",
"server_label": "litellm",
"server_url": "litellm_proxy",
"require_approval": "never"
}
],
stream=True,
tool_choice="required"
)
for chunk in response:
print("response chunk: ", chunk)

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View file

@ -0,0 +1,145 @@
# Custom HTTP Handler
Configure custom aiohttp sessions for better performance and control in LiteLLM completions.
## Overview
You can now inject custom `aiohttp.ClientSession` instances into LiteLLM for:
- Custom connection pooling and timeouts
- Corporate proxy and SSL configurations
- Performance optimization
- Request monitoring
## Basic Usage
### Default (No Changes Required)
```python
import litellm
# Works exactly as before
response = await litellm.acompletion(
model="gpt-3.5-turbo",
messages=[{"role": "user", "content": "Hello!"}]
)
```
### Custom Session
```python
import aiohttp
import litellm
from litellm.llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler
# Create optimized session
session = aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=180),
connector=aiohttp.TCPConnector(limit=300, limit_per_host=75)
)
# Replace global handler
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
# All completions now use your session
response = await litellm.acompletion(model="gpt-3.5-turbo", messages=[...])
```
## Common Patterns
### FastAPI Integration
```python
from contextlib import asynccontextmanager
from fastapi import FastAPI
import aiohttp
import litellm
@asynccontextmanager
async def lifespan(app: FastAPI):
# Startup
session = aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=180),
connector=aiohttp.TCPConnector(limit=300)
)
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(
client_session=session
)
yield
# Shutdown
await session.close()
app = FastAPI(lifespan=lifespan)
@app.post("/chat")
async def chat(messages: list[dict]):
return await litellm.acompletion(model="gpt-3.5-turbo", messages=messages)
```
### Corporate Proxy
```python
import ssl
# Custom SSL context
ssl_context = ssl.create_default_context()
ssl_context.load_cert_chain('cert.pem', 'key.pem')
# Proxy session
session = aiohttp.ClientSession(
connector=aiohttp.TCPConnector(ssl=ssl_context),
trust_env=True # Use environment proxy settings
)
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
```
### High Performance
```python
# Optimized for high throughput
session = aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=300),
connector=aiohttp.TCPConnector(
limit=1000, # High connection limit
limit_per_host=200, # Per host limit
ttl_dns_cache=600, # DNS cache
keepalive_timeout=60, # Keep connections alive
enable_cleanup_closed=True
)
)
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
```
## Constructor Options
```python
BaseLLMAIOHTTPHandler(
client_session=None, # Custom aiohttp.ClientSession
transport=None, # Advanced transport control
connector=None, # Custom aiohttp.BaseConnector
)
```
## Resource Management
- **User sessions**: You manage the lifecycle (call `await session.close()`)
- **Auto-created sessions**: Automatically cleaned up by the handler
- **100% backward compatible**: Existing code works unchanged
## Configuration Tips
### Development
```python
session = aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=60),
connector=aiohttp.TCPConnector(limit=50)
)
```
### Production
```python
session = aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=300),
connector=aiohttp.TCPConnector(
limit=1000,
limit_per_host=200,
keepalive_timeout=60
)
)
```

View file

@ -251,7 +251,7 @@ response = completion(
{
"id": "chatcmpl-565d891b-a42e-4c39-8d14-82a1f5208885",
"created": 1734366691,
"model": "claude-3-sonnet-20240229",
"model": "gpt-4o-2024-08-06",
"object": "chat.completion",
"system_fingerprint": null,
"choices": [

View file

@ -162,3 +162,321 @@ Get more details [here](../observability/lunary_integration.md)
## Use LangChain ChatLiteLLM + Langfuse
Checkout this section [here](../observability/langfuse_integration#use-langchain-chatlitellm--langfuse) for more details on how to integrate Langfuse with ChatLiteLLM.
## Using Tags with LangChain and LiteLLM
Tags are a powerful feature in LiteLLM that allow you to categorize, filter, and track your LLM requests. When using LangChain with LiteLLM, you can pass tags through the `extra_body` parameter in the metadata.
### Basic Tag Usage
<Tabs>
<TabItem value="openai" label="OpenAI">
```python
import os
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage, SystemMessage
os.environ['OPENAI_API_KEY'] = "sk-your-key-here"
chat = ChatOpenAI(
model="gpt-4o",
temperature=0.7,
extra_body={
"metadata": {
"tags": ["production", "customer-support", "high-priority"]
}
}
)
messages = [
SystemMessage(content="You are a helpful customer support assistant."),
HumanMessage(content="How do I reset my password?")
]
response = chat.invoke(messages)
print(response)
```
</TabItem>
<TabItem value="anthropic" label="Anthropic">
```python
import os
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage, SystemMessage
os.environ['ANTHROPIC_API_KEY'] = "sk-ant-your-key-here"
chat = ChatOpenAI(
model="claude-3-sonnet-20240229",
temperature=0.7,
extra_body={
"metadata": {
"tags": ["research", "analysis", "claude-model"]
}
}
)
messages = [
SystemMessage(content="You are a research analyst."),
HumanMessage(content="Analyze this market trend...")
]
response = chat.invoke(messages)
print(response)
```
</TabItem>
<TabItem value="litellm-proxy" label="LiteLLM Proxy">
```python
import os
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage, SystemMessage
# No API key needed when using proxy
chat = ChatOpenAI(
openai_api_base="http://localhost:4000", # Your proxy URL
model="gpt-4o",
temperature=0.7,
extra_body={
"metadata": {
"tags": ["proxy", "team-alpha", "feature-flagged"],
"generation_name": "customer-onboarding",
"trace_user_id": "user-12345"
}
}
)
messages = [
SystemMessage(content="You are an onboarding assistant."),
HumanMessage(content="Welcome our new customer!")
]
response = chat.invoke(messages)
print(response)
```
</TabItem>
</Tabs>
### Advanced Tag Patterns
#### Dynamic Tags Based on Context
```python
import os
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage, SystemMessage
def create_chat_with_tags(user_type: str, feature: str):
"""Create a chat instance with dynamic tags based on context"""
# Build tags dynamically
tags = ["langchain-integration"]
if user_type == "premium":
tags.extend(["premium-user", "high-priority"])
elif user_type == "enterprise":
tags.extend(["enterprise", "custom-sla"])
else:
tags.append("standard-user")
# Add feature-specific tags
if feature == "code-review":
tags.extend(["development", "code-analysis"])
elif feature == "content-gen":
tags.extend(["marketing", "content-creation"])
return ChatOpenAI(
openai_api_base="http://localhost:4000",
model="gpt-4o",
temperature=0.7,
extra_body={
"metadata": {
"tags": tags,
"user_type": user_type,
"feature": feature,
"trace_user_id": f"user-{user_type}-{feature}"
}
}
)
# Usage examples
premium_chat = create_chat_with_tags("premium", "code-review")
enterprise_chat = create_chat_with_tags("enterprise", "content-gen")
messages = [HumanMessage(content="Help me with this task")]
response = premium_chat.invoke(messages)
```
#### Tags for Cost Tracking and Analytics
```python
import os
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage, SystemMessage
# Tags for cost tracking
cost_tracking_chat = ChatOpenAI(
openai_api_base="http://localhost:4000",
model="gpt-4o",
temperature=0.7,
extra_body={
"metadata": {
"tags": [
"cost-center-marketing",
"budget-q4-2024",
"project-launch-campaign",
"high-cost-model" # Flag for expensive models
],
"department": "marketing",
"project_id": "campaign-2024-q4",
"cost_threshold": "high"
}
}
)
messages = [
SystemMessage(content="You are a marketing copywriter."),
HumanMessage(content="Create compelling ad copy for our new product launch.")
]
response = cost_tracking_chat.invoke(messages)
```
#### Tags for A/B Testing
```python
import os
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage, SystemMessage
import random
def create_ab_test_chat(test_variant: str = None):
"""Create chat instance for A/B testing with appropriate tags"""
if test_variant is None:
test_variant = random.choice(["variant-a", "variant-b"])
return ChatOpenAI(
openai_api_base="http://localhost:4000",
model="gpt-4o",
temperature=0.7 if test_variant == "variant-a" else 0.9, # Different temp for variants
extra_body={
"metadata": {
"tags": [
"ab-test-experiment-1",
f"variant-{test_variant}",
"temperature-test",
"user-experience"
],
"experiment_id": "ab-test-001",
"variant": test_variant,
"test_group": "temperature-optimization"
}
}
)
# Run A/B test
variant_a_chat = create_ab_test_chat("variant-a")
variant_b_chat = create_ab_test_chat("variant-b")
test_message = [HumanMessage(content="Explain quantum computing in simple terms")]
response_a = variant_a_chat.invoke(test_message)
response_b = variant_b_chat.invoke(test_message)
```
### Tag Best Practices
#### 1. **Consistent Naming Convention**
```python
# ✅ Good: Consistent, descriptive tags
tags = ["production", "api-v2", "customer-support", "urgent"]
# ❌ Avoid: Inconsistent or unclear tags
tags = ["prod", "v2", "support", "urgent123"]
```
#### 2. **Hierarchical Tags**
```python
# ✅ Good: Hierarchical structure
tags = ["env:production", "team:backend", "service:api", "priority:high"]
# This allows for easy filtering and grouping
```
#### 3. **Include Context Information**
```python
extra_body={
"metadata": {
"tags": ["production", "user-onboarding"],
"user_id": "user-12345",
"session_id": "session-abc123",
"feature_flag": "new-onboarding-flow",
"environment": "production"
}
}
```
#### 4. **Tag Categories**
Consider organizing tags into categories:
- **Environment**: `production`, `staging`, `development`
- **Team/Service**: `backend`, `frontend`, `api`, `worker`
- **Feature**: `authentication`, `payment`, `notification`
- **Priority**: `critical`, `high`, `medium`, `low`
- **User Type**: `premium`, `enterprise`, `free`
### Using Tags with LiteLLM Proxy
When using tags with LiteLLM Proxy, you can:
1. **Filter requests** based on tags
2. **Track costs** by tags in spend reports
3. **Apply routing rules** based on tags
4. **Monitor usage** with tag-based analytics
#### Example Proxy Configuration with Tags
```yaml
# config.yaml
model_list:
- model_name: gpt-4o
litellm_params:
model: gpt-4o
api_key: your-key
# Tag-based routing rules
tag_routing:
- tags: ["premium", "high-priority"]
models: ["gpt-4o", "claude-3-opus"]
- tags: ["standard"]
models: ["gpt-3.5-turbo", "claude-3-haiku"]
```
### Monitoring and Analytics
Tags enable powerful analytics capabilities:
```python
# Example: Get spend reports by tags
import requests
response = requests.get(
"http://localhost:4000/global/spend/report",
headers={"Authorization": "Bearer sk-your-key"},
params={
"start_date": "2024-01-01",
"end_date": "2024-12-31",
"group_by": "tags"
}
)
spend_by_tags = response.json()
```
This documentation covers the essential patterns for using tags effectively with LangChain and LiteLLM, enabling better organization, tracking, and analytics of your LLM requests.

View file

@ -113,6 +113,7 @@ mcp_servers:
transport: "http"
description: "My custom MCP server"
auth_type: "api_key"
auth_value: "abc123"
spec_version: "2025-03-26"
```
@ -128,8 +129,42 @@ mcp_servers:
- **Args**: Array of arguments to pass to the command (optional for stdio)
- **Env**: Environment variables to set for the stdio process (optional for stdio)
- **Description**: Optional description for the server
- **Auth Type**: Optional authentication type
- **Spec Version**: Optional MCP specification version (defaults to `2025-03-26`)
- **Auth Type**: Optional authentication type. Supported values:
| Value | Header sent |
|-------|-------------|
| `api_key` | `X-API-Key: <auth_value>` |
| `bearer_token` | `Authorization: Bearer <auth_value>` |
| `basic` | `Authorization: Basic <auth_value>` |
| `authorization` | `Authorization: <auth_value>` |
- **Spec Version**: Optional MCP specification version (defaults to `2025-06-18`)
Examples for each auth type:
```yaml title="MCP auth examples (config.yaml)" showLineNumbers
mcp_servers:
api_key_example:
url: "https://my-mcp-server.com/mcp"
auth_type: "api_key"
auth_value: "abc123" # headers={"X-API-Key": "abc123"}
bearer_example:
url: "https://my-mcp-server.com/mcp"
auth_type: "bearer_token"
auth_value: "abc123" # headers={"Authorization": "Bearer abc123"}
basic_example:
url: "https://my-mcp-server.com/mcp"
auth_type: "basic"
auth_value: "dXNlcjpwYXNz" # headers={"Authorization": "Basic dXNlcjpwYXNz"}
custom_auth_example:
url: "https://my-mcp-server.com/mcp"
auth_type: "authorization"
auth_value: "Token example123" # headers={"Authorization": "Token example123"}
```
### MCP Aliases
@ -160,70 +195,169 @@ litellm_settings:
## Using your MCP
### Use on LiteLLM UI
Follow this walkthrough to use your MCP on LiteLLM UI
<iframe width="840" height="500" src="https://www.loom.com/embed/57e0763267254bc79dbe6658d0b8758c" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
### Use with Responses API
Replace `http://localhost:4000` with your LiteLLM Proxy base URL.
Demo Video Using Responses API with LiteLLM Proxy: [Demo video here](https://www.loom.com/share/34587e618c5c47c0b0d67b4e4d02718f?sid=2caf3d45-ead4-4490-bcc1-8d6dd6041c02)
<Tabs>
<TabItem value="openai" label="OpenAI API">
#### Connect via OpenAI Responses API
Use the OpenAI Responses API to connect to your LiteLLM MCP server:
<TabItem value="curl" label="cURL">
```bash title="cURL Example" showLineNumbers
curl --location 'https://api.openai.com/v1/responses' \
curl --location 'http://localhost:4000/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer $OPENAI_API_KEY" \
--header "Authorization: Bearer sk-1234" \
--data '{
"model": "gpt-4o",
"model": "gpt-5",
"input": [
{
"role": "user",
"content": "give me TLDR of what BerriAI/litellm repo is about",
"type": "message"
}
],
"tools": [
{
"type": "mcp",
"server_label": "litellm",
"server_url": "litellm_proxy",
"require_approval": "never",
"headers": {
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
}
"require_approval": "never"
}
],
"input": "Run available tools",
"stream": true,
"tool_choice": "required"
}'
```
</TabItem>
<TabItem value="python" label="Python SDK">
<TabItem value="litellm" label="LiteLLM Proxy">
```python title="Python SDK Example" showLineNumbers
"""
Use LiteLLM Proxy MCP Gateway to call MCP tools.
#### Connect via LiteLLM Proxy Responses API
When using LiteLLM Proxy, you can use the same MCP tools across all your LLM providers.
"""
import openai
Use this when calling LiteLLM Proxy for LLM API requests to `/v1/responses` endpoint.
client = openai.OpenAI(
api_key="sk-1234", # paste your litellm proxy api key here
base_url="http://localhost:4000" # paste your litellm proxy base url here
)
print("Making API request to Responses API with MCP tools")
```bash title="cURL Example" showLineNumbers
curl --location '<your-litellm-proxy-base-url>/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer $LITELLM_API_KEY" \
--data '{
"model": "gpt-4o",
"tools": [
response = client.responses.create(
model="gpt-5",
input=[
{
"role": "user",
"content": "give me TLDR of what BerriAI/litellm repo is about",
"type": "message"
}
],
tools=[
{
"type": "mcp",
"server_label": "litellm",
"server_url": "litellm_proxy",
"require_approval": "never",
"headers": {
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
}
"require_approval": "never"
}
],
"input": "Run available tools",
stream=True,
tool_choice="required"
)
for chunk in response:
print("response chunk: ", chunk)
```
</TabItem>
</Tabs>
#### Specifying MCP Tools
You can specify which MCP tools are available by using the `allowed_tools` parameter. This allows you to restrict access to specific tools within an MCP server.
To get the list of allowed tools when using LiteLLM MCP Gateway, you can naigate to the LiteLLM UI on MCP Servers > MCP Tools > Click the Tool > Copy Tool Name.
<Tabs>
<TabItem value="curl" label="cURL">
```bash title="cURL Example with allowed_tools" showLineNumbers
curl --location 'http://localhost:4000/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer sk-1234" \
--data '{
"model": "gpt-5",
"input": [
{
"role": "user",
"content": "give me TLDR of what BerriAI/litellm repo is about",
"type": "message"
}
],
"tools": [
{
"type": "mcp",
"server_label": "litellm",
"server_url": "litellm_proxy/mcp",
"require_approval": "never",
"allowed_tools": ["GitMCP-fetch_litellm_documentation"]
}
],
"stream": true,
"tool_choice": "required"
}'
```
</TabItem>
<TabItem value="python" label="Python SDK">
<TabItem value="cursor" label="Cursor IDE">
```python title="Python SDK Example with allowed_tools" showLineNumbers
import openai
#### Connect via Cursor IDE
client = openai.OpenAI(
api_key="sk-1234",
base_url="http://localhost:4000"
)
response = client.responses.create(
model="gpt-5",
input=[
{
"role": "user",
"content": "give me TLDR of what BerriAI/litellm repo is about",
"type": "message"
}
],
tools=[
{
"type": "mcp",
"server_label": "litellm",
"server_url": "litellm_proxy/mcp",
"require_approval": "never",
"allowed_tools": ["GitMCP-fetch_litellm_documentation"]
}
],
stream=True,
tool_choice="required"
)
print(response)
```
</TabItem>
</Tabs>
### Use with Cursor IDE
Use tools directly from Cursor IDE with LiteLLM MCP:
@ -246,9 +380,6 @@ Use tools directly from Cursor IDE with LiteLLM MCP:
}
```
</TabItem>
</Tabs>
#### How it works when server_url="litellm_proxy"
When server_url="litellm_proxy", LiteLLM bridges non-MCP providers to your MCP tools.

View file

@ -21,7 +21,7 @@ litellm_settings:
failure_callback: ["sentry"] # list of failure callbacks
callbacks: ["otel"] # list of callbacks - runs on success and failure
service_callbacks: ["datadog", "prometheus"] # logs redis, postgres failures on datadog, prometheus
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged.
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data.
redact_user_api_key_info: boolean # Redact information about the user api key (hashed token, user_id, team id, etc.), from logs. Currently supported for Langfuse, OpenTelemetry, Logfire, ArizeAI logging.
langfuse_default_tags: ["cache_hit", "cache_key", "proxy_base_url", "user_api_key_alias", "user_api_key_user_id", "user_api_key_user_email", "user_api_key_team_alias", "semantic-similarity", "proxy_base_url"] # default tags for Langfuse Logging
@ -131,7 +131,7 @@ general_settings:
| failure_callback | array of strings | List of failure callbacks [Doc Proxy logging callbacks](logging), [Doc Metrics](prometheus) |
| callbacks | array of strings | List of callbacks - runs on success and failure [Doc Proxy logging callbacks](logging), [Doc Metrics](prometheus) |
| service_callbacks | array of strings | System health monitoring - Logs redis, postgres failures on specified services (e.g. datadog, prometheus) [Doc Metrics](prometheus) |
| turn_off_message_logging | boolean | If true, prevents messages and responses from being logged to callbacks, but request metadata will still be logged [Proxy Logging](logging) |
| turn_off_message_logging | boolean | If true, prevents messages and responses from being logged to callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data [Proxy Logging](logging) |
| modify_params | boolean | If true, allows modifying the parameters of the request before it is sent to the LLM provider |
| enable_preview_features | boolean | If true, enables preview features - e.g. Azure O1 Models with streaming support.|
| redact_user_api_key_info | boolean | If true, redacts information about the user api key from logs [Proxy Logging](logging#redacting-userapikeyinfo) |
@ -523,6 +523,8 @@ router_settings:
| GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS
| GUARDRAILS_AI_API_BASE | Base URL for Guardrails AI API
| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60
| HEROKU_API_BASE | Base URL for Heroku API
| HEROKU_API_KEY | API key for Heroku services
| HF_API_BASE | Base URL for Hugging Face API
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
@ -566,6 +568,7 @@ router_settings:
| LASSO_USER_ID | User ID for Lasso service
| LASSO_CONVERSATION_ID | Conversation ID for Lasso service
| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16
| LEGACY_MULTI_INSTANCE_RATE_LIMITING | Flag to enable legacy multi-instance rate limiting. **Default is False**
| LITERAL_API_KEY | API key for Literal integration
| LITERAL_API_URL | API URL for Literal service
| LITERAL_BATCH_SIZE | Batch size for Literal operations

View file

@ -13,7 +13,7 @@ End-to-End tutorial for LiteLLM Proxy to:
## Pre-Requisites
- Install LiteLLM Docker Image ** OR ** LiteLLM CLI (pip package)
- Install LiteLLM Docker Image **OR** LiteLLM CLI (pip package)
<Tabs>
@ -278,15 +278,15 @@ See All General Settings [here](http://localhost:3000/docs/proxy/configs#all-set
- **Description**:
- Set a `master key`, this is your Proxy Admin key - you can use this to create other keys (🚨 must start with `sk-`).
- **Usage**:
- ** Set on config.yaml** set your master key under `general_settings:master_key`, example -
- **Set on config.yaml** set your master key under `general_settings:master_key`, example -
`master_key: sk-1234`
- ** Set env variable** set `LITELLM_MASTER_KEY`
- **Set env variable** set `LITELLM_MASTER_KEY`
2. **`database_url`** (str)
- **Description**:
- Set a `database_url`, this is the connection to your Postgres DB, which is used by litellm for generating keys, users, teams.
- **Usage**:
- ** Set on config.yaml** set your `database_url` under `general_settings:database_url`, example -
- **Set on config.yaml** set your `database_url` under `general_settings:database_url`, example -
`database_url: "postgresql://..."`
- Set `DATABASE_URL=postgresql://<user>:<password>@<host>:<port>/<dbname>` in your env

View file

@ -128,8 +128,11 @@ model_list:
api_key: "os.environ/OPENAI_API_KEY"
model_info:
mode: audio_speech
health_check_voice: alloy
```
You can specify a `health_check_voice` if you need to use a voice other than "alloy".
### Rerank Models
To run rerank health checks, specify the mode as "rerank" in your config for the relevant model.

View file

@ -60,7 +60,7 @@ components in your system, including in logging tools.
### Redact Messages, Response Content
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to your logging provider, but request metadata - e.g. spend, will still be tracked.
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to your logging provider, but request metadata - e.g. spend, will still be tracked. Useful for privacy/compliance when handling sensitive data.
<Tabs>

View file

@ -357,6 +357,106 @@ assert user.age == 25
</TabItem>
</Tabs>
## Using Tags for Categorization and Tracking
Tags allow you to categorize, filter, and track your LLM requests. Add tags to your metadata for better organization and analytics.
<Tabs>
<TabItem value="openai-python" label="OpenAI Python">
```python
import openai
client = openai.OpenAI(
api_key="anything",
base_url="http://0.0.0.0:4000"
)
response = client.chat.completions.create(
model="gpt-3.5-turbo",
messages=[{"role": "user", "content": "Hello!"}],
extra_body={
"metadata": {
"tags": ["production", "customer-support", "urgent"],
"generation_name": "support-bot",
"trace_user_id": "user-123"
}
}
)
```
</TabItem>
<TabItem value="langchain-python" label="LangChain Python">
```python
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000",
model="gpt-4o",
extra_body={
"metadata": {
"tags": ["langchain-integration", "content-gen"],
"trace_user_id": "user-456"
}
}
)
response = chat.invoke([HumanMessage(content="Generate a blog post")])
```
</TabItem>
<TabItem value="curl" label="Curl">
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data '{
"model": "gpt-3.5-turbo",
"messages": [{"role": "user", "content": "Hello!"}],
"metadata": {
"tags": ["api-test", "development"],
"trace_user_id": "test-user"
}
}'
```
</TabItem>
<TabItem value="openai-js" label="OpenAI JS">
```js
const { OpenAI } = require('openai');
const openai = new OpenAI({
apiKey: "sk-1234",
baseURL: "http://0.0.0.0:4000"
});
async function main() {
const response = await openai.chat.completions.create({
messages: [{ role: 'user', content: 'Hello!' }],
model: 'gpt-3.5-turbo',
metadata: {
tags: ["javascript-client", "api-test"],
trace_user_id: "js-user-789"
}
});
}
```
</TabItem>
</Tabs>
### Tag Benefits
- **Cost Tracking**: Monitor spending by project/team/feature
- **Analytics**: Filter requests by tags in logs and dashboards
- **Routing**: Use tags for conditional model routing
- **Debugging**: Easier troubleshooting with categorized requests
### Response Format
```json

View file

@ -72,6 +72,7 @@ On the LiteLLM UI, Navigate to `Teams`, You should see the new team `Production
<Image img={require('../../img/msft_auto_team.png')} style={{ width: '900px', height: 'auto' }} />
> **Note:** When a user is removed from your organization via SCIM, all API keys and access tokens associated with that user will be automatically deleted from LiteLLM. This ensures that removed users lose all access immediately and securely.

Binary file not shown.

After

Width:  |  Height:  |  Size: 216 KiB

View file

@ -19,6 +19,13 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
:::warning
This release has a known issue where startup is leading to Out of Memory errors when deploying on Kubernetes. We recommend waiting before upgrading to this version.
:::
## Deploy this version
<Tabs>

View file

@ -261,6 +261,7 @@ const sidebars = {
"completion/input",
"completion/output",
"completion/usage",
"completion/http_handler_config",
],
},
"response_api",

View file

@ -63,7 +63,7 @@ class _ENTERPRISE_LLMGuard(CustomLogger):
analyze_url, json=analyze_payload
) as response:
redacted_text = await response.json()
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"LLM Guard: Received response - {redacted_text}"
)
if redacted_text is not None:

Binary file not shown.

View file

@ -142,7 +142,10 @@ def create_gcp_iam_redis_connect_func(
"""
def iam_connect(self):
"""Initialize the connection and authenticate using GCP IAM"""
from redis.exceptions import AuthenticationError, AuthenticationWrongNumberOfArgsError
from redis.exceptions import (
AuthenticationError,
AuthenticationWrongNumberOfArgsError,
)
from redis.utils import str_if_bytes
self._parser.on_connect(self)
@ -395,7 +398,7 @@ def get_redis_async_client(
# Handle GCP IAM authentication for async clusters
redis_connect_func = cluster_kwargs.pop("redis_connect_func", None)
from litellm import get_secret_str
# Get GCP service account - first try from redis_connect_func, then from environment
gcp_service_account = None
if redis_connect_func and hasattr(redis_connect_func, '_gcp_service_account'):
@ -403,22 +406,22 @@ def get_redis_async_client(
else:
gcp_service_account = redis_kwargs.get("gcp_service_account") or get_secret_str("REDIS_GCP_SERVICE_ACCOUNT")
verbose_logger.info(f"DEBUG: Redis cluster kwargs: redis_connect_func={redis_connect_func is not None}, gcp_service_account_provided={gcp_service_account is not None}")
verbose_logger.debug(f"DEBUG: Redis cluster kwargs: redis_connect_func={redis_connect_func is not None}, gcp_service_account_provided={gcp_service_account is not None}")
# If GCP IAM is configured (indicated by redis_connect_func), generate access token and use as password
if redis_connect_func and gcp_service_account:
verbose_logger.info("DEBUG: Generating IAM token for service account (value not logged for security reasons)")
verbose_logger.debug("DEBUG: Generating IAM token for service account (value not logged for security reasons)")
try:
# Generate IAM access token using the helper function
access_token = _generate_gcp_iam_access_token(gcp_service_account)
cluster_kwargs["password"] = access_token
verbose_logger.info("DEBUG: Successfully generated GCP IAM access token for async Redis cluster")
verbose_logger.debug("DEBUG: Successfully generated GCP IAM access token for async Redis cluster")
except Exception as e:
verbose_logger.error(f"Failed to generate GCP IAM access token: {e}")
from redis.exceptions import AuthenticationError
raise AuthenticationError("Failed to generate GCP IAM access token")
else:
verbose_logger.info(f"DEBUG: Not using GCP IAM auth - redis_connect_func={redis_connect_func is not None}, gcp_service_account={gcp_service_account}")
verbose_logger.debug(f"DEBUG: Not using GCP IAM auth - redis_connect_func={redis_connect_func is not None}, gcp_service_account_provided={gcp_service_account is not None}")
new_startup_nodes: List[ClusterNode] = []

View file

@ -17,7 +17,6 @@ In each method it will call the appropriate method from caching.py
import asyncio
import datetime
import inspect
import threading
from typing import (
TYPE_CHECKING,
Any,
@ -301,10 +300,12 @@ class LLMCachingHandler:
is_async=False,
)
threading.Thread(
target=logging_obj.success_handler,
args=(cached_result, start_time, end_time, cache_hit),
).start()
logging_obj.handle_sync_success_callbacks_for_async_calls(
result=cached_result,
start_time=start_time,
end_time=end_time,
cache_hit=cache_hit
)
cache_key = litellm.cache._get_preset_cache_key_from_kwargs(
**kwargs
)
@ -530,15 +531,17 @@ class LLMCachingHandler:
end_time (datetime): The end time of the operation.
cache_hit (bool): Whether it was a cache hit.
"""
asyncio.create_task(
logging_obj.async_success_handler(
cached_result, start_time, end_time, cache_hit
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
GLOBAL_LOGGING_WORKER.ensure_initialized_and_enqueue(
async_coroutine=logging_obj.async_success_handler(
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
)
)
threading.Thread(
target=logging_obj.success_handler,
args=(cached_result, start_time, end_time, cache_hit),
).start()
logging_obj.handle_sync_success_callbacks_for_async_calls(
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
)
async def _retrieve_from_cache(
self, call_type: str, kwargs: Dict[str, Any], args: Tuple[Any, ...]

View file

@ -193,6 +193,8 @@ class MCPClient:
headers["Authorization"] = f"Basic {self._mcp_auth_value}"
elif self.auth_type == MCPAuth.api_key:
headers["X-API-Key"] = self._mcp_auth_value
elif self.auth_type == MCPAuth.authorization:
headers["Authorization"] = self._mcp_auth_value
# Handle protocol version - it might be a string or enum
if hasattr(self.protocol_version, 'value'):

View file

@ -17,22 +17,60 @@ from litellm.types.utils import ChatCompletionMessageToolCall
########################################################
def transform_mcp_tool_to_openai_tool(mcp_tool: MCPTool) -> ChatCompletionToolParam:
"""Convert an MCP tool to an OpenAI tool."""
normalized_parameters = _normalize_mcp_input_schema(mcp_tool.inputSchema)
return ChatCompletionToolParam(
type="function",
function=FunctionDefinition(
name=mcp_tool.name,
description=mcp_tool.description or "",
parameters=mcp_tool.inputSchema,
parameters=normalized_parameters,
strict=False,
),
)
def _normalize_mcp_input_schema(input_schema: dict) -> dict:
"""
Normalize MCP input schema to ensure it's valid for OpenAI function calling.
OpenAI requires that function parameters have:
- type: 'object'
- properties: dict (can be empty)
- additionalProperties: false (recommended)
"""
if not input_schema:
return {
"type": "object",
"properties": {},
"additionalProperties": False
}
# Make a copy to avoid modifying the original
normalized_schema = dict(input_schema)
# Ensure type is 'object'
if "type" not in normalized_schema:
normalized_schema["type"] = "object"
# Ensure properties exists (can be empty)
if "properties" not in normalized_schema:
normalized_schema["properties"] = {}
# Add additionalProperties if not present (recommended by OpenAI)
if "additionalProperties" not in normalized_schema:
normalized_schema["additionalProperties"] = False
return normalized_schema
def transform_mcp_tool_to_openai_responses_api_tool(mcp_tool: MCPTool) -> FunctionToolParam:
"""Convert an MCP tool to an OpenAI Responses API tool."""
normalized_parameters = _normalize_mcp_input_schema(mcp_tool.inputSchema)
return FunctionToolParam(
name=mcp_tool.name,
parameters=mcp_tool.inputSchema,
parameters=normalized_parameters,
strict=False,
type="function",
description=mcp_tool.description or "",

View file

@ -123,7 +123,7 @@ class CloudZeroLogger(CustomLogger):
)
if data.is_empty():
verbose_logger.info("CloudZero Logger: No usage data found to export")
verbose_logger.debug("CloudZero Logger: No usage data found to export")
return
verbose_logger.debug(f"CloudZero Logger: Processing {len(data)} records")
@ -146,7 +146,7 @@ class CloudZeroLogger(CustomLogger):
verbose_logger.debug(f"CloudZero Logger: Transmitting {len(cbf_data)} records to CloudZero")
streamer.send_batched(cbf_data, operation=operation)
verbose_logger.info(f"CloudZero Logger: Successfully exported {len(cbf_data)} records to CloudZero")
verbose_logger.debug(f"CloudZero Logger: Successfully exported {len(cbf_data)} records to CloudZero")
except Exception as e:
verbose_logger.error(f"CloudZero Logger: Error exporting usage data: {str(e)}")
@ -218,7 +218,7 @@ class CloudZeroLogger(CustomLogger):
unique_services = len(set(record.get('resource/service', '') for record in cbf_data_dict if record.get('resource/service')))
total_tokens = sum(record.get('usage/amount', 0) for record in cbf_data_dict)
verbose_logger.info(f"CloudZero Logger: Dry run completed for {len(cbf_data)} records")
verbose_logger.debug(f"CloudZero Logger: Dry run completed for {len(cbf_data)} records")
return {
"usage_data": usage_data_sample,

View file

@ -352,7 +352,7 @@ class CustomGuardrail(CustomLogger):
self,
guardrail_json_response: Union[Exception, str, dict, List[dict]],
request_data: dict,
guardrail_status: Literal["success", "failure"],
guardrail_status: Literal["success", "failure", "blocked"],
start_time: Optional[float] = None,
end_time: Optional[float] = None,
duration: Optional[float] = None,

View file

@ -44,7 +44,7 @@ try:
request, response, time_elapsed
)
else:
logger.info(f"Unknown OpenAI response object: {response['object']}")
logger.debug(f"Unknown OpenAI response object: {response['object']}")
except Exception as e:
logger.warning(f"Failed to resolve request/response: {e}")
return None

View file

@ -10,7 +10,6 @@ import subprocess
import sys
import time
import traceback
import uuid
from datetime import datetime as dt_object
from functools import lru_cache
from typing import (
@ -27,6 +26,7 @@ from typing import (
cast,
)
import fastuuid as uuid
from httpx import Response
from pydantic import BaseModel
@ -1714,9 +1714,12 @@ class Logging(LiteLLMLoggingBaseClass):
response_obj=result,
start_time=start_time,
end_time=end_time,
litellm_call_id=litellm_params.get(
"litellm_call_id", str(uuid.uuid4())
),
litellm_call_id=current_call_id
if (
current_call_id := litellm_params.get("litellm_call_id")
)
is not None
else str(uuid.uuid4()),
print_verbose=print_verbose,
)
if callback == "wandb" and weightsBiasesLogger is not None:
@ -2774,6 +2777,7 @@ class Logging(LiteLLMLoggingBaseClass):
result: Any,
start_time: datetime.datetime,
end_time: datetime.datetime,
cache_hit: Optional[Any] = None,
) -> None:
"""
Handles calling success callbacks for Async calls.
@ -2788,6 +2792,7 @@ class Logging(LiteLLMLoggingBaseClass):
result,
start_time,
end_time,
cache_hit,
)
def _should_run_sync_callbacks_for_async_calls(self) -> bool:
@ -4499,7 +4504,7 @@ def get_standard_logging_object_payload(
def emit_standard_logging_payload(payload: StandardLoggingPayload):
if os.getenv("LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD"):
verbose_logger.info(json.dumps(payload, indent=4))
print(json.dumps(payload, indent=4)) # noqa
def get_standard_logging_metadata(

View file

@ -1937,7 +1937,7 @@ class CustomStreamWrapper:
)
## Map to OpenAI Exception
try:
exception_type(
raise exception_type(
model=self.model,
custom_llm_provider=self.custom_llm_provider,
original_exception=e,

View file

@ -17,6 +17,7 @@ from litellm.llms.custom_httpx.http_handler import (
HTTPHandler,
_get_httpx_client,
)
from litellm.llms.custom_httpx.aiohttp_transport import LiteLLMAiohttpTransport
from litellm.types.llms.openai import FileTypes
from litellm.types.utils import HttpHandlerRequestFields, ImageResponse, LlmProviders
from litellm.utils import CustomStreamWrapper, ModelResponse, ProviderConfigManager
@ -32,8 +33,71 @@ DEFAULT_TIMEOUT = 600
class BaseLLMAIOHTTPHandler:
def __init__(self):
self.client_session: Optional[aiohttp.ClientSession] = None
def __init__(
self,
client_session: Optional[aiohttp.ClientSession] = None,
transport: Optional[LiteLLMAiohttpTransport] = None,
connector: Optional[aiohttp.BaseConnector] = None,
):
self.client_session = client_session
self._owns_session = (
client_session is None
) # Track if we own the session for cleanup
self.transport = transport
self._owns_transport = (
transport is None
) # Track if we own the transport for cleanup
self.connector = connector
self._owns_connector = (
connector is None
) # Track if we own the connector for cleanup
def _get_or_create_transport(self) -> Optional[LiteLLMAiohttpTransport]:
"""Get existing transport or create a new one if needed."""
if self.transport:
return self.transport
# Create a transport using AsyncHTTPHandler's logic
try:
self.transport = AsyncHTTPHandler._create_aiohttp_transport()
self._owns_transport = True
return self.transport
except Exception:
# If transport creation fails, return None (will use direct session)
return None
def _get_connector(self) -> Optional[aiohttp.BaseConnector]:
"""Get or create a connector for the client session."""
if self.connector:
return self.connector
elif self.transport and hasattr(self.transport, "client"):
# Extract connector from transport if available
client = self.transport.client
if callable(client):
# If client is a factory, we can't extract connector directly
return None
elif hasattr(client, "connector"):
return client.connector
return None
def _create_client_session_with_transport(self) -> ClientSession:
"""Create a new client session using transport or connector configuration."""
connector = self._get_connector()
if self.transport and hasattr(self.transport, "_get_valid_client_session"):
# Use transport's session creation if available
session = self.transport._get_valid_client_session()
return session
elif connector:
# Use provided connector
session = aiohttp.ClientSession(connector=connector)
return session
else:
# Default session creation
session = aiohttp.ClientSession()
return session
def _get_async_client_session(
self, dynamic_client_session: Optional[ClientSession] = None
@ -43,15 +107,33 @@ class BaseLLMAIOHTTPHandler:
elif self.client_session:
return self.client_session
else:
# init client session, and then return new session
self.client_session = aiohttp.ClientSession()
# Create client session using transport/connector if available
self.client_session = self._create_client_session_with_transport()
self._owns_session = True # We created this session, so we own it
return self.client_session
async def close(self):
"""Close the aiohttp client session if it exists."""
if self.client_session and not self.client_session.closed:
"""Close the aiohttp client session and transport if we own them."""
# Close client session if we own it
if (
self.client_session
and not self.client_session.closed
and self._owns_session
):
await self.client_session.close()
# Close transport if we own it
if (
self.transport
and self._owns_transport
and hasattr(self.transport, "aclose")
):
try:
await self.transport.aclose()
except Exception:
# Ignore errors during transport cleanup
pass
async def _make_common_async_call(
self,
async_client_session: Optional[ClientSession],

View file

@ -169,12 +169,18 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
if tool is None:
return None
kwags: dict = {
"name": tool["name"],
"parameters": cast(dict, tool.get("input_schema") or {})
}
description = tool.get("description")
if description is not None:
kwags["description"] = cast(Union[dict, str], description)
return DatabricksTool(
type="function",
function=DatabricksFunction(
name=tool["name"],
parameters=cast(dict, tool.get("input_schema") or {}),
),
function=DatabricksFunction(name=tool["name"], **kwags),
)
def _map_openai_to_dbrx_tool(self, model: str, tools: List) -> List[DatabricksTool]:
@ -331,8 +337,9 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
elif isinstance(content, list):
content_str = ""
for item in content:
if item["type"] == "text":
content_str += item["text"]
if item.get("type") == "text":
text_value = item.get("text", "")
content_str += str(text_value) if text_value is not None else ""
return content_str
else:
raise Exception(f"Unsupported content type: {type(content)}")
@ -361,19 +368,21 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
reasoning_content: Optional[str] = None
if isinstance(content, list):
for item in content:
if item["type"] == "reasoning":
for sum in item["summary"]:
if reasoning_content is None:
reasoning_content = ""
reasoning_content += sum["text"]
thinking_block = ChatCompletionThinkingBlock(
type="thinking",
thinking=sum.get("text", ""),
signature=sum.get("signature", ""),
)
if thinking_blocks is None:
thinking_blocks = []
thinking_blocks.append(thinking_block)
if item.get("type") == "reasoning":
summary_list = item.get("summary", [])
if isinstance(summary_list, list):
for sum in summary_list:
if reasoning_content is None:
reasoning_content = ""
reasoning_content += sum["text"]
thinking_block = ChatCompletionThinkingBlock(
type="thinking",
thinking=sum.get("text", ""),
signature=sum.get("signature", ""),
)
if thinking_blocks is None:
thinking_blocks = []
thinking_blocks.append(thinking_block)
return reasoning_content, thinking_blocks
@staticmethod

View file

@ -272,6 +272,14 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
ResponsesAPIStreamEvents.WEB_SEARCH_CALL_IN_PROGRESS: WebSearchCallInProgressEvent,
ResponsesAPIStreamEvents.WEB_SEARCH_CALL_SEARCHING: WebSearchCallSearchingEvent,
ResponsesAPIStreamEvents.WEB_SEARCH_CALL_COMPLETED: WebSearchCallCompletedEvent,
ResponsesAPIStreamEvents.MCP_LIST_TOOLS_IN_PROGRESS: MCPListToolsInProgressEvent,
ResponsesAPIStreamEvents.MCP_LIST_TOOLS_COMPLETED: MCPListToolsCompletedEvent,
ResponsesAPIStreamEvents.MCP_LIST_TOOLS_FAILED: MCPListToolsFailedEvent,
ResponsesAPIStreamEvents.MCP_CALL_IN_PROGRESS: MCPCallInProgressEvent,
ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DELTA: MCPCallArgumentsDeltaEvent,
ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DONE: MCPCallArgumentsDoneEvent,
ResponsesAPIStreamEvents.MCP_CALL_COMPLETED: MCPCallCompletedEvent,
ResponsesAPIStreamEvents.MCP_CALL_FAILED: MCPCallFailedEvent,
ResponsesAPIStreamEvents.ERROR: ErrorEvent,
}

View file

@ -387,6 +387,19 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
)
if len(tool_call_responses) > 0:
contents.append(ContentType(parts=tool_call_responses))
if len(contents) == 0:
verbose_logger.warning(
"""
No contents in messages. Contents are required. See
https://cloud.google.com/vertex-ai/docs/reference/rest/v1/projects.locations.publishers.models/generateContent#request-body.
If the original request did not comply to OpenAI API requirements it should have failed by now,
but LiteLLM does not check for missing messages.
Setting an empty content to prevent an 400 error.
Relevant Issue - https://github.com/BerriAI/litellm/issues/9733
"""
)
contents.append(ContentType(role="user", parts=[PartType(text=" ")]))
return contents
except Exception as e:
raise e
@ -448,6 +461,17 @@ def _transform_request_body(
) # type: ignore
config_fields = GenerationConfig.__annotations__.keys()
# If the LiteLLM client sends Gemini-supported parameter "labels", add it
# as "labels" field to the request sent to the Gemini backend.
labels: Optional[dict[str, str]] = optional_params.pop("labels", None)
# If the LiteLLM client sends OpenAI-supported parameter "metadata", add it
# as "labels" field to the request sent to the Gemini backend.
if labels is None and "metadata" in litellm_params:
metadata = litellm_params["metadata"]
if metadata is not None and "requester_metadata" in metadata:
rm = metadata["requester_metadata"]
labels = {k: v for k, v in rm.items() if isinstance(v, str)}
filtered_params = {
k: v for k, v in optional_params.items() if k in config_fields
}
@ -468,6 +492,8 @@ def _transform_request_body(
data["generationConfig"] = generation_config
if cached_content is not None:
data["cachedContent"] = cached_content
if labels is not None:
data["labels"] = labels
except Exception as e:
raise e
@ -492,19 +518,21 @@ def sync_transform_request_body(
context_caching_endpoints = ContextCachingEndpoints()
if gemini_api_key is not None:
messages, optional_params, cached_content = (
context_caching_endpoints.check_and_create_cache(
messages=messages,
optional_params=optional_params,
api_key=gemini_api_key,
api_base=api_base,
model=model,
client=client,
timeout=timeout,
extra_headers=extra_headers,
cached_content=optional_params.pop("cached_content", None),
logging_obj=logging_obj,
)
(
messages,
optional_params,
cached_content,
) = context_caching_endpoints.check_and_create_cache(
messages=messages,
optional_params=optional_params,
api_key=gemini_api_key,
api_base=api_base,
model=model,
client=client,
timeout=timeout,
extra_headers=extra_headers,
cached_content=optional_params.pop("cached_content", None),
logging_obj=logging_obj,
)
else: # [TODO] implement context caching for gemini as well
cached_content = optional_params.pop("cached_content", None)

View file

@ -375,10 +375,60 @@ class VertexBase:
url=url,
)
def _handle_reauthentication(
self,
credentials: Optional[VERTEX_CREDENTIALS_TYPES],
project_id: Optional[str],
credential_cache_key: Tuple,
error: Exception,
) -> Tuple[str, str]:
"""
Handle reauthentication when credentials refresh fails.
This method clears the cached credentials and attempts to reload them once.
It should only be called when "Reauthentication is needed" error occurs.
Args:
credentials: The original credentials
project_id: The project ID
credential_cache_key: The cache key to clear
error: The original error that triggered reauthentication
Returns:
Tuple of (access_token, project_id)
Raises:
The original error if reauthentication fails
"""
verbose_logger.debug(
f"Handling reauthentication for project_id: {project_id}. "
f"Clearing cache and retrying once."
)
# Clear the cached credentials
if credential_cache_key in self._credentials_project_mapping:
del self._credentials_project_mapping[credential_cache_key]
# Retry once with _retry_reauth=True to prevent infinite recursion
try:
return self.get_access_token(
credentials=credentials,
project_id=project_id,
_retry_reauth=True,
)
except Exception as retry_error:
verbose_logger.error(
f"Reauthentication retry failed for project_id: {project_id}. "
f"Original error: {str(error)}. Retry error: {str(retry_error)}"
)
# Re-raise the original error for better context
raise error
def get_access_token(
self,
credentials: Optional[VERTEX_CREDENTIALS_TYPES],
project_id: Optional[str],
_retry_reauth: bool = False,
) -> Tuple[str, str]:
"""
Get access token and project id
@ -388,6 +438,14 @@ class VertexBase:
3. Check if loaded credentials have expired
4. If expired, refresh credentials
5. Return access token and project id
Args:
credentials: The credentials to use for authentication
project_id: The Google Cloud project ID
_retry_reauth: Internal flag to prevent infinite recursion during reauthentication
Returns:
Tuple of (access_token, project_id)
"""
# Convert dict credentials to string for caching
@ -481,14 +539,12 @@ class VertexBase:
except Exception as e:
# if refresh fails, it's possible the user has re-authenticated via `gcloud auth application-default login`
# in this case, we should try to reload the credentials by clearing the cache and retrying
if "Reauthentication is needed" in str(e):
verbose_logger.debug(
f"Credential refresh failed for project_id: {project_id}. Deleting from cache and retrying."
)
del self._credentials_project_mapping[credential_cache_key]
return self.get_access_token(
if "Reauthentication is needed" in str(e) and not _retry_reauth:
return self._handle_reauthentication(
credentials=credentials,
project_id=project_id,
credential_cache_key=credential_cache_key,
error=e,
)
raise e

View file

@ -3854,7 +3854,7 @@ def embedding( # noqa: PLR0915
max_retries = kwargs.get("max_retries", None)
litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
mock_response: Optional[List[float]] = kwargs.get("mock_response", None) # type: ignore
azure_ad_token_provider = kwargs.pop("azure_ad_token_provider", None)
azure_ad_token_provider = kwargs.get("azure_ad_token_provider", None)
aembedding = kwargs.get("aembedding", None)
extra_headers = kwargs.get("extra_headers", None)
headers = kwargs.get("headers", None)
@ -5780,9 +5780,8 @@ async def ahealth_check(
input=input or ["test"],
),
"audio_speech": lambda: litellm.aspeech(
**_filter_model_params(model_params),
**{**_filter_model_params(model_params), **({"voice": "alloy"} if "voice" not in _filter_model_params(model_params) else {})},
input=prompt or "test",
voice="alloy",
),
"audio_transcription": lambda: litellm.atranscription(
**_filter_model_params(model_params),

View file

@ -13123,6 +13123,7 @@
"mode": "chat",
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_function_calling": true,
"supports_reasoning": true
},
"openai.gpt-oss-120b-1:0": {
@ -13135,6 +13136,7 @@
"mode": "chat",
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_function_calling": true,
"supports_reasoning": true
},
"anthropic.claude-opus-4-1-20250805-v1:0": {
@ -13877,136 +13879,6 @@
"litellm_provider": "bedrock",
"mode": "chat"
},
"anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-east-1/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-west-2/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0455,
"output_cost_per_second": 0.0455,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.02527,
"output_cost_per_second": 0.02527,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/eu-central-1/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0415,
"output_cost_per_second": 0.0415,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.02305,
"output_cost_per_second": 0.02305,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0175,
"output_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.00972,
"output_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0175,
"output_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.00972,
"output_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"anthropic.claude-v2:1": {
"max_tokens": 8191,
"max_input_tokens": 100000,
@ -15245,7 +15117,7 @@
"mode": "chat",
"source": "https://www.together.ai/models/gpt-oss-120b"
},
"together_ai/OpenAI/gpt-oss-20B": {
"together_ai/openai/gpt-oss-20b": {
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"max_input_tokens": 128000,
@ -15517,16 +15389,6 @@
"litellm_provider": "ollama",
"mode": "completion"
},
"deepinfra/Austism/chronos-hermes-13b-v2": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.3e-07,
"output_cost_per_token": 1.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Gryphe/MythoMax-L2-13b": {
"max_tokens": 4096,
"max_input_tokens": 4096,
@ -15537,26 +15399,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Gryphe/MythoMax-L2-13b-turbo": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.3e-07,
"output_cost_per_token": 1.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1e-07,
"output_cost_per_token": 1e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -15575,78 +15417,18 @@
"output_cost_per_token": 2.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/NovaSky-AI/Sky-T1-32B-Preview": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 1.2e-07,
"output_cost_per_token": 1.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Phind/Phind-CodeLlama-34B-v2": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 6e-07,
"output_cost_per_token": 6e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/QVQ-72B-Preview": {
"max_tokens": 32000,
"max_input_tokens": 32000,
"max_output_tokens": 32000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/QwQ-32B": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/QwQ-32B-Preview": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 1.2e-07,
"output_cost_per_token": 1.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2-72B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 3.5e-07,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/Qwen2-7B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/Qwen2.5-72B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
@ -15667,26 +15449,6 @@
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 6e-08,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2.5-Coder-7B": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 2.5e-08,
"output_cost_per_token": 5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": {
"max_tokens": 128000,
"max_input_tokens": 128000,
@ -15773,30 +15535,11 @@
"max_output_tokens": 262144,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"cache_read_input_token_cost": 2.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Sao10K/L3-70B-Euryale-v2.1": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 7e-07,
"output_cost_per_token": 8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Sao10K/L3-8B-Lunaris-v1": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3e-08,
"output_cost_per_token": 6e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": {
"max_tokens": 8192,
"max_input_tokens": 8192,
@ -15843,6 +15586,7 @@
"max_output_tokens": 200000,
"input_cost_per_token": 3.3e-06,
"output_cost_per_token": 1.65e-05,
"cache_read_input_token_cost": 3.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
@ -15867,67 +15611,15 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/bigcode/starcoder2-15b-instruct-v0.1": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 2.4e-07,
"output_cost_per_token": 2.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/deepinfra/airoboros-70b": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 7e-07,
"output_cost_per_token": 9e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 5e-07,
"output_cost_per_token": 2.18e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false,
"supports_reasoning": true
},
"deepinfra/deepseek-ai/DeepSeek-R1": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 4.5e-07,
"output_cost_per_token": 2.15e-06,
"input_cost_per_token": 7e-07,
"output_cost_per_token": 2.4e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-0528": {
"max_tokens": 163840,
@ -15935,10 +15627,10 @@
"max_output_tokens": 163840,
"input_cost_per_token": 5e-07,
"output_cost_per_token": 2.15e-06,
"cache_read_input_token_cost": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": {
"max_tokens": 32768,
@ -15948,8 +15640,7 @@
"output_cost_per_token": 3e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": {
"max_tokens": 131072,
@ -15959,8 +15650,7 @@
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false,
"supports_reasoning": true
"supports_tool_choice": false
},
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": {
"max_tokens": 131072,
@ -15970,19 +15660,17 @@
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-Turbo": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 40960,
"max_input_tokens": 40960,
"max_output_tokens": 40960,
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-V3": {
"max_tokens": 163840,
@ -15992,8 +15680,7 @@
"output_cost_per_token": 8.9e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-V3-0324": {
"max_tokens": 163840,
@ -16001,63 +15688,23 @@
"max_output_tokens": 163840,
"input_cost_per_token": 2.8e-07,
"output_cost_per_token": 8.8e-07,
"cache_read_input_token_cost": 2.24e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
},
"deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-V3.1": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 3e-07,
"input_cost_per_token": 2.7e-07,
"output_cost_per_token": 1e-06,
"cache_read_input_token_cost": 2.16e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false,
"supports_tool_choice": true,
"supports_reasoning": true
},
"deepinfra/google/codegemma-7b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 7e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/google/gemini-1.5-flash": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemini-1.5-flash-8b": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"input_cost_per_token": 3.75e-08,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemini-2.0-flash-001": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
@ -16088,36 +15735,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemma-1.1-7b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 7e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemma-2-27b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 2.7e-07,
"output_cost_per_token": 2.7e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/google/gemma-2-9b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3e-08,
"output_cost_per_token": 6e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/google/gemma-3-12b-it": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16142,48 +15759,8 @@
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 2e-08,
"output_cost_per_token": 4e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/lizpreciatior/lzlv_70b_fp16_hf": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mattshumer/Reflection-Llama-3.1-70B": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-2-13b-chat-hf": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.3e-07,
"output_cost_per_token": 1.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-2-70b-chat-hf": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 6.4e-07,
"output_cost_per_token": 8e-07,
"input_cost_per_token": 4e-08,
"output_cost_per_token": 8e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
@ -16198,16 +15775,6 @@
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-3.2-1B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 5e-09,
"output_cost_per_token": 1e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-3.2-3B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16218,16 +15785,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-3.3-70B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16258,16 +15815,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 5e-07,
"output_cost_per_token": 5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
"max_tokens": 327680,
"max_input_tokens": 327680,
@ -16298,16 +15845,6 @@
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": {
"max_tokens": 8192,
"max_input_tokens": 8192,
@ -16318,16 +15855,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 8e-07,
"output_cost_per_token": 8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16368,36 +15895,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/microsoft/Phi-3-medium-4k-instruct": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 1.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/microsoft/Phi-4-multimodal-instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 1e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/microsoft/WizardLM-2-7B": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/microsoft/WizardLM-2-8x22B": {
"max_tokens": 65536,
"max_input_tokens": 65536,
@ -16418,66 +15915,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/microsoft/phi-4-reasoning-plus": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 3.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mistralai/Devstral-Small-2505": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 6e-08,
"output_cost_per_token": 1.2e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Devstral-Small-2507": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 2.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-7B-Instruct-v0.1": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-7B-Instruct-v0.2": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mistralai/Mistral-7B-Instruct-v0.3": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 2.8e-08,
"output_cost_per_token": 5.4e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-Nemo-Instruct-2407": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16498,16 +15935,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 1e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": {
"max_tokens": 128000,
"max_input_tokens": 128000,
@ -16518,16 +15945,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": {
"max_tokens": 65536,
"max_input_tokens": 65536,
"max_output_tokens": 65536,
"input_cost_per_token": 6.5e-07,
"output_cost_per_token": 6.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": {
"max_tokens": 32768,
"max_input_tokens": 32768,
@ -16558,16 +15975,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/nvidia/Nemotron-4-340B-Instruct": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 4.2e-06,
"output_cost_per_token": 4.2e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/openai/gpt-oss-120b": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16588,36 +15995,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/openbmb/MiniCPM-Llama3-V-2_5": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3.4e-07,
"output_cost_per_token": 3.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/openchat/openchat-3.6-8b": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/openchat/openchat_3.5": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/zai-org/GLM-4.5": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -21126,4 +20503,4 @@
"notes": "Volcengine Doubao embedding model - text-240715 version with 2560 dimensions"
}
}
}
}

View file

@ -241,6 +241,9 @@ class MCPServerManager:
transport=server_config.get("transport", MCPTransport.http),
spec_version=server_config.get("spec_version", MCPSpecVersion.jun_2025),
auth_type=server_config.get("auth_type", None),
authentication_token=server_config.get(
"authentication_token", server_config.get("auth_value", None)
),
mcp_info=mcp_info,
access_groups=server_config.get("access_groups", None),
)
@ -716,8 +719,8 @@ class MCPServerManager:
tasks = []
if proxy_logging_obj:
# Create synthetic LLM data for during hook processing
from litellm.types.mcp import MCPDuringCallRequestObject
from litellm.types.llms.base import HiddenParams
from litellm.types.mcp import MCPDuringCallRequestObject
request_obj = MCPDuringCallRequestObject(
tool_name=name,

View file

@ -215,9 +215,9 @@ if MCP_AVAILABLE:
"""
from fastapi import Request
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
from litellm.proxy.litellm_pre_call_utils import add_litellm_data_to_request
from litellm.proxy.proxy_server import proxy_config
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
# Validate arguments
user_api_key_auth, mcp_auth_header, _, mcp_server_auth_headers, mcp_protocol_version = get_auth_context()
@ -279,33 +279,15 @@ if MCP_AVAILABLE:
############ Helper Functions ##########################
########################################################
async def _get_tools_from_mcp_servers(
user_api_key_auth: Optional[UserAPIKeyAuth],
mcp_auth_header: Optional[str],
async def _get_allowed_mcp_servers_from_mcp_server_names(
mcp_servers: Optional[List[str]],
mcp_server_auth_headers: Optional[Dict[str, str]] = None,
mcp_protocol_version: Optional[str] = None,
) -> List[MCPTool]:
allowed_mcp_servers: List[str],
) -> List[str]:
"""
Helper method to fetch tools from MCP servers based on server filtering criteria.
Args:
user_api_key_auth: User authentication info for access control
mcp_auth_header: Optional auth header for MCP server (deprecated)
mcp_servers: Optional list of server names/aliases to filter by
mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
Returns:
List[MCPTool]: Combined list of tools from filtered servers
Get the filtered MCP servers from the MCP server names
"""
if not MCP_AVAILABLE:
return []
# Get allowed MCP servers based on user permissions
allowed_mcp_servers = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
filtered_server_ids = set()
from typing import Set
filtered_server_ids: Set[str] = set()
# Filter servers based on mcp_servers parameter if provided
if mcp_servers is not None:
for server_or_group in mcp_servers:
@ -336,6 +318,40 @@ if MCP_AVAILABLE:
if filtered_server_ids:
allowed_mcp_servers = list(filtered_server_ids)
return allowed_mcp_servers
async def _get_tools_from_mcp_servers(
user_api_key_auth: Optional[UserAPIKeyAuth],
mcp_auth_header: Optional[str],
mcp_servers: Optional[List[str]],
mcp_server_auth_headers: Optional[Dict[str, str]] = None,
mcp_protocol_version: Optional[str] = None,
) -> List[MCPTool]:
"""
Helper method to fetch tools from MCP servers based on server filtering criteria.
Args:
user_api_key_auth: User authentication info for access control
mcp_auth_header: Optional auth header for MCP server (deprecated)
mcp_servers: Optional list of server names/aliases to filter by
mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
Returns:
List[MCPTool]: Combined list of tools from filtered servers
"""
if not MCP_AVAILABLE:
return []
# Get allowed MCP servers based on user permissions
allowed_mcp_servers = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
if mcp_servers is not None:
allowed_mcp_servers = await _get_allowed_mcp_servers_from_mcp_server_names(
mcp_servers=mcp_servers,
allowed_mcp_servers=allowed_mcp_servers,
)
# Get tools from each allowed server
all_tools = []
@ -556,20 +572,25 @@ if MCP_AVAILABLE:
except Exception as e:
return [TextContent(text=f"Error: {str(e)}", type="text")]
async def extract_mcp_auth_context(scope, path):
def _get_mcp_servers_in_path(path: str) -> Optional[List[str]]:
"""
Extracts mcp_servers from the path and processes the MCP request for auth context.
Returns: (user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers)
Get the MCP servers from the path
"""
import re
mcp_servers_from_path = None
mcp_servers_from_path: Optional[List[str]] = None
mcp_path_match = re.match(r"^/mcp/([^/]+)(/.*)?$", path)
if mcp_path_match:
mcp_servers_str = mcp_path_match.group(1)
if mcp_servers_str:
mcp_servers_from_path = [s.strip() for s in mcp_servers_str.split(",") if s.strip()]
return mcp_servers_from_path
async def extract_mcp_auth_context(scope, path):
"""
Extracts mcp_servers from the path and processes the MCP request for auth context.
Returns: (user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers)
"""
mcp_servers_from_path = _get_mcp_servers_in_path(path)
if mcp_servers_from_path is not None:
(
user_api_key_auth,

View file

@ -7,24 +7,3 @@ model_list:
- model_name: wildcard_models/*
litellm_params:
model: openai/*
- model_name: gpt-5-mini
litellm_params:
model: azure/gpt-5-mini
api_base: os.environ/AZURE_GPT_5_MINI_API_BASE # runs os.getenv("AZURE_API_BASE")
api_key: os.environ/AZURE_GPT_5_MINI_API_KEY # runs os.getenv("AZURE_API_KEY")
stream_timeout: 60
merge_reasoning_content_in_choices: true
model_info:
mode: chat
- model_name: ollama-deepseek-r1
litellm_params:
model: ollama/deepseek-r1:1.5b
model_info:
mode: chat
router_settings:
model_group_alias: {"my-fake-gpt-4": "fake-openai-endpoint"}
litellm_settings:
callbacks: ["otel"]
success_callback: ["braintrust"]

View file

@ -205,7 +205,7 @@ async def anthropic_response( # noqa: PLR0915
data=data, user_api_key_dict=user_api_key_dict, response=response # type: ignore
)
verbose_proxy_logger.info("\nResponse from Litellm:\n{}".format(response))
verbose_proxy_logger.debug("\nResponse from Litellm:\n{}".format(response))
return response
except Exception as e:
await proxy_logging_obj.post_call_failure_hook(

View file

@ -154,7 +154,7 @@ class DBSpendUpdateWriter:
prisma_client=prisma_client,
)
else:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"disable_spend_logs=True. Skipping writing spend logs to db. Other spend updates - Key/User/Team table will still occur."
)
@ -252,7 +252,7 @@ class DBSpendUpdateWriter:
)
)
except Exception as e:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"\033[91m"
+ f"Update User DB call failed to execute {str(e)}\n{traceback.format_exc()}"
)
@ -294,7 +294,7 @@ class DBSpendUpdateWriter:
except Exception:
pass
except Exception as e:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Update Team DB failed to execute - {str(e)}\n{traceback.format_exc()}"
)
raise e
@ -320,7 +320,7 @@ class DBSpendUpdateWriter:
)
)
except Exception as e:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Update Org DB failed to execute - {str(e)}\n{traceback.format_exc()}"
)
raise e
@ -331,7 +331,7 @@ class DBSpendUpdateWriter:
prisma_client: Optional[PrismaClient] = None,
spend_logs_url: Optional[str] = os.getenv("SPEND_LOGS_URL"),
) -> Optional[PrismaClient]:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"Writing spend log to db - request_id: {}, spend: {}".format(
payload.get("request_id"), payload.get("spend")
)
@ -959,7 +959,7 @@ class DBSpendUpdateWriter:
},
)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Processed {len(transactions_to_process)} daily {entity_type} transactions in {time.time() - start_time:.2f}s"
)
@ -1087,7 +1087,7 @@ class DBSpendUpdateWriter:
return None
request_status = prisma_client.get_request_status(payload)
verbose_proxy_logger.info(f"Logged request status: {request_status}")
verbose_proxy_logger.debug(f"Logged request status: {request_status}")
_metadata: SpendLogsMetadata = json.loads(payload["metadata"])
usage_obj = _metadata.get("usage_object", {}) or {}
if isinstance(payload["startTime"], datetime):

View file

@ -73,7 +73,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
self.api_base = api_base
self.api_version = kwargs.get("api_version") or "2024-09-01"
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Initialized Azure Prompt Shield Guardrail: {guardrail_name}"
)
@ -131,7 +131,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
Raises HTTPException if content should be blocked.
"""
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"Azure Prompt Shield: Running pre-call prompt scan, on call_type: %s",
call_type,
)
@ -145,7 +145,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
user_prompt = self.get_user_prompt(new_messages)
if user_prompt:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Azure Prompt Shield: User prompt: {user_prompt}"
)
azure_prompt_shield_response = await self.async_make_request(
@ -180,7 +180,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
Raises HTTPException if response should be blocked.
"""
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"Azure Prompt Shield: Running post-call response scan"
)

View file

@ -232,7 +232,7 @@ class LakeraAIGuardrail(CustomGuardrail):
lakera_response=lakera_guardrail_response,
masked_entity_count=masked_entity_count,
)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"Lakera AI: Masked PII in messages instead of blocking request"
)
else:
@ -299,7 +299,7 @@ class LakeraAIGuardrail(CustomGuardrail):
lakera_response=lakera_guardrail_response,
masked_entity_count=masked_entity_count,
)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"Lakera AI: Masked PII in messages instead of blocking request"
)
else:

View file

@ -39,4 +39,4 @@ guardrail_initializer_registry = {
guardrail_class_registry = {
SupportedGuardrailIntegrations.MODEL_ARMOR.value: ModelArmorGuardrail,
}
}

View file

@ -58,7 +58,7 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
# Initialize parent classes first
super().__init__(**kwargs)
VertexBase.__init__(self)
# Then set our attributes (this ensures project_id is not overwritten)
self.async_handler = get_async_httpx_client(
llm_provider=httpxSpecialProvider.GuardrailCallback
@ -94,14 +94,12 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
else:
return {"model_response_data": {"text": content}}
def _extract_content_from_response(
self, response: Union[Any, ModelResponse]
) -> str:
"""
Extract text content from model response.
Returns empty string for non-text responses (TTS, images, etc.) to skip guardrail processing.
"""
from litellm.litellm_core_utils.prompt_templates.common_utils import (
@ -193,22 +191,90 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
def _should_block_content(self, armor_response: dict) -> bool:
"""Check if Model Armor response indicates content should be blocked."""
# Model Armor may return different response structures
# This is a basic implementation - adjust based on actual API response
if armor_response.get("blocked", False):
# Check the sanitizationResult from Model Armor API
sanitization_result = armor_response.get("sanitizationResult", {})
filter_results = sanitization_result.get("filterResults", {})
# Check blocking filters (these should cause the request to be blocked)
# RAI (Responsible AI) filters
rai_results = filter_results.get("rai", {}).get("raiFilterResult", {})
if rai_results.get("matchState") == "MATCH_FOUND":
return True
# Check for sanitization actions
if armor_response.get("action") == "BLOCK":
# Prompt injection and jailbreak filters
pi_jailbreak = filter_results.get("piAndJailbreakFilterResult", {})
if pi_jailbreak.get("matchState") == "MATCH_FOUND":
return True
# Malicious URI filters
malicious_uri = filter_results.get("maliciousUriFilterResult", {})
if malicious_uri.get("matchState") == "MATCH_FOUND":
return True
# CSAM filters
csam = filter_results.get("csamFilterFilterResult", {})
if csam.get("matchState") == "MATCH_FOUND":
return True
# Virus scan filters
virus_scan = filter_results.get("virusScanFilterResult", {})
if virus_scan.get("matchState") == "MATCH_FOUND":
return True
return False
def _get_sanitized_content(self, armor_response: dict) -> Optional[str]:
"""Extract sanitized content from Model Armor response."""
# This depends on the actual Model Armor API response structure
# Adjust based on documentation
return armor_response.get("sanitized_text") or armor_response.get("text")
# Model Armor returns sanitized content in the sanitizationResult
sanitization_result = armor_response.get("sanitizationResult", {})
# Check for sdp structure (for deidentification)
filter_results = sanitization_result.get("filterResults", {})
sdp = filter_results.get("sdp", {}).get("sdpFilterResult")
if sdp is not None:
# Model Armor returns sanitized text under deidentifyResult in sdp
deidentify_result = sdp.get("deidentifyResult", {})
sanitized_text = deidentify_result.get("data", {}).get("text", "")
if deidentify_result.get("matchState") == "MATCH_FOUND" and sanitized_text:
return sanitized_text
# Fallback to checking root level
return armor_response.get("sanitizedText") or armor_response.get("text")
def _process_response(
self,
response: Optional[dict],
request_data: dict,
start_time: Optional[float] = None,
end_time: Optional[float] = None,
duration: Optional[float] = None,
):
"""
Override to store only the Model Armor API response, not the entire data dict.
This prevents circular references in logging.
"""
# Retrieve the Model Armor response & status stored on the per-request `metadata` object.
metadata = (
request_data.get("metadata", {}) if isinstance(request_data, dict) else {}
)
guardrail_response = metadata.get("_model_armor_response", {})
# Determine status – default to "success" but prefer the explicit value if present.
guardrail_status: Literal["success", "failure", "blocked"] = metadata.get(
"_model_armor_status", "success"
) # type: ignore
self.add_standard_logging_guardrail_information_to_request_data(
guardrail_json_response=guardrail_response,
request_data=request_data,
guardrail_status=guardrail_status, # type: ignore
duration=duration,
start_time=start_time,
end_time=end_time,
)
return response
@log_guardrail_information
async def async_pre_call_hook(
@ -263,6 +329,24 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
request_data=data,
)
# Store the armor response for logging
# Attach Model Armor response + evaluation status directly to the per-request metadata to avoid
# race-conditions between concurrent requests which share the same guardrail instance.
# This ensures each request logs its own Model Armor response instead of a potentially stale value
# overwritten by another coroutine.
if isinstance(data, dict):
metadata = data.setdefault(
"metadata", {}
) # ensures metadata exists and is unique per request
metadata["_model_armor_response"] = armor_response
# Pre-compute guardrail status for downstream logging. A blocked response will eventually raise
# an HTTPException, however in scenarios where the caller decides to ignore the exception (e.g.
# fail_on_error=False) we still want the correct status reflected.
metadata["_model_armor_status"] = (
"blocked"
if self._should_block_content(armor_response)
else "success"
)
# Check if content should be blocked
if self._should_block_content(armor_response):
raise HTTPException(
@ -339,6 +423,16 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
request_data=data,
)
# Attach Model Armor response & status to this request's metadata to prevent race conditions
if isinstance(data, dict):
metadata = data.setdefault("metadata", {})
metadata["_model_armor_response"] = armor_response
metadata["_model_armor_status"] = (
"blocked"
if self._should_block_content(armor_response)
else "success"
)
# Check if content should be blocked
if self._should_block_content(armor_response):
raise HTTPException(
@ -406,6 +500,16 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
request_data=request_data,
)
# Attach Model Armor response & status to this request's metadata to avoid race conditions
if isinstance(request_data, dict):
metadata = request_data.setdefault("metadata", {})
metadata["_model_armor_response"] = armor_response
metadata["_model_armor_status"] = (
"blocked"
if self._should_block_content(armor_response)
else "success"
)
# Check if blocked
if self._should_block_content(armor_response):
raise HTTPException(

View file

@ -86,7 +86,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
if not self.api_key:
raise ValueError("OpenAI Moderation: api_key is required. Set OPENAI_API_KEY environment variable or pass it in configuration.")
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Initialized OpenAI Moderation Guardrail: {guardrail_name} with model: {self.model}"
)
@ -201,7 +201,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
Raises HTTPException if content should be blocked.
"""
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"OpenAI Moderation: Running pre-call prompt scan, on call_type: %s",
call_type,
)
@ -219,7 +219,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
user_prompt = self.get_user_prompt(new_messages)
if user_prompt:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"OpenAI Moderation: User prompt: {user_prompt[:100]}..." # Log first 100 chars for debugging
)
@ -256,7 +256,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
Raises HTTPException if content should be blocked.
"""
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"OpenAI Moderation: Running moderation hook, on call_type: %s",
call_type,
)
@ -295,14 +295,14 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
Raises HTTPException if response should be blocked.
"""
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"OpenAI Moderation: Running post-call response scan"
)
# Extract response text for moderation
response_text = self._extract_response_text(response)
if response_text:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"OpenAI Moderation: Response text: {response_text[:100]}..." # Log first 100 chars
)
@ -333,7 +333,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
from litellm.main import stream_chunk_builder
from litellm.types.utils import TextCompletionResponse
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"OpenAI Moderation: Running streaming response scan"
)
@ -362,7 +362,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
# Extract response text for moderation
response_text = self._extract_response_text(assembled_model_response)
if response_text:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"OpenAI Moderation: Streaming response text: {response_text[:100]}..." # Log first 100 chars
)

View file

@ -19,7 +19,12 @@ from litellm.proxy.common_utils.callback_utils import (
add_guardrail_to_applied_guardrails_header,
)
from litellm.types.guardrails import GuardrailEventHooks
from litellm.types.utils import Choices, LLMResponseTypes, ModelResponse, TextCompletionResponse
from litellm.types.utils import (
Choices,
LLMResponseTypes,
ModelResponse,
TextCompletionResponse,
)
if TYPE_CHECKING:
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
@ -92,7 +97,7 @@ class PangeaHandler(CustomGuardrail):
# Pass relevant kwargs to the parent class
super().__init__(guardrail_name=guardrail_name, **kwargs)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Initialized Pangea Guardrail: name={guardrail_name}, recipe={pangea_input_recipe}, api_base={self.api_base}"
)
@ -147,7 +152,7 @@ class PangeaHandler(CustomGuardrail):
"guardrail_name": self.guardrail_name,
},
)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Pangea Guardrail ({hook_name}): Request passed. Response: {result.get('result', {}).get('detectors')}"
)

View file

@ -64,7 +64,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
)
self.profile_name = profile_name
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Initialized PANW Prisma AIRS Guardrail: {guardrail_name}"
)
@ -253,7 +253,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
Raises HTTPException if content should be blocked.
"""
verbose_proxy_logger.info("PANW Prisma AIRS: Running pre-call prompt scan")
verbose_proxy_logger.debug("PANW Prisma AIRS: Running pre-call prompt scan")
# Extract prompt text from messages
messages = data.get("messages", [])
@ -280,7 +280,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
category = scan_result.get("category", "unknown")
if action == "allow":
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"PANW Prisma AIRS: Response allowed (Category: {category})"
)
@ -305,7 +305,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
Raises HTTPException if response should be blocked.
"""
verbose_proxy_logger.info("PANW Prisma AIRS: Running post-call response scan")
verbose_proxy_logger.debug("PANW Prisma AIRS: Running post-call response scan")
# Extract response text
response_text = self._extract_response_text(response)
@ -331,7 +331,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
category = scan_result.get("category", "unknown")
if action == "allow":
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"PANW Prisma AIRS: Response allowed (Category: {category})"
)

View file

@ -428,7 +428,7 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
messages[index][
"content"
] = r # replace content with redacted string
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Presidio PII Masking: Redacted pii message: {data['messages']}"
)
data["messages"] = messages
@ -513,7 +513,7 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
messages[index][
"content"
] = r # replace content with redacted string
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Presidio PII Masking: Redacted pii message: {messages}"
)
kwargs["messages"] = messages

View file

@ -137,11 +137,14 @@ def _update_litellm_params_for_health_check(
- gets a short `messages` param for health check
- updates the `model` param with the `health_check_model` if it exists Doc: https://docs.litellm.ai/docs/proxy/health#wildcard-routes
- updates the `voice` param with the `health_check_voice` for `audio_speech` mode if it exists Doc: https://docs.litellm.ai/docs/proxy/health#text-to-speech-models
"""
litellm_params["messages"] = _get_random_llm_message()
_health_check_model = model_info.get("health_check_model", None)
if _health_check_model is not None:
litellm_params["model"] = _health_check_model
if model_info.get("mode", None) == "audio_speech":
litellm_params["voice"] = model_info.get("health_check_voice", "alloy")
return litellm_params

View file

@ -128,7 +128,7 @@ class _ProxyDBLogger(CustomLogger):
user_api_key = metadata.get("user_api_key", None)
if kwargs.get("cache_hit", False) is True:
response_cost = 0.0
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Cache Hit: response_cost {response_cost}, for user_id {user_id}"
)

View file

@ -473,7 +473,7 @@ async def _common_key_generation_helper( # noqa: PLR0915
data = apply_enterprise_key_management_params(data, team_table)
except Exception as e:
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"litellm.proxy.proxy_server.generate_key_fn(): Enterprise key management params not applied - {}".format(
str(e)
)
@ -551,6 +551,15 @@ async def _common_key_generation_helper( # noqa: PLR0915
prisma_client=prisma_client,
)
# Validate user-provided key format
if data.key is not None and not data.key.startswith("sk-"):
raise HTTPException(
status_code=400,
detail={
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {data.key}"
}
)
response = await generate_key_helper_fn(
request_type="key", **data_json, table_name="key"
)
@ -2004,7 +2013,7 @@ async def _rotate_master_key(
# 2. process model table
if models:
decrypted_models = proxy_config.decrypt_model_list_from_db(new_models=models)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
"ABLE TO DECRYPT MODELS - len(decrypted_models): %s", len(decrypted_models)
)
new_models = []
@ -2018,9 +2027,9 @@ async def _rotate_master_key(
)
if new_model:
new_models.append(jsonify_object(new_model.model_dump()))
verbose_proxy_logger.info("Resetting proxy model table")
verbose_proxy_logger.debug("Resetting proxy model table")
await prisma_client.db.litellm_proxymodeltable.delete_many()
verbose_proxy_logger.info("Creating %s models", len(new_models))
verbose_proxy_logger.debug("Creating %s models", len(new_models))
await prisma_client.db.litellm_proxymodeltable.create_many(
data=new_models,
)

View file

@ -17,8 +17,8 @@ Endpoints here:
"""
import importlib
from typing import Iterable, List, Optional
from datetime import datetime
from typing import Iterable, List, Optional
from fastapi import APIRouter, Depends, Header, HTTPException, Response, status
from fastapi.responses import JSONResponse
@ -26,7 +26,9 @@ from fastapi.responses import JSONResponse
import litellm
from litellm._logging import verbose_logger, verbose_proxy_logger
from litellm.constants import LITELLM_PROXY_ADMIN_NAME
from litellm.proxy._experimental.mcp_server.utils import validate_and_normalize_mcp_server_payload
from litellm.proxy._experimental.mcp_server.utils import (
validate_and_normalize_mcp_server_payload,
)
router = APIRouter(prefix="/v1/mcp", tags=["mcp"])
MCP_AVAILABLE: bool = True
@ -94,34 +96,17 @@ if MCP_AVAILABLE:
"""
Get all MCP tools available for the current key, including those from access groups
"""
from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
MCPRequestHandler,
)
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
global_mcp_server_manager,
from litellm.proxy._experimental.mcp_server.server import _list_mcp_tools
tools = await _list_mcp_tools(
user_api_key_auth=user_api_key_dict,
mcp_auth_header=None,
mcp_servers=None,
mcp_server_auth_headers=None,
mcp_protocol_version=None,
)
dumped_tools = [dict(tool) for tool in tools]
# This now includes both direct and access group servers
server_ids = await MCPRequestHandler._get_allowed_mcp_servers_for_key(user_api_key_dict)
tools = []
errors = []
for server_id in server_ids:
try:
server_tools = await global_mcp_server_manager.get_tools_for_server(server_id)
tools.extend(server_tools)
verbose_proxy_logger.debug(f"Successfully fetched {len(server_tools)} tools from server {server_id}")
except Exception as e:
error_msg = f"Failed to get tools from server {server_id}: {str(e)}"
verbose_proxy_logger.warning(error_msg)
errors.append(error_msg)
# Continue with other servers instead of failing completely
verbose_proxy_logger.debug(f"Available tools: {tools}")
if errors:
verbose_proxy_logger.warning(f"Some servers failed to respond: {errors}")
return {"tools": tools}
return {"tools": dumped_tools}
@router.get(
"/access_groups",
@ -134,8 +119,10 @@ if MCP_AVAILABLE:
"""
Get all available MCP access groups from the database AND config
"""
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
global_mcp_server_manager,
)
from litellm.proxy.proxy_server import prisma_client
from litellm.proxy._experimental.mcp_server.mcp_server_manager import global_mcp_server_manager
access_groups = set()

View file

@ -1023,7 +1023,7 @@ async def update_public_model_groups(
# Save the updated config
await proxy_config.save_config(new_config=config)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Updated public model groups to: {request.model_groups} by user: {user_api_key_dict.user_id}"
)
@ -1090,7 +1090,7 @@ async def update_useful_links(
# Save the updated config
await proxy_config.save_config(new_config=config)
verbose_proxy_logger.info(
verbose_proxy_logger.debug(
f"Updated useful links to: {request.useful_links} by user: {user_api_key_dict.user_id}"
)

View file

@ -264,7 +264,7 @@ async def chat_completion_pass_through_endpoint( # noqa: PLR0915
)
)
verbose_proxy_logger.info("\nResponse from Litellm:\n{}".format(response))
verbose_proxy_logger.debug("\nResponse from Litellm:\n{}".format(response))
return response
except Exception as e:
await proxy_logging_obj.post_call_failure_hook(

View file

@ -311,7 +311,7 @@ class ProxyInitializationHelpers:
@click.option(
"--num_workers",
default=DEFAULT_NUM_WORKERS_LITELLM_PROXY,
help="Number of uvicorn / gunicorn workers to spin up. By default, 4 uvicorn workers are used.",
help="Number of uvicorn / gunicorn workers to spin up. By default, it equals the number of logical CPUs in the system, or 4 workers if that cannot be determined.",
envvar="NUM_WORKERS",
)
@click.option("--api_base", default=None, help="API base URL.")

View file

@ -3,5 +3,16 @@ model_list:
litellm_params:
model: openai/*
api_base: https://exampleopenaiendpoint-production-0ee2.up.railway.app/
- model_name: bedrock/*
litellm_params:
model: bedrock/*
- model_name: openai/*
litellm_params:
model: openai/*
- model_name: gemini/*
litellm_params:
model: gemini/*
litellm_settings:
callbacks: ["cloudzero"]

View file

@ -248,9 +248,7 @@ from litellm.proxy.management_endpoints.customer_endpoints import (
from litellm.proxy.management_endpoints.internal_user_endpoints import (
router as internal_user_router,
)
from litellm.proxy.management_endpoints.internal_user_endpoints import (
user_update,
)
from litellm.proxy.management_endpoints.internal_user_endpoints import user_update
from litellm.proxy.management_endpoints.key_management_endpoints import (
delete_verification_tokens,
duration_in_seconds,
@ -297,9 +295,7 @@ from litellm.proxy.middleware.prometheus_auth_middleware import PrometheusAuthMi
from litellm.proxy.openai_files_endpoints.files_endpoints import (
router as openai_files_router,
)
from litellm.proxy.openai_files_endpoints.files_endpoints import (
set_files_config,
)
from litellm.proxy.openai_files_endpoints.files_endpoints import set_files_config
from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import (
passthrough_endpoint_router,
)
@ -3548,7 +3544,7 @@ def giveup(e):
return True # giveup if queuing max parallel request limits is disabled
if result:
verbose_proxy_logger.info(json.dumps({"event": "giveup", "exception": str(e)}))
verbose_proxy_logger.debug(json.dumps({"event": "giveup", "exception": str(e)}))
return result
@ -3812,9 +3808,7 @@ class ProxyStartupEvent:
# CloudZero Background Job
########################################################
from litellm.integrations.cloudzero.cloudzero import CloudZeroLogger
from litellm.proxy.spend_tracking.cloudzero_endpoints import (
is_cloudzero_setup,
)
from litellm.proxy.spend_tracking.cloudzero_endpoints import is_cloudzero_setup
if await is_cloudzero_setup():
await CloudZeroLogger.init_cloudzero_background_job(scheduler=scheduler)
@ -7601,7 +7595,10 @@ async def login(request: Request): # noqa: PLR0915
data=UpdateUserRequest(
user_id=key_user_id,
user_role=user_role,
)
),
user_api_key_dict=UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN,
),
)
if os.getenv("DATABASE_URL") is not None:
response = await generate_key_helper_fn(

View file

@ -149,13 +149,6 @@ async def view_spend_tags(
```
"""
try:
from enterprise.utils import get_spend_by_tags
except ImportError:
raise Exception(
"Trying to use Spend by Tags"
+ CommonProxyErrors.missing_enterprise_package_docker.value
)
from litellm.proxy.proxy_server import prisma_client
try:

View file

@ -1,12 +1,24 @@
import asyncio
import contextvars
from functools import partial
from typing import Any, Coroutine, Dict, Iterable, List, Literal, Optional, Type, Union
from typing import (
TYPE_CHECKING,
Any,
Coroutine,
Dict,
Iterable,
List,
Literal,
Optional,
Type,
Union,
)
import httpx
from pydantic import BaseModel
import litellm
from litellm._logging import verbose_logger
from litellm.constants import request_timeout
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfig
@ -22,15 +34,27 @@ from litellm.types.llms.openai import (
ResponseInputParam,
ResponsesAPIOptionalRequestParams,
ResponsesAPIResponse,
ResponseText,
ToolChoice,
ToolParam,
)
# Handle ResponseText import with fallback
if TYPE_CHECKING:
from litellm.types.llms.openai import ResponseText
else:
ResponseText = str # Fallback for ResponseText import
from litellm.types.responses.main import *
from litellm.types.router import GenericLiteLLMParams
from litellm.utils import ProviderConfigManager, client
from .streaming_iterator import BaseResponsesAPIStreamingIterator
if TYPE_CHECKING:
from mcp.types import Tool as MCPTool
else:
MCPTool = Any
from .streaming_iterator import (
BaseResponsesAPIStreamingIterator,
)
####### ENVIRONMENT VARIABLES ###################
# Initialize any necessary instances or variables here
@ -141,17 +165,15 @@ async def aresponses_api_with_mcp(
other_tools,
) = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools)
# Get available tools from MCP manager if we have MCP tools
openai_tools = []
mcp_tools_fetched = []
if mcp_tools_with_litellm_proxy:
user_api_key_auth = kwargs.get("user_api_key_auth")
mcp_tools_fetched = await LiteLLM_Proxy_MCP_Handler._get_mcp_tools_from_manager(
user_api_key_auth
)
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(
mcp_tools_fetched
)
# Process MCP tools through the complete pipeline (fetch + filter + deduplicate + transform)
user_api_key_auth = kwargs.get("user_api_key_auth")
# Get original MCP tools (for events) and OpenAI tools (for LLM) by reusing existing methods
original_mcp_tools = await LiteLLM_Proxy_MCP_Handler._process_mcp_tools_without_openai_transform(
user_api_key_auth=user_api_key_auth,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
)
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(original_mcp_tools)
# Combine with other tools
all_tools = openai_tools + other_tools if (openai_tools or other_tools) else None
@ -182,23 +204,68 @@ async def aresponses_api_with_mcp(
**kwargs,
}
# Handle MCP streaming if requested
if stream and mcp_tools_with_litellm_proxy:
# Generate MCP discovery events using the already processed tools
import uuid
from litellm.responses.mcp.mcp_streaming_iterator import (
create_mcp_list_tools_events,
)
base_item_id = f"mcp_{uuid.uuid4().hex[:8]}"
mcp_discovery_events = await create_mcp_list_tools_events(
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
user_api_key_auth=user_api_key_auth,
base_item_id=base_item_id,
pre_processed_mcp_tools=original_mcp_tools
)
return LiteLLM_Proxy_MCP_Handler._create_mcp_streaming_response(
input=input,
model=model,
all_tools=all_tools,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
mcp_discovery_events=mcp_discovery_events,
call_params=call_params,
previous_response_id=previous_response_id,
**kwargs
)
# Determine if we should auto-execute tools
should_auto_execute = (
bool(mcp_tools_with_litellm_proxy)
and LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
)
)
# Prepare parameters for the initial call
initial_call_params = LiteLLM_Proxy_MCP_Handler._prepare_initial_call_params(
call_params=call_params,
should_auto_execute=should_auto_execute
)
#########################################################
# Make initial response API call
# TODO: if should auto-execute is True, then this first response should not be streamed
#########################################################
response = await aresponses(
input=input,
model=model,
tools=all_tools,
previous_response_id=previous_response_id,
**call_params,
**initial_call_params,
)
# Check if we need to auto-execute tool calls (only for non-streaming responses)
verbose_logger.debug("Initial response %s", response)
#########################################################
# Auto-Execute Tools Handling
# If auto-execute tools is True, then we need to execute the tool calls
#########################################################
if (
mcp_tools_with_litellm_proxy
should_auto_execute
and isinstance(response, ResponsesAPIResponse)
and LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
)
): # type: ignore
tool_calls = LiteLLM_Proxy_MCP_Handler._extract_tool_calls_from_response(
response=response
@ -217,20 +284,49 @@ async def aresponses_api_with_mcp(
response=response, tool_results=tool_results, original_input=input
)
# Prepare parameters for follow-up call (restores original stream setting)
follow_up_call_params = LiteLLM_Proxy_MCP_Handler._prepare_follow_up_call_params(
call_params=call_params,
original_stream_setting=stream or False
)
# Create tool execution events for streaming if needed
tool_execution_events = []
if stream:
tool_execution_events = LiteLLM_Proxy_MCP_Handler._create_tool_execution_events(
tool_calls=tool_calls,
tool_results=tool_results
)
final_response = await LiteLLM_Proxy_MCP_Handler._make_follow_up_call(
follow_up_input=follow_up_input,
model=model,
all_tools=all_tools,
response_id=response.id,
**call_params,
**follow_up_call_params,
)
# Add custom output elements to the final response
if isinstance(final_response, ResponsesAPIResponse):
# If streaming and we have tool execution events, wrap the response
if stream and tool_execution_events and (hasattr(final_response, '__aiter__') or hasattr(final_response, '__iter__')):
from litellm.responses.mcp.mcp_streaming_iterator import (
MCPEnhancedStreamingIterator,
)
final_response = MCPEnhancedStreamingIterator(
base_iterator=final_response,
mcp_events=tool_execution_events
)
# Add custom output elements to the final response (for non-streaming)
elif isinstance(final_response, ResponsesAPIResponse):
# Fetch MCP tools again for output elements (without OpenAI transformation)
mcp_tools_for_output = await LiteLLM_Proxy_MCP_Handler._process_mcp_tools_without_openai_transform(
user_api_key_auth=user_api_key_auth,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
)
final_response = (
LiteLLM_Proxy_MCP_Handler._add_mcp_output_elements_to_response(
response=final_response,
mcp_tools_fetched=mcp_tools_fetched,
mcp_tools_fetched=mcp_tools_for_output,
tool_results=tool_results,
)
)
@ -401,13 +497,13 @@ def responses(
Synchronous version of the Responses API.
Uses the synchronous HTTP handler to make requests.
"""
local_vars = locals()
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
LiteLLM_Proxy_MCP_Handler,
)
local_vars = locals()
try:
litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
litellm_logging_obj: LiteLLMLoggingObj = kwargs.pop("litellm_logging_obj") # type: ignore
litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
_is_async = kwargs.pop("aresponses", False) is True
@ -448,7 +544,32 @@ def responses(
#########################################################
if LiteLLM_Proxy_MCP_Handler._should_use_litellm_mcp_gateway(tools=tools):
return aresponses_api_with_mcp(
**local_vars,
input=input,
model=model,
include=include,
instructions=instructions,
max_output_tokens=max_output_tokens,
prompt=prompt,
metadata=metadata,
parallel_tool_calls=parallel_tool_calls,
previous_response_id=previous_response_id,
reasoning=reasoning,
store=store,
background=background,
stream=stream,
temperature=temperature,
text=text,
tool_choice=tool_choice,
tools=tools,
top_p=top_p,
truncation=truncation,
user=user,
extra_headers=extra_headers,
extra_query=extra_query,
extra_body=extra_body,
timeout=timeout,
custom_llm_provider=custom_llm_provider,
**kwargs,
)
# get provider config

View file

@ -1,10 +1,17 @@
from typing import Any, Dict, Iterable, List, Optional, Tuple, Union
from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Tuple, Union
from litellm._logging import verbose_logger
from litellm.responses.main import aresponses
from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator
from litellm.types.llms.openai import ResponsesAPIResponse, ToolParam
if TYPE_CHECKING:
from mcp.types import Tool as MCPTool
else:
MCPTool = Any
LITELLM_PROXY_MCP_SERVER_URL = "litellm_proxy"
LITELLM_PROXY_MCP_SERVER_URL_PREFIX = f"{LITELLM_PROXY_MCP_SERVER_URL}/mcp/"
class LiteLLM_Proxy_MCP_Handler:
"""
@ -20,10 +27,10 @@ class LiteLLM_Proxy_MCP_Handler:
"""
if tools:
for tool in tools:
if (isinstance(tool, dict) and
tool.get("type") == "mcp" and
tool.get("server_url") == "litellm_proxy"):
return True
if isinstance(tool, dict) and tool.get("type") == "mcp":
server_url = tool.get("server_url", "")
if isinstance(server_url, str) and server_url.startswith(LITELLM_PROXY_MCP_SERVER_URL):
return True
return False
@staticmethod
@ -39,23 +46,180 @@ class LiteLLM_Proxy_MCP_Handler:
if tools:
for tool in tools:
if (isinstance(tool, dict) and
tool.get("type") == "mcp" and
tool.get("server_url") == "litellm_proxy"):
mcp_tools_with_litellm_proxy.append(tool)
if isinstance(tool, dict) and tool.get("type") == "mcp":
server_url = tool.get("server_url", "")
if isinstance(server_url, str) and server_url.startswith(LITELLM_PROXY_MCP_SERVER_URL):
mcp_tools_with_litellm_proxy.append(tool)
else:
other_tools.append(tool)
else:
other_tools.append(tool)
return mcp_tools_with_litellm_proxy, other_tools
@staticmethod
async def _get_mcp_tools_from_manager(user_api_key_auth: Any) -> List[Any]:
"""Get available tools from the MCP server manager."""
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
global_mcp_server_manager,
async def _get_mcp_tools_from_manager(
user_api_key_auth: Any,
mcp_tools_with_litellm_proxy: Optional[Iterable[ToolParam]],
) -> List[MCPTool]:
"""
Get available tools from the MCP server manager.
Args:
user_api_key_auth: User authentication info for access control
mcp_tools_with_litellm_proxy: ToolParam objects with server_url starting with "litellm_proxy"
"""
from litellm.proxy._experimental.mcp_server.server import (
_get_tools_from_mcp_servers,
)
mcp_servers: List[str] = []
if mcp_tools_with_litellm_proxy:
for _tool in mcp_tools_with_litellm_proxy:
# if user specifies servers as server_url: litellm_proxy/mcp/zapier,github then return zapier,github
server_url = _tool.get("server_url", "") if isinstance(_tool, dict) else ""
if isinstance(server_url, str) and server_url.startswith(LITELLM_PROXY_MCP_SERVER_URL_PREFIX):
mcp_servers.append(server_url.split("/")[-1])
return await _get_tools_from_mcp_servers(
user_api_key_auth=user_api_key_auth,
mcp_auth_header=None,
mcp_servers=mcp_servers,
mcp_server_auth_headers=None,
mcp_protocol_version=None,
)
@staticmethod
def _deduplicate_mcp_tools(mcp_tools: List[Any]) -> List[Any]:
"""
Deduplicate MCP tools by name, keeping the first occurrence of each tool.
Args:
mcp_tools: List of MCP tools that may contain duplicates
Returns:
List of deduplicated MCP tools
"""
seen_names = set()
deduplicated_tools = []
for tool in mcp_tools:
tool_name = getattr(tool, 'name', None) if hasattr(tool, 'name') else tool.get('name') if isinstance(tool, dict) else None
if tool_name and tool_name not in seen_names:
seen_names.add(tool_name)
deduplicated_tools.append(tool)
return deduplicated_tools
@staticmethod
def _filter_mcp_tools_by_allowed_tools(
mcp_tools: List[Any],
mcp_tools_with_litellm_proxy: List[ToolParam]
) -> List[Any]:
"""Filter MCP tools based on allowed_tools parameter from the original tool configs."""
# Collect all allowed tool names from all MCP tool configs
allowed_tool_names = set()
for tool_config in mcp_tools_with_litellm_proxy:
if isinstance(tool_config, dict) and "allowed_tools" in tool_config:
allowed_tools = tool_config.get("allowed_tools", [])
if isinstance(allowed_tools, list):
allowed_tool_names.update(allowed_tools)
# If no allowed_tools specified, return all tools
if not allowed_tool_names:
return mcp_tools
# Filter tools based on allowed names
filtered_tools = []
for mcp_tool in mcp_tools:
tool_name = getattr(mcp_tool, 'name', None) if hasattr(mcp_tool, 'name') else mcp_tool.get('name') if isinstance(mcp_tool, dict) else None
if tool_name and tool_name in allowed_tool_names:
filtered_tools.append(mcp_tool)
return filtered_tools
@staticmethod
async def _process_mcp_tools_to_openai_format(
user_api_key_auth: Any,
mcp_tools_with_litellm_proxy: List[ToolParam]
) -> List[Any]:
"""
Centralized method to process MCP tools through the complete pipeline:
1. Fetch tools from MCP manager
2. Filter based on allowed_tools parameter
3. Deduplicate tools by name
4. Transform to OpenAI format
Args:
user_api_key_auth: User authentication info for access control
mcp_tools_with_litellm_proxy: ToolParam objects with server_url starting with "litellm_proxy"
Returns:
List of tools in OpenAI format ready to be sent to the LLM
"""
if not mcp_tools_with_litellm_proxy:
return []
# Step 1: Fetch MCP tools from manager
mcp_tools_fetched = await LiteLLM_Proxy_MCP_Handler._get_mcp_tools_from_manager(
user_api_key_auth=user_api_key_auth,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
)
return await global_mcp_server_manager.list_tools(user_api_key_auth=user_api_key_auth)
# Step 2: Filter tools based on allowed_tools parameter
filtered_mcp_tools = LiteLLM_Proxy_MCP_Handler._filter_mcp_tools_by_allowed_tools(
mcp_tools=mcp_tools_fetched,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
)
# Step 3: Deduplicate tools after filtering
deduplicated_mcp_tools = LiteLLM_Proxy_MCP_Handler._deduplicate_mcp_tools(
filtered_mcp_tools
)
# Step 4: Transform to OpenAI format
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(
deduplicated_mcp_tools
)
return openai_tools
@staticmethod
async def _process_mcp_tools_without_openai_transform(
user_api_key_auth: Any,
mcp_tools_with_litellm_proxy: List[ToolParam]
) -> List[Any]:
"""
Process MCP tools through filtering and deduplication pipeline without OpenAI transformation.
This is useful for cases where we need the original MCP tool objects (e.g., for events).
Args:
user_api_key_auth: User authentication info for access control
mcp_tools_with_litellm_proxy: ToolParam objects with server_url starting with "litellm_proxy"
Returns:
List of filtered and deduplicated MCP tools in their original format
"""
if not mcp_tools_with_litellm_proxy:
return []
# Step 1: Fetch MCP tools from manager
mcp_tools_fetched = await LiteLLM_Proxy_MCP_Handler._get_mcp_tools_from_manager(
user_api_key_auth=user_api_key_auth,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
)
# Step 2: Filter tools based on allowed_tools parameter
filtered_mcp_tools = LiteLLM_Proxy_MCP_Handler._filter_mcp_tools_by_allowed_tools(
mcp_tools=mcp_tools_fetched,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
)
# Step 3: Deduplicate tools after filtering
deduplicated_mcp_tools = LiteLLM_Proxy_MCP_Handler._deduplicate_mcp_tools(
filtered_mcp_tools
)
return deduplicated_mcp_tools
@staticmethod
def _transform_mcp_tools_to_openai(mcp_tools: List[Any]) -> List[Any]:
@ -178,11 +342,12 @@ class LiteLLM_Proxy_MCP_Handler:
user_api_key_auth: Any
) -> List[Dict[str, Any]]:
"""Execute tool calls and return results."""
from fastapi import HTTPException
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
global_mcp_server_manager,
)
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
from fastapi import HTTPException
tool_results = []
tool_call_id: Optional[str] = None
@ -331,6 +496,170 @@ class LiteLLM_Proxy_MCP_Handler:
**call_params
)
@staticmethod
def _create_mcp_streaming_response(
input: Union[str, Any],
model: str,
all_tools: Optional[List[Any]],
mcp_tools_with_litellm_proxy: List[Any],
mcp_discovery_events: List[Any],
call_params: Dict[str, Any],
previous_response_id: Optional[str],
**kwargs
) -> Any:
"""
Create MCP enhanced streaming response that handles the full MCP workflow.
This creates a streaming iterator that:
1. Immediately emits MCP discovery events
2. Makes the LLM call and streams the response
3. Handles tool execution and follow-up calls
"""
from litellm.responses.mcp.mcp_streaming_iterator import (
MCPEnhancedStreamingIterator,
)
# Build the complete request parameters by merging all sources
request_params = LiteLLM_Proxy_MCP_Handler._build_request_params(
input=input,
model=model,
all_tools=all_tools,
call_params=call_params,
previous_response_id=previous_response_id,
**kwargs
)
# Create the enhanced streaming iterator that will handle everything
return MCPEnhancedStreamingIterator(
base_iterator=None, # Will be created internally
mcp_events=mcp_discovery_events, # Pre-generated MCP discovery events
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
user_api_key_auth=kwargs.get("user_api_key_auth"),
original_request_params=request_params
)
@staticmethod
def _build_request_params(
input: Union[str, Any],
model: str,
all_tools: Optional[List[Any]],
call_params: Dict[str, Any],
previous_response_id: Optional[str],
**kwargs
) -> Dict[str, Any]:
"""
Build a clean request parameters dictionary for MCP streaming.
Combines input, model, tools with call_params and additional kwargs
in a clean, maintainable way.
"""
# Start with the core required parameters
request_params = {
'input': input,
'model': model,
'tools': all_tools,
}
# Add previous_response_id if provided
if previous_response_id is not None:
request_params['previous_response_id'] = previous_response_id
# Merge in all call_params (which contains most of the API parameters)
request_params.update(call_params)
# Merge in any additional kwargs
request_params.update(kwargs)
return request_params
@staticmethod
def _create_tool_execution_events(
tool_calls: List[Any],
tool_results: List[Dict[str, Any]]
) -> List[Any]:
"""
Create MCP tool execution events for streaming.
Args:
tool_calls: List of tool calls from the LLM response
tool_results: List of tool execution results
Returns:
List of MCP tool execution events for streaming
"""
import uuid
from litellm.responses.mcp.mcp_streaming_iterator import create_mcp_call_events
tool_execution_events: List[Any] = []
# Create events for each tool execution
for tool_result in tool_results:
tool_call_id = tool_result.get("tool_call_id", "unknown")
result_text = tool_result.get("result", "")
# Extract tool name and arguments from tool calls
tool_name = "unknown"
tool_arguments = "{}"
for tool_call in tool_calls:
name, args, call_id = LiteLLM_Proxy_MCP_Handler._extract_tool_call_details(tool_call)
if call_id == tool_call_id:
tool_name = name or "unknown"
tool_arguments = args or "{}"
break
execution_events = create_mcp_call_events(
tool_name=tool_name,
tool_call_id=tool_call_id,
arguments=tool_arguments, # Use actual arguments
result=result_text,
base_item_id=f"mcp_{uuid.uuid4().hex[:8]}", # Unique ID for each tool call
sequence_start=len(tool_execution_events) + 1
)
tool_execution_events.extend(execution_events)
return tool_execution_events
@staticmethod
def _prepare_initial_call_params(
call_params: Dict[str, Any],
should_auto_execute: bool
) -> Dict[str, Any]:
"""
Prepare call parameters for the initial LLM call.
For auto-execute scenarios, we need to disable streaming for the initial call
so we can process the tool calls before streaming the final response.
"""
initial_params = call_params.copy()
if should_auto_execute:
# Disable streaming for initial call when auto-executing tools
initial_params["stream"] = False
return initial_params
@staticmethod
def _prepare_follow_up_call_params(
call_params: Dict[str, Any],
original_stream_setting: bool
) -> Dict[str, Any]:
"""
Prepare call parameters for the follow-up LLM call after tool execution.
Restores the original streaming setting and removes tool_choice since
we're now providing tool results, not requesting tool calls.
"""
follow_up_params = call_params.copy()
# Restore original streaming setting for follow-up call
follow_up_params["stream"] = original_stream_setting
# Remove tool_choice since we're providing results, not requesting tool calls
follow_up_params.pop("tool_choice", None)
return follow_up_params
@staticmethod
def _add_mcp_output_elements_to_response(
response: ResponsesAPIResponse,

View file

@ -0,0 +1,600 @@
import uuid
from typing import (
TYPE_CHECKING,
Any,
Dict,
List,
Optional,
Union,
cast,
)
from litellm._logging import verbose_logger
from litellm.responses.streaming_iterator import (
BaseResponsesAPIStreamingIterator,
)
from litellm.types.llms.openai import (
MCPCallArgumentsDeltaEvent,
MCPCallArgumentsDoneEvent,
MCPCallCompletedEvent,
MCPCallFailedEvent,
MCPCallInProgressEvent,
MCPListToolsCompletedEvent,
MCPListToolsFailedEvent,
MCPListToolsInProgressEvent,
ResponsesAPIResponse,
ResponsesAPIStreamEvents,
ResponsesAPIStreamingResponse,
ToolParam,
)
if TYPE_CHECKING:
from mcp.types import Tool as MCPTool
else:
MCPTool = Any
async def create_mcp_list_tools_events(
mcp_tools_with_litellm_proxy: List[ToolParam],
user_api_key_auth: Any,
base_item_id: str,
pre_processed_mcp_tools: List[Any]
) -> List[ResponsesAPIStreamingResponse]:
"""Create MCP discovery events using pre-processed tools from the parent"""
events: List[ResponsesAPIStreamingResponse] = []
try:
# Extract MCP server names
mcp_servers = []
for tool in mcp_tools_with_litellm_proxy:
if isinstance(tool, dict) and "server_url" in tool:
server_url = tool.get("server_url")
if isinstance(server_url, str) and server_url.startswith("litellm_proxy/mcp/"):
server_name = server_url.split("/")[-1]
mcp_servers.append(server_name)
# Emit list tools in progress event
in_progress_event = MCPListToolsInProgressEvent(
type=ResponsesAPIStreamEvents.MCP_LIST_TOOLS_IN_PROGRESS,
sequence_number=1,
output_index=0,
item_id=base_item_id,
)
events.append(in_progress_event)
# Use the pre-processed MCP tools that were already fetched, filtered, and deduplicated by the parent
filtered_mcp_tools = pre_processed_mcp_tools
# Convert tools to dict format for the event
mcp_tools_dict = []
for tool in filtered_mcp_tools:
if hasattr(tool, 'model_dump') and callable(getattr(tool, 'model_dump')):
mcp_tools_dict.append(tool.model_dump())
elif hasattr(tool, '__dict__'):
mcp_tools_dict.append(tool.__dict__)
else:
mcp_tools_dict.append({"name": getattr(tool, 'name', str(tool))})
# Emit list tools completed event
completed_event = MCPListToolsCompletedEvent(
type=ResponsesAPIStreamEvents.MCP_LIST_TOOLS_COMPLETED,
sequence_number=2,
output_index=0,
item_id=base_item_id,
)
events.append(completed_event)
# Add output_item.done event with the actual tools list (matching OpenAI format)
from litellm.types.llms.openai import OutputItemDoneEvent
# Extract server label from the first MCP tool config
server_label = ""
if mcp_tools_with_litellm_proxy:
first_tool = mcp_tools_with_litellm_proxy[0]
if isinstance(first_tool, dict):
server_label_value = first_tool.get("server_label", "")
server_label = str(server_label_value) if server_label_value is not None else ""
# Format tools for OpenAI output_item.done format
formatted_tools = []
for tool in filtered_mcp_tools:
tool_dict = {
"name": getattr(tool, 'name', 'unknown'),
"description": getattr(tool, 'description', ''),
"annotations": {"read_only": False},
}
# Add input_schema if available
if hasattr(tool, 'inputSchema'):
tool_dict["input_schema"] = getattr(tool, 'inputSchema')
elif hasattr(tool, 'input_schema'):
tool_dict["input_schema"] = getattr(tool, 'input_schema')
formatted_tools.append(tool_dict)
# Create the output_item.done event with MCP tools list
output_item_done_event = OutputItemDoneEvent(
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
output_index=0,
item={
"id": base_item_id,
"type": "mcp_list_tools",
"server_label": server_label,
"tools": formatted_tools
}
)
events.append(output_item_done_event)
verbose_logger.debug(f"Created {len(events)} MCP discovery events")
except Exception as e:
verbose_logger.error(f"Error creating MCP list tools events: {e}")
import traceback
traceback.print_exc()
# Emit failed event on error
failed_event = MCPListToolsFailedEvent(
type=ResponsesAPIStreamEvents.MCP_LIST_TOOLS_FAILED,
sequence_number=2,
output_index=0,
item_id=base_item_id,
)
events.append(failed_event)
# Still emit output_item.done event even on failure (with empty tools list)
from litellm.types.llms.openai import OutputItemDoneEvent
output_item_done_event = OutputItemDoneEvent(
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
output_index=0,
item={
"id": base_item_id,
"type": "mcp_list_tools",
"server_label": "",
"tools": []
}
)
events.append(output_item_done_event)
return events
def create_mcp_call_events(
tool_name: str,
tool_call_id: str,
arguments: str,
result: Optional[str] = None,
base_item_id: Optional[str] = None,
sequence_start: int = 1
) -> List[ResponsesAPIStreamingResponse]:
"""Create MCP call events following OpenAI's specification"""
events: List[ResponsesAPIStreamingResponse] = []
item_id = base_item_id or f"mcp_{uuid.uuid4().hex[:8]}"
# MCP call in progress event
in_progress_event = MCPCallInProgressEvent(
type=ResponsesAPIStreamEvents.MCP_CALL_IN_PROGRESS,
sequence_number=sequence_start,
output_index=0,
item_id=item_id,
)
events.append(in_progress_event)
# MCP call arguments delta event (streaming the arguments)
arguments_delta_event = MCPCallArgumentsDeltaEvent(
type=ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DELTA,
output_index=0,
item_id=item_id,
delta=arguments, # JSON string with arguments
sequence_number=sequence_start + 1,
)
events.append(arguments_delta_event)
# MCP call arguments done event
arguments_done_event = MCPCallArgumentsDoneEvent(
type=ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DONE,
output_index=0,
item_id=item_id,
arguments=arguments, # Complete JSON string with finalized arguments
sequence_number=sequence_start + 2,
)
events.append(arguments_done_event)
# MCP call completed event (or failed if result indicates failure)
if result is not None:
completed_event = MCPCallCompletedEvent(
type=ResponsesAPIStreamEvents.MCP_CALL_COMPLETED,
sequence_number=sequence_start + 3,
item_id=item_id,
output_index=0,
)
events.append(completed_event)
# Add output_item.done event with the tool call result
from litellm.types.llms.openai import OutputItemDoneEvent
output_item_done_event = OutputItemDoneEvent(
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
output_index=0,
item={
"id": item_id,
"type": "mcp_call",
"approval_request_id": f"mcpr_{uuid.uuid4().hex[:8]}",
"arguments": arguments,
"error": None,
"name": tool_name,
"output": result,
"server_label": "litellm"
},
)
events.append(output_item_done_event)
else:
failed_event = MCPCallFailedEvent(
type=ResponsesAPIStreamEvents.MCP_CALL_FAILED,
sequence_number=sequence_start + 3,
item_id=item_id,
output_index=0,
)
events.append(failed_event)
return events
class MCPEnhancedStreamingIterator(BaseResponsesAPIStreamingIterator):
"""
A complete MCP streaming iterator that handles the entire flow:
1. Immediately emits MCP discovery events
2. Makes the first LLM call and streams its response
3. Handles tool execution and follow-up calls for auto-execute tools
4. Emits tool execution events in the stream
"""
def __init__(
self,
base_iterator: Any, # Can be None - will be created internally
mcp_events: List[ResponsesAPIStreamingResponse],
mcp_tools_with_litellm_proxy: Optional[List[Any]] = None,
user_api_key_auth: Any = None,
original_request_params: Optional[Dict[str, Any]] = None
):
# MCP setup
self.mcp_tools_with_litellm_proxy = mcp_tools_with_litellm_proxy or []
self.user_api_key_auth = user_api_key_auth
self.original_request_params = original_request_params or {}
self.should_auto_execute = self._should_auto_execute_tools()
# Streaming state management
self.phase = "mcp_discovery" # mcp_discovery -> initial_response -> tool_execution -> follow_up_response -> finished
self.finished = False
# Event queues and generation flags
self.mcp_discovery_events: List[ResponsesAPIStreamingResponse] = mcp_events # Pre-generated MCP discovery events
self.tool_execution_events: List[ResponsesAPIStreamingResponse] = []
self.mcp_discovery_generated = True # Events are already generated
self.mcp_events = mcp_events # Store the initial MCP events for backward compatibility
# Iterator references
self.base_iterator: Optional[Union[Any, ResponsesAPIResponse]] = base_iterator # Will be created when needed
self.follow_up_iterator: Optional[Any] = None
# Response collection for tool execution
self.collected_response: Optional[ResponsesAPIResponse] = None
# Set up model metadata (will be updated when we get the real iterator)
self.model = self.original_request_params.get('model', 'unknown')
self.litellm_metadata = {}
self.custom_llm_provider = self.original_request_params.get('custom_llm_provider', None)
# Mark as async iterator
self.is_async = True
def _should_auto_execute_tools(self) -> bool:
"""Check if tools should be auto-executed"""
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
LiteLLM_Proxy_MCP_Handler,
)
return LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
self.mcp_tools_with_litellm_proxy
)
def __aiter__(self):
return self
async def __anext__(self) -> ResponsesAPIStreamingResponse:
"""
Phase-based streaming:
1. mcp_discovery - Emit MCP discovery events
2. initial_response - Stream the first LLM response
3. tool_execution - Emit tool execution events
4. follow_up_response - Stream the follow-up response
5. finished - End iteration
"""
# Phase 1: MCP Discovery Events
if self.phase == "mcp_discovery":
# Generate MCP discovery events if not already done
# MCP discovery events are already generated and available
# Emit MCP discovery events
if self.mcp_discovery_events:
return self.mcp_discovery_events.pop(0)
# All MCP discovery events emitted, move to next phase
verbose_logger.debug("MCP discovery phase complete, transitioning to initial_response")
self.phase = "initial_response"
await self._create_initial_response_iterator()
# Fall through to process the initial response immediately
# Phase 2: Initial Response Stream
if self.phase == "initial_response":
if self.base_iterator:
# Check if base_iterator is actually iterable
if hasattr(self.base_iterator, '__anext__'):
try:
chunk = await cast(Any, self.base_iterator).__anext__() # type: ignore[attr-defined]
# If auto-execution is enabled, check for completed responses
if self.should_auto_execute and self._is_response_completed(chunk):
# Collect the response for tool execution
response_obj = getattr(chunk, 'response', None)
if isinstance(response_obj, ResponsesAPIResponse):
self.collected_response = response_obj
# Move to tool execution phase after emitting this chunk
self.phase = "tool_execution"
await self._generate_tool_execution_events()
return chunk
except StopAsyncIteration:
# Initial response ended, move to next phase
if self.should_auto_execute and self.collected_response:
self.phase = "tool_execution"
await self._generate_tool_execution_events()
else:
self.phase = "finished"
raise
else:
# base_iterator is not async iterable (likely a ResponsesAPIResponse)
# Collect it for tool execution if needed
if self.should_auto_execute and isinstance(self.base_iterator, ResponsesAPIResponse):
self.collected_response = self.base_iterator
self.phase = "tool_execution"
await self._generate_tool_execution_events()
else:
self.phase = "finished"
raise StopAsyncIteration
# Phase 3: Tool Execution Events
if self.phase == "tool_execution":
# Emit any queued tool execution events
if self.tool_execution_events:
return self.tool_execution_events.pop(0)
# Move to follow-up response phase
self.phase = "follow_up_response"
await self._create_follow_up_iterator()
# Phase 4: Follow-up Response Stream
if self.phase == "follow_up_response":
if self.follow_up_iterator:
try:
return await cast(Any, self.follow_up_iterator).__anext__() # type: ignore[attr-defined]
except StopAsyncIteration:
self.phase = "finished"
raise
else:
self.phase = "finished"
raise StopAsyncIteration
# Phase 5: Finished
if self.phase == "finished":
raise StopAsyncIteration
# Should not reach here
raise StopAsyncIteration
def _is_response_completed(self, chunk: ResponsesAPIStreamingResponse) -> bool:
"""Check if this chunk indicates the response is completed"""
from litellm.types.llms.openai import ResponsesAPIStreamEvents
return getattr(chunk, 'type', None) == ResponsesAPIStreamEvents.RESPONSE_COMPLETED
async def _create_initial_response_iterator(self) -> None:
"""Create the initial response iterator by making the first LLM call"""
try:
# Import the core aresponses function that doesn't have MCP logic
from litellm.responses.main import aresponses
# Make the initial response API call - but avoid the MCP wrapper
params = self.original_request_params.copy()
params['stream'] = True # Ensure streaming
# Use the pre-fetched all_tools from original_request_params (no re-processing needed)
params_for_llm = {}
for key, value in params.items():
params_for_llm[key] = value # Copy all params as-is since tools are already processed
tools_count = len(params_for_llm.get('tools', []))
verbose_logger.debug(f"Making LLM call with {tools_count} tools")
response = await aresponses(**params_for_llm)
# Set the base iterator
if hasattr(response, '__aiter__') or hasattr(response, '__iter__'):
self.base_iterator = response
# Copy metadata from the real iterator
self.model = getattr(response, 'model', self.model)
self.litellm_metadata = getattr(response, 'litellm_metadata', {})
self.custom_llm_provider = getattr(response, 'custom_llm_provider', self.custom_llm_provider)
verbose_logger.debug(f"Created base iterator: {type(self.base_iterator)}")
else:
# Non-streaming response - this shouldn't happen but handle it
verbose_logger.warning(f"Got non-streaming response: {type(response)}")
self.base_iterator = None
self.phase = "finished"
except Exception as e:
verbose_logger.error(f"Error creating initial response iterator: {e}")
import traceback
traceback.print_exc()
self.base_iterator = None
self.phase = "finished"
async def _generate_tool_execution_events(self) -> None:
"""Generate tool execution events and execute tools"""
if not self.collected_response:
return
import uuid
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
LiteLLM_Proxy_MCP_Handler,
)
try:
# Extract tool calls from the response
if self.collected_response is not None:
tool_calls = LiteLLM_Proxy_MCP_Handler._extract_tool_calls_from_response(self.collected_response) # type: ignore[arg-type]
else:
tool_calls = []
if not tool_calls:
return
for tool_call in tool_calls:
tool_name, tool_arguments, tool_call_id = LiteLLM_Proxy_MCP_Handler._extract_tool_call_details(tool_call)
if tool_name and tool_call_id:
# Create MCP call events for this tool execution
call_events = create_mcp_call_events(
tool_name=tool_name,
tool_call_id=tool_call_id,
arguments=tool_arguments or "{}", # JSON string with arguments
result=None, # Will be set after execution
base_item_id=f"mcp_{uuid.uuid4().hex[:8]}",
sequence_start=len(self.tool_execution_events) + 1
)
# Add the in_progress and arguments events (not the completed event yet)
self.tool_execution_events.extend(call_events[:-1])
# Execute the tools
tool_results = await LiteLLM_Proxy_MCP_Handler._execute_tool_calls(
tool_calls=tool_calls,
user_api_key_auth=self.user_api_key_auth
)
# Create completion events and output_item.done events for tool execution
for tool_result in tool_results:
tool_call_id = tool_result.get("tool_call_id", "unknown")
result_text = tool_result.get("result", "")
# Find matching tool name and arguments
tool_name = "unknown"
tool_arguments = "{}"
for tool_call in tool_calls:
name, args, call_id = LiteLLM_Proxy_MCP_Handler._extract_tool_call_details(tool_call)
if call_id == tool_call_id:
tool_name = name or "unknown"
tool_arguments = args or "{}"
break
item_id = f"mcp_{uuid.uuid4().hex[:8]}"
# Create the completion event
completed_event = MCPCallCompletedEvent(
type=ResponsesAPIStreamEvents.MCP_CALL_COMPLETED,
sequence_number=len(self.tool_execution_events) + 1,
item_id=item_id,
output_index=0,
)
self.tool_execution_events.append(completed_event)
# Create output_item.done event with the tool call result
from litellm.types.llms.openai import OutputItemDoneEvent
output_item_done_event = OutputItemDoneEvent(
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
output_index=0,
item={
"id": item_id,
"type": "mcp_call",
"approval_request_id": f"mcpr_{uuid.uuid4().hex[:8]}",
"arguments": tool_arguments,
"error": None,
"name": tool_name,
"output": result_text,
"server_label": "litellm" # or extract from tool config
},
)
self.tool_execution_events.append(output_item_done_event)
# Store tool results for follow-up call
self.tool_results = tool_results
except Exception as e:
verbose_logger.error(f"Error in tool execution: {e}")
import traceback
traceback.print_exc()
self.tool_results = []
async def _create_follow_up_iterator(self) -> None:
"""Create the follow-up response iterator with tool results"""
if not self.collected_response or not hasattr(self, 'tool_results'):
return
from litellm.responses.main import aresponses
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
LiteLLM_Proxy_MCP_Handler,
)
try:
# Create follow-up input
if self.collected_response is not None:
follow_up_input = LiteLLM_Proxy_MCP_Handler._create_follow_up_input(
response=self.collected_response, # type: ignore[arg-type]
tool_results=self.tool_results,
original_input=self.original_request_params.get('input')
)
# Make follow-up call with streaming
follow_up_params = self.original_request_params.copy()
follow_up_params.update({
'input': follow_up_input,
'previous_response_id': self.collected_response.id, # type: ignore[attr-defined]
'stream': True
})
else:
return
# Remove tool_choice to avoid forcing more tool calls
follow_up_params.pop('tool_choice', None)
follow_up_response = await aresponses(**follow_up_params)
# Set up the follow-up iterator
if hasattr(follow_up_response, '__aiter__'):
self.follow_up_iterator = follow_up_response
except Exception as e:
verbose_logger.error(f"Error creating follow-up iterator: {e}")
import traceback
traceback.print_exc()
self.follow_up_iterator = None
def __iter__(self):
return self
def __next__(self) -> ResponsesAPIStreamingResponse:
# First, emit any queued MCP events
if self.mcp_events: # type: ignore[attr-defined]
return self.mcp_events.pop(0) # type: ignore[attr-defined]
# Then delegate to the base iterator
if not self.is_async:
try:
if self.base_iterator and hasattr(self.base_iterator, '__next__'):
return next(cast(Any, self.base_iterator)) # type: ignore[arg-type]
else:
raise StopIteration
except StopIteration:
self.finished = True
raise
else:
raise RuntimeError("Cannot use sync iteration on async iterator")

View file

@ -4031,7 +4031,7 @@ class Router:
else:
raise
verbose_router_logger.info(
verbose_router_logger.debug(
f"Retrying request with num_retries: {num_retries}"
)
# decides how long to sleep before retry
@ -4681,7 +4681,7 @@ class Router:
elif self._has_default_fallbacks(): # default fallbacks set
return True
verbose_router_logger.info(
verbose_router_logger.debug(
"Content Policy Error occurred. No available fallbacks. Returning original response. model={}, content_policy_fallbacks={}".format(
model, content_policy_fallbacks
)

View file

@ -106,7 +106,7 @@ class HashicorpSecretManager(BaseSecretManager):
resp.raise_for_status()
token = resp.json()["auth"]["client_token"]
_lease_duration = resp.json()["auth"]["lease_duration"]
verbose_logger.info("Successfully obtained Vault token via TLS cert auth.")
verbose_logger.debug("Successfully obtained Vault token via TLS cert auth.")
self.cache.set_cache(
key="hcp_vault_token", value=token, ttl=_lease_duration
)

View file

@ -51,7 +51,7 @@ AllDatabricksContentValues = Union[str, List[AllDatabricksContentListValues]]
class DatabricksFunction(TypedDict, total=False):
name: Required[str]
description: dict
description: Union[dict, str]
parameters: dict
strict: bool

View file

@ -121,7 +121,7 @@ class OCICompletionTokenDetails(BaseModel):
reasoningTokens: int
class OCIPropmtTokensDetails(BaseModel):
class OCIPromptTokensDetails(BaseModel):
"""Prompt token details in the OCI response."""
cachedTokens: int
@ -129,12 +129,12 @@ class OCIPropmtTokensDetails(BaseModel):
class OCIResponseUsage(BaseModel):
"""Token usage in the OCI response."""
promptTokens: int
completionTokens: int
totalTokens: int
completionTokensDetails: OCICompletionTokenDetails
promptTokensDetails: OCIPropmtTokensDetails
completionTokensDetails: Optional[OCICompletionTokenDetails] = None
promptTokensDetails: Optional[OCIPromptTokensDetails] = None
class OCIResponseChoice(BaseModel):

View file

@ -1112,6 +1112,16 @@ class ResponsesAPIStreamEvents(str, Enum):
WEB_SEARCH_CALL_SEARCHING = "response.web_search_call.searching"
WEB_SEARCH_CALL_COMPLETED = "response.web_search_call.completed"
# MCP events - matching OpenAI's official specification
MCP_LIST_TOOLS_IN_PROGRESS = "response.mcp_list_tools.in_progress"
MCP_LIST_TOOLS_COMPLETED = "response.mcp_list_tools.completed"
MCP_LIST_TOOLS_FAILED = "response.mcp_list_tools.failed"
MCP_CALL_IN_PROGRESS = "response.mcp_call.in_progress"
MCP_CALL_ARGUMENTS_DELTA = "response.mcp_call_arguments.delta"
MCP_CALL_ARGUMENTS_DONE = "response.mcp_call_arguments.done"
MCP_CALL_COMPLETED = "response.mcp_call.completed"
MCP_CALL_FAILED = "response.mcp_call.failed"
# Error event
ERROR = "error"
@ -1275,6 +1285,66 @@ class WebSearchCallCompletedEvent(BaseLiteLLMOpenAIResponseObject):
item_id: str
# MCP List Tools Events
class MCPListToolsInProgressEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_LIST_TOOLS_IN_PROGRESS]
sequence_number: int
output_index: int
item_id: str
class MCPListToolsCompletedEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_LIST_TOOLS_COMPLETED]
sequence_number: int
output_index: int
item_id: str
class MCPListToolsFailedEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_LIST_TOOLS_FAILED]
sequence_number: int
output_index: int
item_id: str
# MCP Call Events
class MCPCallInProgressEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_IN_PROGRESS]
sequence_number: int
output_index: int
item_id: str
class MCPCallArgumentsDeltaEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DELTA]
output_index: int
item_id: str
delta: str # JSON string containing partial update to arguments
sequence_number: int
class MCPCallArgumentsDoneEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DONE]
output_index: int
item_id: str
arguments: str # JSON string containing finalized arguments
sequence_number: int
class MCPCallCompletedEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_COMPLETED]
sequence_number: int
item_id: str
output_index: int
class MCPCallFailedEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_FAILED]
sequence_number: int
item_id: str
output_index: int
class ErrorEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.ERROR]
code: Optional[str]
@ -1315,6 +1385,14 @@ ResponsesAPIStreamingResponse = Annotated[
WebSearchCallInProgressEvent,
WebSearchCallSearchingEvent,
WebSearchCallCompletedEvent,
MCPListToolsInProgressEvent,
MCPListToolsCompletedEvent,
MCPListToolsFailedEvent,
MCPCallInProgressEvent,
MCPCallArgumentsDeltaEvent,
MCPCallArgumentsDoneEvent,
MCPCallCompletedEvent,
MCPCallFailedEvent,
ErrorEvent,
GenericEvent,
],

View file

@ -281,6 +281,7 @@ class RequestBody(TypedDict, total=False):
safetySettings: List[SafetSettingsConfig]
generationConfig: GenerationConfig
cachedContent: str
labels: Dict[str, str]
speechConfig: SpeechConfig

View file

@ -31,13 +31,20 @@ class MCPAuth(str, enum.Enum):
api_key = "api_key"
bearer_token = "bearer_token"
basic = "basic"
authorization = "authorization"
# MCP Literals
MCPTransportType = Literal[MCPTransport.sse, MCPTransport.http, MCPTransport.stdio]
MCPSpecVersionType = Literal[MCPSpecVersion.nov_2024, MCPSpecVersion.mar_2025, MCPSpecVersion.jun_2025]
MCPAuthType = Optional[
Literal[MCPAuth.none, MCPAuth.api_key, MCPAuth.bearer_token, MCPAuth.basic]
Literal[
MCPAuth.none,
MCPAuth.api_key,
MCPAuth.bearer_token,
MCPAuth.basic,
MCPAuth.authorization,
]
]

View file

@ -1995,7 +1995,7 @@ class StandardLoggingGuardrailInformation(TypedDict, total=False):
]
guardrail_request: Optional[dict]
guardrail_response: Optional[Union[dict, str, List[dict]]]
guardrail_status: Literal["success", "failure"]
guardrail_status: Literal["success", "failure","blocked"]
start_time: Optional[float]
end_time: Optional[float]
duration: Optional[float]

View file

@ -2437,6 +2437,7 @@ def get_optional_params_transcription(
"prompt": None,
"response_format": None,
"temperature": None, # openai defaults this to 0
"timestamp_granularities": None
}
non_default_params = {

View file

@ -13123,6 +13123,7 @@
"mode": "chat",
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_function_calling": true,
"supports_reasoning": true
},
"openai.gpt-oss-120b-1:0": {
@ -13135,6 +13136,7 @@
"mode": "chat",
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_function_calling": true,
"supports_reasoning": true
},
"anthropic.claude-opus-4-1-20250805-v1:0": {
@ -13877,136 +13879,6 @@
"litellm_provider": "bedrock",
"mode": "chat"
},
"anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-east-1/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-west-2/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0455,
"output_cost_per_second": 0.0455,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.02527,
"output_cost_per_second": 0.02527,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/eu-central-1/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_token": 8e-06,
"output_cost_per_token": 2.4e-05,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0415,
"output_cost_per_second": 0.0415,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.02305,
"output_cost_per_second": 0.02305,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0175,
"output_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.00972,
"output_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.0175,
"output_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2": {
"max_tokens": 8191,
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"input_cost_per_second": 0.00972,
"output_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"mode": "chat",
"supports_tool_choice": true
},
"anthropic.claude-v2:1": {
"max_tokens": 8191,
"max_input_tokens": 100000,
@ -15245,7 +15117,7 @@
"mode": "chat",
"source": "https://www.together.ai/models/gpt-oss-120b"
},
"together_ai/OpenAI/gpt-oss-20B": {
"together_ai/openai/gpt-oss-20b": {
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"max_input_tokens": 128000,
@ -15517,16 +15389,6 @@
"litellm_provider": "ollama",
"mode": "completion"
},
"deepinfra/Austism/chronos-hermes-13b-v2": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.3e-07,
"output_cost_per_token": 1.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Gryphe/MythoMax-L2-13b": {
"max_tokens": 4096,
"max_input_tokens": 4096,
@ -15537,26 +15399,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Gryphe/MythoMax-L2-13b-turbo": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.3e-07,
"output_cost_per_token": 1.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1e-07,
"output_cost_per_token": 1e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -15575,78 +15417,18 @@
"output_cost_per_token": 2.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/NovaSky-AI/Sky-T1-32B-Preview": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 1.2e-07,
"output_cost_per_token": 1.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Phind/Phind-CodeLlama-34B-v2": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 6e-07,
"output_cost_per_token": 6e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/QVQ-72B-Preview": {
"max_tokens": 32000,
"max_input_tokens": 32000,
"max_output_tokens": 32000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/QwQ-32B": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/QwQ-32B-Preview": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 1.2e-07,
"output_cost_per_token": 1.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2-72B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 3.5e-07,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/Qwen2-7B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Qwen/Qwen2.5-72B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
@ -15667,26 +15449,6 @@
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 6e-08,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2.5-Coder-7B": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 2.5e-08,
"output_cost_per_token": 5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": {
"max_tokens": 128000,
"max_input_tokens": 128000,
@ -15773,30 +15535,11 @@
"max_output_tokens": 262144,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"cache_read_input_token_cost": 2.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/Sao10K/L3-70B-Euryale-v2.1": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 7e-07,
"output_cost_per_token": 8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Sao10K/L3-8B-Lunaris-v1": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3e-08,
"output_cost_per_token": 6e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": {
"max_tokens": 8192,
"max_input_tokens": 8192,
@ -15843,6 +15586,7 @@
"max_output_tokens": 200000,
"input_cost_per_token": 3.3e-06,
"output_cost_per_token": 1.65e-05,
"cache_read_input_token_cost": 3.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
@ -15867,67 +15611,15 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/bigcode/starcoder2-15b-instruct-v0.1": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 2.4e-07,
"output_cost_per_token": 2.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/deepinfra/airoboros-70b": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 7e-07,
"output_cost_per_token": 9e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 5e-07,
"output_cost_per_token": 2.18e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false,
"supports_reasoning": true
},
"deepinfra/deepseek-ai/DeepSeek-R1": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 4.5e-07,
"output_cost_per_token": 2.15e-06,
"input_cost_per_token": 7e-07,
"output_cost_per_token": 2.4e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-0528": {
"max_tokens": 163840,
@ -15935,10 +15627,10 @@
"max_output_tokens": 163840,
"input_cost_per_token": 5e-07,
"output_cost_per_token": 2.15e-06,
"cache_read_input_token_cost": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": {
"max_tokens": 32768,
@ -15948,8 +15640,7 @@
"output_cost_per_token": 3e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": {
"max_tokens": 131072,
@ -15959,8 +15650,7 @@
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false,
"supports_reasoning": true
"supports_tool_choice": false
},
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": {
"max_tokens": 131072,
@ -15970,19 +15660,17 @@
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-R1-Turbo": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 40960,
"max_input_tokens": 40960,
"max_output_tokens": 40960,
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-V3": {
"max_tokens": 163840,
@ -15992,8 +15680,7 @@
"output_cost_per_token": 8.9e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-V3-0324": {
"max_tokens": 163840,
@ -16001,63 +15688,23 @@
"max_output_tokens": 163840,
"input_cost_per_token": 2.8e-07,
"output_cost_per_token": 8.8e-07,
"cache_read_input_token_cost": 2.24e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
},
"deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true,
"supports_reasoning": true
"supports_tool_choice": true
},
"deepinfra/deepseek-ai/DeepSeek-V3.1": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 3e-07,
"input_cost_per_token": 2.7e-07,
"output_cost_per_token": 1e-06,
"cache_read_input_token_cost": 2.16e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false,
"supports_tool_choice": true,
"supports_reasoning": true
},
"deepinfra/google/codegemma-7b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 7e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/google/gemini-1.5-flash": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemini-1.5-flash-8b": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"input_cost_per_token": 3.75e-08,
"output_cost_per_token": 1.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemini-2.0-flash-001": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
@ -16088,36 +15735,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemma-1.1-7b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 7e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/google/gemma-2-27b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 2.7e-07,
"output_cost_per_token": 2.7e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/google/gemma-2-9b-it": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3e-08,
"output_cost_per_token": 6e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/google/gemma-3-12b-it": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16142,48 +15759,8 @@
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 2e-08,
"output_cost_per_token": 4e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/lizpreciatior/lzlv_70b_fp16_hf": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mattshumer/Reflection-Llama-3.1-70B": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-2-13b-chat-hf": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.3e-07,
"output_cost_per_token": 1.3e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-2-70b-chat-hf": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 6.4e-07,
"output_cost_per_token": 8e-07,
"input_cost_per_token": 4e-08,
"output_cost_per_token": 8e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
@ -16198,16 +15775,6 @@
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-3.2-1B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 5e-09,
"output_cost_per_token": 1e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-3.2-3B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16218,16 +15785,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 3.5e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-3.3-70B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16258,16 +15815,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 5e-07,
"output_cost_per_token": 5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
"max_tokens": 327680,
"max_input_tokens": 327680,
@ -16298,16 +15845,6 @@
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": {
"max_tokens": 8192,
"max_input_tokens": 8192,
@ -16318,16 +15855,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 8e-07,
"output_cost_per_token": 8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16368,36 +15895,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/microsoft/Phi-3-medium-4k-instruct": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 1.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/microsoft/Phi-4-multimodal-instruct": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 1e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/microsoft/WizardLM-2-7B": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/microsoft/WizardLM-2-8x22B": {
"max_tokens": 65536,
"max_input_tokens": 65536,
@ -16418,66 +15915,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/microsoft/phi-4-reasoning-plus": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 3.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mistralai/Devstral-Small-2505": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 6e-08,
"output_cost_per_token": 1.2e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Devstral-Small-2507": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 7e-08,
"output_cost_per_token": 2.8e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-7B-Instruct-v0.1": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-7B-Instruct-v0.2": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mistralai/Mistral-7B-Instruct-v0.3": {
"max_tokens": 32768,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"input_cost_per_token": 2.8e-08,
"output_cost_per_token": 5.4e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-Nemo-Instruct-2407": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16498,16 +15935,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 1e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": {
"max_tokens": 128000,
"max_input_tokens": 128000,
@ -16518,16 +15945,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": {
"max_tokens": 65536,
"max_input_tokens": 65536,
"max_output_tokens": 65536,
"input_cost_per_token": 6.5e-07,
"output_cost_per_token": 6.5e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": {
"max_tokens": 32768,
"max_input_tokens": 32768,
@ -16558,16 +15975,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/nvidia/Nemotron-4-340B-Instruct": {
"max_tokens": 4096,
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"input_cost_per_token": 4.2e-06,
"output_cost_per_token": 4.2e-06,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/openai/gpt-oss-120b": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -16588,36 +15995,6 @@
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/openbmb/MiniCPM-Llama3-V-2_5": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 3.4e-07,
"output_cost_per_token": 3.4e-07,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/openchat/openchat-3.6-8b": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": false
},
"deepinfra/openchat/openchat_3.5": {
"max_tokens": 8192,
"max_input_tokens": 8192,
"max_output_tokens": 8192,
"input_cost_per_token": 5.5e-08,
"output_cost_per_token": 5.5e-08,
"litellm_provider": "deepinfra",
"mode": "chat",
"supports_tool_choice": true
},
"deepinfra/zai-org/GLM-4.5": {
"max_tokens": 131072,
"max_input_tokens": 131072,
@ -21126,4 +20503,4 @@
"notes": "Volcengine Doubao embedding model - text-240715 version with 2560 dimensions"
}
}
}
}

3596
poetry.lock generated

File diff suppressed because it is too large Load diff

View file

@ -1,8 +1,8 @@
[tool.poetry]
name = "litellm"
version = "1.77.0"
version = "1.77.1"
description = "Library to easily interface with LLM API providers"
authors = ["BerriAI, AndrewDoan"]
authors = ["BerriAI"]
license = "MIT"
readme = "README.md"
packages = [
@ -156,7 +156,7 @@ requires = ["poetry-core", "wheel"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "1.77.0"
version = "1.77.1"
version_files = [
"pyproject.toml:^version"
]

Binary file not shown.

View file

@ -2,7 +2,8 @@
anyio==4.8.0 # openai + http req.
httpx==0.28.1
openai==1.99.5 # openai req.
fastapi==0.115.5 # server dep
fastapi==0.116.1 # server dep
starlette==0.47.2 # starlette fastapi dep
backoff==2.2.1 # server dep
pyyaml==6.0.2 # server dep
uvicorn==0.29.0 # server dep

View file

@ -0,0 +1,337 @@
import ast
import os
import re
from typing import List, Dict, Any
class SensitiveLogDetector(ast.NodeVisitor):
"""
Detects logger.info() statements that might log sensitive request/response data.
"""
def __init__(self):
self.violations = []
self.current_file = None
def set_file(self, file_path: str):
"""Set the current file being analyzed"""
self.current_file = file_path
def visit_Call(self, node):
"""Visit function calls to detect logger.info() with sensitive data"""
if self._is_logger_info_call(node):
# Check all arguments to the logger.info() call
for arg in node.args:
if self._contains_sensitive_data(arg):
violation = {
"file": self.current_file,
"line": node.lineno,
"call": self._get_call_string(node),
"reason": self._get_violation_reason(arg),
"arg": self._get_arg_string(arg)
}
self.violations.append(violation)
self.generic_visit(node)
def _is_logger_info_call(self, node) -> bool:
"""Check if this is a logger.info() call"""
if not isinstance(node.func, ast.Attribute):
return False
# Check for various logger patterns:
# logger.info(), verbose_logger.info(), verbose_proxy_logger.info(), etc.
if node.func.attr == "info":
if isinstance(node.func.value, ast.Name):
logger_name = node.func.value.id
return any(pattern in logger_name.lower() for pattern in ["logger", "log"])
return False
def _contains_sensitive_data(self, arg) -> bool:
"""Check if the argument might contain sensitive data"""
# Convert argument to string for analysis
arg_str = self._get_arg_string(arg).lower()
# Skip obvious non-sensitive patterns
non_sensitive_patterns = [
r'^["\'][\w\s\-_:.,!?]*["\']$', # Simple static strings
r'^["\'][^{%]*["\']$', # Strings without format placeholders
]
# Skip common safe phrases that contain sensitive keywords
safe_phrases = [
r'request\s+(completed|finished|started|processing)',
r'response\s+(sent|received|processed)',
r'data\s+(inserted|updated|deleted|saved)\s+into',
r'(successfully|failed)\s+(request|response)',
r'(starting|ending|completed)\s+(request|response)',
r'no\s+(usage\s+)?data\s+found',
r'found\s+\d+.*records',
r'exported\s+\d+.*records',
]
for pattern in non_sensitive_patterns:
if re.search(pattern, arg_str):
# Check if it's a safe phrase first
for safe_pattern in safe_phrases:
if re.search(safe_pattern, arg_str, re.IGNORECASE):
return False
# Then check if the static string mentions sensitive keywords
if not any(keyword in arg_str for keyword in
['request', 'response', 'data', 'body', 'payload', 'token', 'auth', 'credential']):
return False
# Direct variable/attribute patterns that are likely sensitive
sensitive_patterns = [
r'\brequest\b(?!\s*(id|status|method))', # request but not request_id, request_status, request_method
r'\bresponse\b(?!\s*(status|code|time))', # response but not response_status, response_code
r'\bdata\b(?=[\.\[\s]|$)', # data followed by . [ space or end
r'\bbody\b(?=[\.\[\s]|$)',
r'\bpayload\b(?=[\.\[\s]|$)',
r'\bmessages?\b(?=[\.\[\s]|$)',
r'\bcontent\b(?=[\.\[\s]|$)',
r'\binput\b(?=[\.\[\s]|$)',
r'\boutput\b(?=[\.\[\s]|$)',
r'\bargs\b(?=[\.\[\s]|$)',
r'\bkwargs\b(?=[\.\[\s]|$)',
r'\bparams\b(?=[\.\[\s]|$)',
r'\bheaders\b(?=[\.\[\s]|$)',
r'\bapi_key\b',
r'\btoken\b(?!\s*(name|id))', # token but not token_name, token_id
r'\bauth\b(?=[\.\[\s]|$)',
r'\bcredentials?\b'
]
# Check for direct variable references with context
for pattern in sensitive_patterns:
if re.search(pattern, arg_str):
return True
# Check for format strings that might interpolate sensitive data
if self._is_format_string_with_sensitive_data(arg):
return True
# Check for JSON dumps or string formatting of objects
if self._is_object_serialization(arg):
return True
return False
def _is_format_string_with_sensitive_data(self, arg) -> bool:
"""Check if this is a format string that might contain sensitive data"""
# Check for f-strings
if isinstance(arg, ast.JoinedStr):
for value in arg.values:
if isinstance(value, ast.FormattedValue):
value_str = self._get_arg_string(value.value).lower()
if any(pattern in value_str for pattern in
['request', 'response', 'data', 'body', 'content', 'messages']):
return True
# Check for .format() calls
if isinstance(arg, ast.Call) and isinstance(arg.func, ast.Attribute):
if arg.func.attr == "format":
# Check the base string for suspicious patterns
base_str = self._get_arg_string(arg.func.value).lower()
if "{}" in base_str or "{" in base_str:
# Check format arguments for sensitive data
for format_arg in arg.args:
format_str = self._get_arg_string(format_arg).lower()
if any(pattern in format_str for pattern in
['request', 'response', 'data', 'body', 'content']):
return True
return False
def _is_object_serialization(self, arg) -> bool:
"""Check if this is serializing an object that might contain sensitive data"""
arg_str = self._get_arg_string(arg)
# Check for json.dumps() calls
if isinstance(arg, ast.Call):
if (isinstance(arg.func, ast.Attribute) and
arg.func.attr == "dumps" and
isinstance(arg.func.value, ast.Name) and
arg.func.value.id == "json"):
return True
# Check for str() calls on potentially sensitive objects
if (isinstance(arg.func, ast.Name) and arg.func.id == "str" and
len(arg.args) > 0):
obj_str = self._get_arg_string(arg.args[0]).lower()
if any(pattern in obj_str for pattern in
['request', 'response', 'data', 'body']):
return True
return False
def _get_violation_reason(self, arg) -> str:
"""Get a human-readable reason for the violation"""
arg_str = self._get_arg_string(arg).lower()
if 'request' in arg_str:
return "Potentially logging request data"
elif 'response' in arg_str:
return "Potentially logging response data"
elif any(pattern in arg_str for pattern in ['data', 'body', 'payload', 'content']):
return "Potentially logging sensitive data/body/content"
elif any(pattern in arg_str for pattern in ['messages', 'input', 'output']):
return "Potentially logging message/input/output data"
elif any(pattern in arg_str for pattern in ['api_key', 'token', 'auth', 'credentials']):
return "Potentially logging authentication data"
else:
return "Potentially logging sensitive data"
def _get_call_string(self, node) -> str:
"""Get string representation of the function call"""
try:
if hasattr(ast, 'unparse'):
return ast.unparse(node)
else:
# Fallback for older Python versions
return f"{self._get_arg_string(node.func)}(...)"
except:
return "logger.info(...)"
def _get_arg_string(self, arg) -> str:
"""Get string representation of an argument"""
try:
if hasattr(ast, 'unparse'):
return ast.unparse(arg)
else:
# Fallback for older Python versions
if isinstance(arg, ast.Name):
return arg.id
elif isinstance(arg, ast.Attribute):
return f"{self._get_arg_string(arg.value)}.{arg.attr}"
elif isinstance(arg, ast.Str):
return repr(arg.s)
elif isinstance(arg, ast.Constant):
return repr(arg.value)
else:
return str(type(arg).__name__)
except:
return "unknown"
def check_sensitive_logging(base_dir: str) -> List[Dict[str, Any]]:
"""
Check for logger.info() statements that might log sensitive data.
Args:
base_dir: Base directory to scan (typically the litellm root)
Returns:
List of violations found
"""
detector = SensitiveLogDetector()
all_violations = []
# Directories to scan - only main litellm codebase
scan_dirs = [
"litellm",
"enterprise" # Include enterprise directory if it exists
]
# Directories to exclude (third-party code, venvs, etc.)
exclude_dirs = {
"venv", "venv313", ".venv", "env", ".env",
"node_modules", "__pycache__", ".git",
"build", "dist", ".tox", "clean_env",
"litellm_env", "myenv", "py313_env",
"venv_sip_bypass", "mypyc_env"
}
for scan_dir in scan_dirs:
dir_path = os.path.join(base_dir, scan_dir)
if not os.path.exists(dir_path):
print(f"Warning: Directory {dir_path} does not exist, skipping.")
continue
print(f"Scanning directory: {dir_path}")
for root, dirs, files in os.walk(dir_path):
# Skip excluded directories
dirs[:] = [d for d in dirs if d not in exclude_dirs]
# Skip if we're in a virtual environment or third-party directory
relative_root = os.path.relpath(root, base_dir)
if any(excluded in relative_root.split(os.sep) for excluded in exclude_dirs):
continue
for file in files:
if file.endswith(".py"):
file_path = os.path.join(root, file)
relative_path = os.path.relpath(file_path, base_dir)
# Skip files that are clearly third-party or generated
if any(excluded in relative_path for excluded in exclude_dirs):
continue
try:
with open(file_path, "r", encoding="utf-8") as f:
content = f.read()
tree = ast.parse(content)
detector.set_file(relative_path)
detector.visit(tree)
except SyntaxError as e:
print(f"Warning: Syntax error in file {relative_path}: {e}")
continue
except UnicodeDecodeError as e:
print(f"Warning: Unicode decode error in file {relative_path}: {e}")
continue
except Exception as e:
print(f"Warning: Error processing file {relative_path}: {e}")
continue
return detector.violations
def main():
"""Main function to run the sensitive logging check"""
# Get the base directory (assume we're running from tests/code_coverage_tests/)
###################
# Running locally
###################
# current_dir = os.path.dirname(os.path.abspath(__file__))
# base_dir = os.path.join(current_dir, "..", "..")
# base_dir = os.path.abspath(base_dir)
###################
# Running in CI/CD
###################
base_dir = "./litellm" # Adjust this path as needed
print(f"Checking for sensitive logging in: {base_dir}")
violations = check_sensitive_logging(base_dir)
if violations:
print(f"\n❌ Found {len(violations)} potential violations:")
print("=" * 80)
for i, violation in enumerate(violations, 1):
print(f"\n{i}. {violation['file']}:{violation['line']}")
print(f" Reason: {violation['reason']}")
print(f" Call: {violation['call']}")
print(f" Argument: {violation['arg']}")
print("\n" + "=" * 80)
print("⚠️ SECURITY WARNING:")
print("These logger.info() statements may log sensitive request/response data.")
print("Consider changing them to logger.debug() or removing sensitive data.")
print("This is critical for PII compliance and security.")
print("Please contact @ishaan-jaff for more details about this check. DO NOT VIOLATE THIS CHECK.")
return 1 # Exit with error code
else:
print("\n✅ No sensitive logging violations found!")
return 0
if __name__ == "__main__":
exit(main())

View file

@ -80,6 +80,7 @@ lmdb: >=1.5.1
openai: >=1.1.0 # APACHE 2.0 License
httpx: >=0.25.0 # BSD 3-Clause License
fastapi: >=0.115.5 # MIT License
starlette: >=0.47.2 # MIT License
uvicorn: >=0.29.0 # BSD 3-Clause License
anthropic: >=0.21.3 # MIT License
detect-secrets: >=1.5.0 # MIT License

View file

@ -28,6 +28,7 @@ IGNORE_FUNCTIONS = [
"_remove_json_schema_refs", # max depth set.,
"_convert_schema_types", # max depth set.,
"_fix_enum_empty_strings", # max depth set.,
"get_access_token", # max depth set.,
]

View file

@ -198,17 +198,6 @@ class TestAzureOpenAIDalle3(BaseImageGenTest):
}
class TestAzureFoundryFlux(BaseImageGenTest):
def get_base_image_generation_call_args(self) -> dict:
litellm.set_verbose = True
return {
"model": "azure_ai/FLUX.1-Kontext-pro",
"api_base": os.getenv("AZURE_FLUX_API_BASE"),
"api_key": os.getenv("AZURE_GPT5_API_KEY"),
"n": 1,
"quality": "standard",
}
@pytest.mark.flaky(retries=3, delay=1)
def test_image_generation_azure_dall_e_3():

View file

@ -0,0 +1,119 @@
import os
import sys
import pytest
sys.path.insert(
0, os.path.abspath("../../../../..")
) # Adds the parent directory to the system path
from litellm.llms.vertex_ai.gemini import transformation
from litellm.types.llms import openai
from litellm.types import completion
from litellm.types.llms.vertex_ai import RequestBody
@pytest.mark.asyncio
async def test__transform_request_body_labels():
"""
Test that Vertex AI requests use the optional Vertex AI
"labels" parameters sent by client.
"""
# Set up the test parameters
model = "vertex_ai/gemini-1.5-pro"
messages = [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "Hello! How can I assist you today?"},
{"role": "user", "content": "hi"},
]
optional_params = {
"labels": {"lparam1": "lvalue1", "lparam2": "lvalue2"}
}
litellm_params = {}
transform_request_params = {
"messages": messages,
"model": model,
"optional_params": optional_params,
"custom_llm_provider": "vertex_ai",
"litellm_params": litellm_params,
"cached_content": None,
}
rb: RequestBody = transformation._transform_request_body(**transform_request_params)
# Check URL
assert rb["contents"] == [{'parts': [{'text': 'hi'}], 'role': 'user'}, {'parts': [{'text': 'Hello! How can I assist you today?'}], 'role': 'model'}, {'parts': [{'text': 'hi'}], 'role': 'user'}]
assert "labels" in rb and rb["labels"] == {"lparam1": "lvalue1", "lparam2": "lvalue2"}
@pytest.mark.asyncio
async def test__transform_request_body_metadata():
"""
Test that Vertex AI requests use the optional Open AI
"metadata" parameters sent by client.
"""
# Set up the test parameters
model = "vertex_ai/gemini-1.5-pro"
messages = [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "Hello! How can I assist you today?"},
{"role": "user", "content": "hi"},
]
optional_params = {}
litellm_params = {
"metadata": {
"requester_metadata": {"rparam1": "rvalue1", "rparam2": "rvalue2"}
}
}
transform_request_params = {
"messages": messages,
"model": model,
"optional_params": optional_params,
"custom_llm_provider": "vertex_ai",
"litellm_params": litellm_params,
"cached_content": None,
}
rb: RequestBody = transformation._transform_request_body(**transform_request_params)
# Check URL
assert rb["contents"] == [{'parts': [{'text': 'hi'}], 'role': 'user'}, {'parts': [{'text': 'Hello! How can I assist you today?'}], 'role': 'model'}, {'parts': [{'text': 'hi'}], 'role': 'user'}]
assert "labels" in rb and rb["labels"] == {"rparam1": "rvalue1", "rparam2": "rvalue2"}
@pytest.mark.asyncio
async def test__transform_request_body_labels_and_metadata():
"""
Test that Vertex AI requests use the optional Vertex AI
"labels" parameters sent by client and that the "metadata"
optional Open AI parameters are ignored if the client uses
"labels" parameters.
"""
# Set up the test parameters
model = "vertex_ai/gemini-1.5-pro"
messages = [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "Hello! How can I assist you today?"},
{"role": "user", "content": "hi"},
]
optional_params = {
"labels": {"lparam1": "lvalue1", "lparam2": "lvalue2"}
}
litellm_params = {
"metadata": {
"requester_metadata": {"rparam1": "rvalue1", "rparam2": "rvalue2"}
}
}
transform_request_params = {
"messages": messages,
"model": model,
"optional_params": optional_params,
"custom_llm_provider": "vertex_ai",
"litellm_params": litellm_params,
"cached_content": None,
}
rb: RequestBody = transformation._transform_request_body(**transform_request_params)
# Check URL
assert rb["contents"] == [{'parts': [{'text': 'hi'}], 'role': 'user'}, {'parts': [{'text': 'Hello! How can I assist you today?'}], 'role': 'model'}, {'parts': [{'text': 'hi'}], 'role': 'user'}]
assert "labels" in rb and rb["labels"] == {"lparam1": "lvalue1", "lparam2": "lvalue2"}

View file

@ -338,7 +338,9 @@ def test_aget_valid_models():
print(valid_models)
# list of openai supported llms on litellm
expected_models = litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models
expected_models = (
litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models
)
assert set(valid_models) == set(expected_models)
@ -410,7 +412,12 @@ def test_validate_environment_api_key():
def test_validate_environment_api_version():
response_obj = validate_environment(model="azure/openai-deployment", api_key="sk-my-test-key", api_base="https://fake.openai.azure.com/", api_version="2024-02-15")
response_obj = validate_environment(
model="azure/openai-deployment",
api_key="sk-my-test-key",
api_base="https://fake.openai.azure.com/",
api_version="2024-02-15",
)
assert (
response_obj["keys_in_environment"] is True
), f"Missing keys={response_obj['missing_keys']}"
@ -513,7 +520,6 @@ def test_function_to_dict():
("gpt-3.5-turbo", True),
("azure/gpt-4-1106-preview", True),
("groq/gemma-7b-it", True),
("anthropic.claude-instant-v1", False),
("gemini/gemini-1.5-flash", True),
],
)
@ -1690,15 +1696,6 @@ def test_pick_cheapest_chat_model_from_llm_provider():
assert len(pick_cheapest_chat_models_from_llm_provider("unknown", n=1)) == 0
def test_get_potential_model_names():
from litellm.utils import _get_potential_model_names
assert _get_potential_model_names(
model="bedrock/ap-northeast-1/anthropic.claude-instant-v1",
custom_llm_provider="bedrock",
)
@pytest.mark.parametrize("num_retries", [0, 1, 5])
def test_get_num_retries(num_retries):
from litellm.utils import _get_wrapper_num_retries

View file

@ -171,17 +171,14 @@ def test_azure_extra_headers(input, call_type, header_value):
"api_base, model, expected_endpoint",
[
(
os.getenv("AZURE_SWEDEN_API_BASE"),
"https://my-endpoint-sweden-berri992.openai.azure.com",
"dall-e-3-test",
os.getenv("AZURE_SWEDEN_API_BASE")
+ "/openai/deployments/dall-e-3-test/images/generations?api-version=2023-12-01-preview",
"https://my-endpoint-sweden-berri992.openai.azure.com/openai/deployments/dall-e-3-test/images/generations?api-version=2023-12-01-preview",
),
(
os.getenv("AZURE_SWEDEN_API_BASE")
+ "/openai/deployments/my-custom-deployment",
"https://my-endpoint-sweden-berri992.openai.azure.com/openai/deployments/my-custom-deployment",
"dall-e-3",
os.getenv("AZURE_SWEDEN_API_BASE")
+ "/openai/deployments/my-custom-deployment/images/generations?api-version=2023-12-01-preview",
"https://my-endpoint-sweden-berri992.openai.azure.com/openai/deployments/my-custom-deployment/images/generations?api-version=2023-12-01-preview",
),
],
)
@ -261,7 +258,7 @@ def test_azure_openai_gpt_4o_naming(monkeypatch):
client = AzureOpenAI(
api_key="test-api-key",
base_url=os.getenv("AZURE_SWEDEN_API_BASE"),
base_url="https://my-endpoint-sweden-berri992.openai.azure.com",
api_version="2023-12-01-preview",
)

View file

@ -20,6 +20,7 @@ import pytest
@pytest.mark.asyncio
@pytest.mark.skip(reason="Skipping bedrock agents test - arn not working")
async def test_bedrock_agents():
litellm._turn_on_debug()
response = litellm.completion(
@ -44,6 +45,7 @@ async def test_bedrock_agents():
@pytest.mark.asyncio
@pytest.mark.skip(reason="Skipping bedrock agents test - arn not working")
async def test_bedrock_agents_with_streaming():
# litellm._turn_on_debug()
response = litellm.completion(

View file

@ -69,7 +69,7 @@ def test_completion_bedrock_claude_completion_auth():
try:
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,
@ -105,7 +105,7 @@ def test_completion_bedrock_guardrails(streaming):
try:
if streaming is False:
response = completion(
model="anthropic.claude-v2",
model="anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=[
{
"content": "where do i buy coffee from? ",
@ -133,7 +133,7 @@ def test_completion_bedrock_guardrails(streaming):
else:
litellm.set_verbose = True
response = completion(
model="anthropic.claude-v2",
model="anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=[
{
"content": "where do i buy coffee from? ",
@ -166,39 +166,6 @@ def test_completion_bedrock_guardrails(streaming):
pytest.fail(f"Error occurred: {e}")
def test_completion_bedrock_claude_2_1_completion_auth():
print("calling bedrock claude 2.1 completion params auth")
import os
aws_access_key_id = os.environ["AWS_ACCESS_KEY_ID"]
aws_secret_access_key = os.environ["AWS_SECRET_ACCESS_KEY"]
aws_region_name = os.environ["AWS_REGION_NAME"]
os.environ.pop("AWS_ACCESS_KEY_ID", None)
os.environ.pop("AWS_SECRET_ACCESS_KEY", None)
os.environ.pop("AWS_REGION_NAME", None)
try:
response = completion(
model="bedrock/anthropic.claude-v2:1",
messages=messages,
max_tokens=10,
temperature=0.1,
aws_access_key_id=aws_access_key_id,
aws_secret_access_key=aws_secret_access_key,
aws_region_name=aws_region_name,
)
# Add any assertions here to check the response
print(response)
os.environ["AWS_ACCESS_KEY_ID"] = aws_access_key_id
os.environ["AWS_SECRET_ACCESS_KEY"] = aws_secret_access_key
os.environ["AWS_REGION_NAME"] = aws_region_name
except RateLimitError:
pass
except Exception as e:
pytest.fail(f"Error occurred: {e}")
# test_completion_bedrock_claude_2_1_completion_auth()
@ -228,7 +195,7 @@ def test_completion_bedrock_claude_external_client_auth():
)
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,
@ -265,7 +232,7 @@ def test_completion_bedrock_claude_sts_client_auth():
litellm.set_verbose = True
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,
@ -734,8 +701,6 @@ def test_bedrock_stop_value(stop, model):
"model",
[
"anthropic.claude-3-sonnet-20240229-v1:0",
# "meta.llama3-70b-instruct-v1:0",
"anthropic.claude-v2",
"mistral.mixtral-8x7b-instruct-v0:1",
],
)
@ -939,7 +904,7 @@ def test_bedrock_ptu():
)
try:
response = litellm.completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=[{"role": "user", "content": "What's AWS?"}],
model_id=model_id,
client=client,
@ -1105,7 +1070,7 @@ def test_completion_bedrock_external_client_region():
with patch.object(client, "post", new=Mock()) as mock_client_post:
try:
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,

View file

@ -265,9 +265,11 @@ def test_gemini_image_generation():
#########################################################
# Important: Validate we did get an image in the response
#########################################################
assert response.choices[0].message.image is not None
assert response.choices[0].message.image["url"] is not None
assert response.choices[0].message.image["url"].startswith("data:image/png;base64,")
assert response.choices[0].message.images is not None
assert len(response.choices[0].message.images) > 0
assert response.choices[0].message.images[0]["image_url"] is not None
assert response.choices[0].message.images[0]["image_url"]["url"] is not None
assert response.choices[0].message.images[0]["image_url"]["url"].startswith("data:image/png;base64,")
def test_gemini_thinking():

View file

@ -451,6 +451,8 @@ class TestOpenAIGPT4OAudioTranscription(BaseLLMAudioTranscriptionTest):
def get_base_audio_transcription_call_args(self) -> dict:
return {
"model": "openai/gpt-4o-transcribe",
# "response_format": "verbose_json",
"timestamp_granularities": ["word"],
}
def get_custom_llm_provider(self) -> litellm.LlmProviders:
@ -655,6 +657,7 @@ def test_openai_tool_calling():
response = litellm.completion(**completion_params)
@pytest.mark.asyncio
async def test_openai_gpt5_reasoning():
response = await litellm.acompletion(
@ -665,11 +668,12 @@ async def test_openai_gpt5_reasoning():
print("response: ", response)
assert response.choices[0].message.content is not None
@pytest.mark.asyncio
async def test_openai_safety_identifier_parameter():
"""Test that safety_identifier parameter is correctly passed to the OpenAI API."""
from openai import AsyncOpenAI
litellm.set_verbose = True
client = AsyncOpenAI(api_key="fake-api-key")
@ -698,7 +702,7 @@ async def test_openai_safety_identifier_parameter():
def test_openai_safety_identifier_parameter_sync():
"""Test that safety_identifier parameter is correctly passed to the OpenAI API."""
from openai import OpenAI
litellm.set_verbose = True
client = OpenAI(api_key="fake-api-key")
@ -722,4 +726,3 @@ def test_openai_safety_identifier_parameter_sync():
assert "safety_identifier" in request_body
# Verify safety_identifier is correctly sent to the API
assert request_body["safety_identifier"] == "user_code_123456"

View file

@ -7,7 +7,7 @@ import pytest
sys.path.insert(0, os.path.abspath("../.."))
from typing import Union
from typing import Union, List
# from litellm.litellm_core_utils.prompt_templates.factory import prompt_factory
import litellm
@ -28,6 +28,7 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
from litellm.llms.vertex_ai.gemini.transformation import (
_gemini_convert_messages_with_history,
)
from litellm.types.llms.openai import AllMessageValues
from unittest.mock import AsyncMock, MagicMock, patch
@ -472,6 +473,20 @@ def test_vertex_only_image_user_message():
)
def test_no_messages_yields_user_text():
"""
Test that contents are not empty and have text when called without messages
This is to support blha blah
"""
messages: List[AllMessageValues] = []
contents = _gemini_convert_messages_with_history(messages=messages)
expected_output = [{"role": "user", "parts": [{"text": " "}]}]
assert contents == expected_output
def test_convert_url():
convert_url_to_base64("https://picsum.photos/id/237/200/300")
@ -630,7 +645,6 @@ def test_azure_tool_call_invoke_helper():
def test_ensure_alternating_roles(
messages, expected_messages, user_continue_message, assistant_continue_message
):
messages = get_completion_messages(
messages=messages,
assistant_continue_message=assistant_continue_message,
@ -651,7 +665,7 @@ def test_alternating_roles_e2e():
http_handler = HTTPHandler()
with patch.object(http_handler, "post", new=MagicMock()) as mock_post:
try:
try:
response = litellm.completion(
**{
"model": "databricks/databricks-meta-llama-3-1-70b-instruct",
@ -663,7 +677,10 @@ def test_alternating_roles_e2e():
},
{"role": "user", "content": "What is Databricks?"},
{"role": "user", "content": "What is Azure?"},
{"role": "assistant", "content": "I don't know anyything, do you?"},
{
"role": "assistant",
"content": "I don't know anyything, do you?",
},
{"role": "assistant", "content": "I can't repeat sentences."},
],
"user_continue_message": {
@ -712,7 +729,7 @@ def test_alternating_roles_e2e():
"role": "user",
"content": "Ok",
},
]
],
}
)

View file

@ -314,6 +314,7 @@ async def test_caching_with_cache_controls(sync_flag):
# test_caching_with_cache_controls()
@pytest.mark.flaky(retries=3, delay=1)
def test_caching_with_models_v2():
messages = [
@ -449,6 +450,7 @@ def test_embedding_caching():
# test_embedding_caching()
@pytest.mark.asyncio
async def test_embedding_caching_individual_items_and_then_list():
litellm._turn_on_debug()
@ -473,7 +475,7 @@ async def test_embedding_caching_individual_items_and_then_list():
assert embedding3["data"][0]["embedding"] == embedding1["data"][0]["embedding"]
assert embedding3["data"][1]["embedding"] == embedding2["data"][0]["embedding"]
assert embedding3._hidden_params["cache_hit"] == True
assert embedding3.usage.prompt_tokens != 0
assert embedding3.usage.prompt_tokens != 0
## with new input, check that prompt tokens increase
additional_text = "this is a new text"
@ -483,6 +485,7 @@ async def test_embedding_caching_individual_items_and_then_list():
)
assert embedding4.usage.prompt_tokens > embedding3.usage.prompt_tokens
@pytest.mark.asyncio
async def test_embedding_caching_individual_items():
litellm.cache = Cache()
@ -500,7 +503,7 @@ async def test_embedding_caching_individual_items():
assert embedding3["data"][0]["embedding"] == embedding1["data"][0]["embedding"]
assert len(embedding3.data) == 1
assert embedding3._hidden_params["cache_hit"] == True
assert embedding3.usage.prompt_tokens != 0
assert embedding3.usage.prompt_tokens != 0
def test_embedding_caching_azure():
@ -1156,7 +1159,7 @@ async def test_redis_cache_acompletion_stream_bedrock():
response_2_content = ""
response1 = await litellm.acompletion(
model="bedrock/anthropic.claude-v2",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=40,
temperature=1,
@ -1171,7 +1174,7 @@ async def test_redis_cache_acompletion_stream_bedrock():
print("\n\n Response 1 content: ", response_1_content, "\n\n")
response2 = await litellm.acompletion(
model="bedrock/anthropic.claude-v2",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=40,
temperature=1,
@ -1883,9 +1886,6 @@ def test_caching_redis_simple(caplog, capsys):
assert "async success_callback: reaches cache for logging" not in captured.out
@pytest.mark.asyncio()
async def test_cache_default_off_acompletion():
litellm.set_verbose = True
@ -2417,7 +2417,7 @@ async def test_redis_increment_pipeline():
results = await redis_cache.async_increment_pipeline(increment_list)
# Verify results
assert len(results) == 4
assert len(results) == 4
# Verify the values were actually set in Redis
value1 = await redis_cache.async_get_cache("test_key1")
@ -2502,116 +2502,136 @@ def test_redis_caching_multiple_namespaces():
# Use a fixed uuid to ensure consistent cache keys
test_uuid = "12345678-1234-1234-1234-123456789abc"
messages = [{"role": "user", "content": f"what is litellm? {test_uuid}"}]
# Mock the Redis client creation from the _redis module
with patch('litellm._redis.get_redis_client') as mock_get_redis_client, \
patch('litellm._redis.get_redis_connection_pool') as mock_get_redis_connection_pool:
with patch("litellm._redis.get_redis_client") as mock_get_redis_client, patch(
"litellm._redis.get_redis_connection_pool"
) as mock_get_redis_connection_pool:
# Create a mock Redis client that simulates real Redis behavior
mock_redis_client = MagicMock()
mock_get_redis_client.return_value = mock_redis_client
# Mock the connection pool
mock_connection_pool = MagicMock()
mock_get_redis_connection_pool.return_value = mock_connection_pool
# Dictionary to simulate Redis storage with namespace support
redis_storage = {}
def mock_redis_get(key):
print(f"Redis GET: {key}")
value = redis_storage.get(key, None)
# Convert to bytes to match real Redis behavior
if value is not None:
import json
return json.dumps(value).encode('utf-8')
return json.dumps(value).encode("utf-8")
return None
def mock_redis_set(name, value, ex=None, **kwargs):
print(f"Redis SET: {name} = {value}")
redis_storage[name] = value
return True
def mock_redis_ping():
return True
def mock_redis_info():
return {"redis_version": "7.0.0"}
mock_redis_client.get = mock_redis_get
mock_redis_client.set = mock_redis_set
mock_redis_client.ping = mock_redis_ping
mock_redis_client.info = mock_redis_info
# Initialize the cache
litellm.cache = Cache(type="redis")
namespace_1 = "org-id1"
namespace_2 = "org-id2"
# Use mock_response to ensure deterministic responses without external API calls
response_1 = completion(
model="gpt-3.5-turbo",
messages=messages,
model="gpt-3.5-turbo",
messages=messages,
cache={"namespace": namespace_1},
mock_response="Response for namespace 1"
mock_response="Response for namespace 1",
)
response_2 = completion(
model="gpt-3.5-turbo",
messages=messages,
model="gpt-3.5-turbo",
messages=messages,
cache={"namespace": namespace_2},
mock_response="Response for namespace 2"
mock_response="Response for namespace 2",
)
response_3 = completion(
model="gpt-3.5-turbo",
messages=messages,
model="gpt-3.5-turbo",
messages=messages,
cache={"namespace": namespace_1},
mock_response="This should be cached"
mock_response="This should be cached",
)
response_4 = completion(
model="gpt-3.5-turbo",
model="gpt-3.5-turbo",
messages=messages,
mock_response="Response without namespace"
mock_response="Response without namespace",
)
print(
f"Response 1 type: {type(response_1)} - ID: {getattr(response_1, 'id', 'N/A')}"
)
print(
f"Response 2 type: {type(response_2)} - ID: {getattr(response_2, 'id', 'N/A')}"
)
print(
f"Response 3 type: {type(response_3)} - Cache hit: {isinstance(response_3, str)}"
)
print(
f"Response 4 type: {type(response_4)} - ID: {getattr(response_4, 'id', 'N/A')}"
)
print(f"Response 1 type: {type(response_1)} - ID: {getattr(response_1, 'id', 'N/A')}")
print(f"Response 2 type: {type(response_2)} - ID: {getattr(response_2, 'id', 'N/A')}")
print(f"Response 3 type: {type(response_3)} - Cache hit: {isinstance(response_3, str)}")
print(f"Response 4 type: {type(response_4)} - ID: {getattr(response_4, 'id', 'N/A')}")
print(f"Redis storage keys: {list(redis_storage.keys())}")
# Verify that different namespaces created different cache keys
cache_keys = list(redis_storage.keys())
namespace_1_keys = [k for k in cache_keys if k.startswith(f"{namespace_1}:")]
namespace_2_keys = [k for k in cache_keys if k.startswith(f"{namespace_2}:")]
no_namespace_keys = [k for k in cache_keys if not k.startswith(f"{namespace_1}:") and not k.startswith(f"{namespace_2}:")]
no_namespace_keys = [
k
for k in cache_keys
if not k.startswith(f"{namespace_1}:")
and not k.startswith(f"{namespace_2}:")
]
print(f"Namespace 1 keys: {namespace_1_keys}")
print(f"Namespace 2 keys: {namespace_2_keys}")
print(f"No namespace keys: {no_namespace_keys}")
# Should have at least one key for each namespace
assert len(namespace_1_keys) > 0, "Should have cache keys for namespace 1"
assert len(namespace_2_keys) > 0, "Should have cache keys for namespace 2"
assert len(no_namespace_keys) > 0, "Should have cache keys for no namespace"
# The main test: response 3 should be a cache hit (string) because it uses same namespace as response 1
assert isinstance(response_3, str), "Response 3 should be a cache hit (string) for same namespace"
assert isinstance(
response_3, str
), "Response 3 should be a cache hit (string) for same namespace"
# response 1 & 2 should be ModelResponse objects (cache misses)
assert hasattr(response_1, 'id'), "Response 1 should be a ModelResponse object"
assert hasattr(response_2, 'id'), "Response 2 should be a ModelResponse object"
assert hasattr(response_4, 'id'), "Response 4 should be a ModelResponse object"
assert hasattr(response_1, "id"), "Response 1 should be a ModelResponse object"
assert hasattr(response_2, "id"), "Response 2 should be a ModelResponse object"
assert hasattr(response_4, "id"), "Response 4 should be a ModelResponse object"
# response 1 & 2 should have different IDs (different namespaces)
assert response_1.id != response_2.id, f"Expected different response ID for different namespace. Got {response_1.id} and {response_2.id}"
assert (
response_1.id != response_2.id
), f"Expected different response ID for different namespace. Got {response_1.id} and {response_2.id}"
# response 1 & 4 should have different IDs (different namespaces)
assert response_1.id != response_4.id, f"Expected different response ID for no namespace vs namespaced. Got {response_1.id} and {response_4.id}"
assert (
response_1.id != response_4.id
), f"Expected different response ID for no namespace vs namespaced. Got {response_1.id} and {response_4.id}"
def test_caching_with_reasoning_content():
@ -2643,12 +2663,22 @@ def test_caching_with_reasoning_content():
def test_caching_reasoning_args_miss(): # test in memory cache
try:
#litellm._turn_on_debug()
# litellm._turn_on_debug()
litellm.set_verbose = True
litellm.cache = Cache(
litellm.cache = Cache()
response1 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
reasoning_effort="low",
mock_response="My response",
)
response2 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
mock_response="My response",
)
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, reasoning_effort="low", mock_response="My response")
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, mock_response="My response")
print(f"response1: {response1}")
print(f"response2: {response2}")
assert response1.id != response2.id
@ -2656,29 +2686,52 @@ def test_caching_reasoning_args_miss(): # test in memory cache
print(f"error occurred: {traceback.format_exc()}")
pytest.fail(f"Error occurred: {e}")
def test_caching_reasoning_args_hit(): # test in memory cache
try:
#litellm._turn_on_debug()
# litellm._turn_on_debug()
litellm.set_verbose = True
litellm.cache = Cache(
litellm.cache = Cache()
response1 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
reasoning_effort="low",
mock_response="My response",
)
response2 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
reasoning_effort="low",
mock_response="My response",
)
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, reasoning_effort="low", mock_response="My response")
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, reasoning_effort="low", mock_response="My response")
print(f"response1: {response1}")
print(f"response2: {response2}")
assert response1.id == response2.id
except Exception as e:
print(f"error occurred: {traceback.format_exc()}")
pytest.fail(f"Error occurred: {e}")
def test_caching_thinking_args_miss(): # test in memory cache
try:
#litellm._turn_on_debug()
# litellm._turn_on_debug()
litellm.set_verbose = True
litellm.cache = Cache(
litellm.cache = Cache()
response1 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
thinking={"type": "enabled", "budget_tokens": 1024},
mock_response="My response",
)
response2 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
mock_response="My response",
)
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, thinking={"type": "enabled", "budget_tokens": 1024}, mock_response="My response")
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, mock_response="My response")
print(f"response1: {response1}")
print(f"response2: {response2}")
assert response1.id != response2.id
@ -2686,18 +2739,29 @@ def test_caching_thinking_args_miss(): # test in memory cache
print(f"error occurred: {traceback.format_exc()}")
pytest.fail(f"Error occurred: {e}")
def test_caching_thinking_args_hit(): # test in memory cache
try:
#litellm._turn_on_debug()
# litellm._turn_on_debug()
litellm.set_verbose = True
litellm.cache = Cache(
litellm.cache = Cache()
response1 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
thinking={"type": "enabled", "budget_tokens": 1024},
mock_response="My response",
)
response2 = completion(
model="claude-3-7-sonnet-latest",
messages=messages,
caching=True,
thinking={"type": "enabled", "budget_tokens": 1024},
mock_response="My response",
)
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, thinking={"type": "enabled", "budget_tokens": 1024}, mock_response="My response" )
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, thinking={"type": "enabled", "budget_tokens": 1024}, mock_response="My response")
print(f"response1: {response1}")
print(f"response2: {response2}")
assert response1.id == response2.id
except Exception as e:
print(f"error occurred: {traceback.format_exc()}")
pytest.fail(f"Error occurred: {e}")

View file

@ -134,6 +134,7 @@ async def test_async_log_cache_hit_on_callbacks():
mock_logging_obj = MagicMock()
mock_logging_obj.async_success_handler = AsyncMock()
mock_logging_obj.success_handler = MagicMock()
mock_logging_obj.handle_sync_success_callbacks_for_async_calls = MagicMock()
cached_result = "Mocked cached result"
start_time = datetime.now()
@ -156,14 +157,14 @@ async def test_async_log_cache_hit_on_callbacks():
# Assertions
mock_logging_obj.async_success_handler.assert_called_once_with(
cached_result, start_time, end_time, cache_hit
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
)
# Wait for the thread to complete
await asyncio.sleep(0.5)
mock_logging_obj.success_handler.assert_called_once_with(
cached_result, start_time, end_time, cache_hit
mock_logging_obj.handle_sync_success_callbacks_for_async_calls.assert_called_once_with(
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
)

View file

@ -759,6 +759,7 @@ def test_completion_base64(model):
else:
pytest.fail(f"An exception occurred - {str(e)}")
def test_completion_mistral_api():
try:
litellm.set_verbose = True
@ -3190,7 +3191,6 @@ def response_format_tests(response: litellm.ModelResponse):
"bedrock/mistral.mistral-large-2407-v1:0",
"bedrock/cohere.command-r-plus-v1:0",
"anthropic.claude-3-sonnet-20240229-v1:0",
"anthropic.claude-instant-v1",
"mistral.mistral-7b-instruct-v0:2",
# "bedrock/amazon.titan-tg1-large",
"meta.llama3-8b-instruct-v1:0",

View file

@ -319,64 +319,9 @@ def test_cost_openai_image_gen():
assert cost == 0.019922944
def test_cost_bedrock_pricing():
"""
- get pricing specific to region for a model
"""
from litellm import Choices, Message, ModelResponse
from litellm.utils import Usage
litellm.set_verbose = True
input_tokens = litellm.token_counter(
model="bedrock/anthropic.claude-instant-v1",
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
print(f"input_tokens: {input_tokens}")
output_tokens = litellm.token_counter(
model="bedrock/anthropic.claude-instant-v1",
text="It's all going well",
count_response_tokens=True,
)
print(f"output_tokens: {output_tokens}")
resp = ModelResponse(
id="chatcmpl-e41836bb-bb8b-4df2-8e70-8f3e160155ac",
choices=[
Choices(
finish_reason=None,
index=0,
message=Message(
content="It's all going well",
role="assistant",
),
)
],
created=1700775391,
model="anthropic.claude-instant-v1",
object="chat.completion",
system_fingerprint=None,
usage=Usage(
prompt_tokens=input_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + output_tokens,
),
)
resp._hidden_params = {
"custom_llm_provider": "bedrock",
"region_name": "ap-northeast-1",
}
cost = litellm.completion_cost(
model="anthropic.claude-instant-v1",
completion_response=resp,
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
predicted_cost = input_tokens * 0.00000223 + 0.00000755 * output_tokens
assert cost == predicted_cost
def test_cost_bedrock_pricing_actual_calls():
litellm.set_verbose = True
model = "anthropic.claude-instant-v1"
model = "anthropic.claude-3-5-sonnet-20240620-v1:0"
messages = [{"role": "user", "content": "Hey, how's it going?"}]
response = litellm.completion(
model=model, messages=messages, mock_response="hello cool one"
@ -384,7 +329,7 @@ def test_cost_bedrock_pricing_actual_calls():
print("response", response)
cost = litellm.completion_cost(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
completion_response=response,
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
@ -864,6 +809,7 @@ def test_vertex_ai_embedding_completion_cost(caplog):
# assert False
@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.asyncio
async def test_completion_cost_hidden_params(sync_mode):
@ -949,7 +895,9 @@ def test_vertex_ai_mistral_predict_cost(usage):
assert predictive_cost > 0
@pytest.mark.parametrize("model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"])
@pytest.mark.parametrize(
"model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"]
)
def test_completion_cost_tts(model):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
@ -1225,7 +1173,10 @@ def test_get_model_params_fireworks_ai(model, base_model):
@pytest.mark.parametrize(
"model",
["fireworks_ai/llama-v3p1-405b-instruct", "fireworks_ai/llama4-maverick-instruct-basic"],
[
"fireworks_ai/llama-v3p1-405b-instruct",
"fireworks_ai/llama4-maverick-instruct-basic",
],
)
def test_completion_cost_fireworks_ai(model):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
@ -2862,6 +2813,7 @@ def test_cost_calculator_with_custom_pricing():
@pytest.mark.asyncio
async def test_cost_calculator_with_custom_pricing_router(model_item, custom_pricing):
from litellm import Router
if custom_pricing == "litellm_params":
model_item["litellm_params"]["input_cost_per_token"] = 0.0000008
model_item["litellm_params"]["output_cost_per_token"] = 0.0000032

View file

@ -554,91 +554,6 @@ async def test_async_chat_openai_stream_options():
pytest.fail(f"An exception occurred: {str(e)}")
## Test Bedrock + sync
def test_chat_bedrock_stream():
try:
customHandler = CompletionCustomHandler()
litellm.callbacks = [customHandler]
response = litellm.completion(
model="bedrock/anthropic.claude-v2",
messages=[{"role": "user", "content": "Hi 👋 - i'm sync bedrock"}],
)
# test streaming
response = litellm.completion(
model="bedrock/anthropic.claude-v2",
messages=[{"role": "user", "content": "Hi 👋 - i'm sync bedrock"}],
stream=True,
)
for chunk in response:
continue
# test failure callback
try:
response = litellm.completion(
model="bedrock/anthropic.claude-v2",
messages=[{"role": "user", "content": "Hi 👋 - i'm sync bedrock"}],
aws_region_name="my-bad-region",
stream=True,
)
for chunk in response:
continue
except Exception:
pass
time.sleep(1)
print(f"customHandler.errors: {customHandler.errors}")
assert len(customHandler.errors) == 0
litellm.callbacks = []
except Exception as e:
pytest.fail(f"An exception occurred: {str(e)}")
# test_chat_bedrock_stream()
## Test Bedrock + Async
@pytest.mark.asyncio
async def test_async_chat_bedrock_stream():
try:
litellm.set_verbose = True
customHandler = CompletionCustomHandler()
litellm.callbacks = [customHandler]
response = await litellm.acompletion(
model="bedrock/anthropic.claude-v2",
messages=[{"role": "user", "content": "Hi 👋 - i'm async bedrock"}],
)
# test streaming
response = await litellm.acompletion(
model="bedrock/anthropic.claude-v2",
messages=[{"role": "user", "content": "Hi 👋 - i'm async bedrock"}],
stream=True,
)
print(f"response: {response}")
async for chunk in response:
print(f"chunk: {chunk}")
continue
await asyncio.sleep(1)
## test failure callback
try:
response = await litellm.acompletion(
model="bedrock/anthropic.claude-v2",
messages=[{"role": "user", "content": "Hi 👋 - i'm async bedrock"}],
aws_region_name="my-bad-key",
stream=True,
)
async for chunk in response:
continue
await asyncio.sleep(1)
except Exception:
pass
await asyncio.sleep(1)
print(f"customHandler.errors: {customHandler.errors}")
assert len(customHandler.errors) == 0
litellm.callbacks = []
except Exception as e:
pytest.fail(f"An exception occurred: {str(e)}")
# asyncio.run(test_async_chat_bedrock_stream())

Some files were not shown because too many files have changed in this diff Show more