Merge pull request #19210 from BerriAI/main

merge main in bedrock passthrough
This commit is contained in:
Sameer Kankute 2026-01-16 17:00:11 +05:30 committed by GitHub
commit b6aa05df16
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
158 changed files with 7202 additions and 2640 deletions

View file

@ -1,4 +0,0 @@
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Hello, how are you?"}]}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "What is the weather today?"}]}}
{"custom_id": "request-3", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Tell me a short joke"}]}}

View file

@ -1,3 +1,3 @@
ignore:
- vulnerability: CVE-2019-1010022
reason: no fixed glibc package is available yet in the Wolfi repositories, so this is ignored temporarily until an upstream release exists
- vulnerability: CVE-2026-22184
reason: no fixed zlib package is available yet in the Wolfi repositories, so this is ignored temporarily until an upstream release exists

View file

@ -129,11 +129,14 @@ run_grype_scans() {
"CVE-2025-13836" # Python 3.13 HTTP response reading OOM/DoS - no fix available in base image
"CVE-2025-12084" # Python 3.13 xml.dom.minidom quadratic algorithm - no fix available in base image
"CVE-2025-60876" # BusyBox wget HTTP request splitting - no fix available in Chainguard Wolfi base image
"CVE-2026-0861" # Wolfi glibc still flagged even on 2.42-r5; upstream patched build unavailable yet
"CVE-2010-4756" # glibc glob DoS - awaiting patched Wolfi glibc build
"CVE-2019-1010022" # glibc stack guard bypass - awaiting patched Wolfi glibc build
"CVE-2019-1010023" # glibc ldd remap issue - awaiting patched Wolfi glibc build
"CVE-2019-1010024" # glibc ASLR mitigation bypass - awaiting patched Wolfi glibc build
"CVE-2019-1010025" # glibc pthread heap address leak - awaiting patched Wolfi glibc build
"CVE-2026-22184" # zlib untgz buffer overflow - untgz unused + no fixed Wolfi build yet
"GHSA-58pv-8j8x-9vj2" # jaraco.context path traversal - setuptools vendored only (v5.3.0), not used in application code (using v6.1.0+)
)
# Build JSON array of allowlisted CVE IDs for jq

View file

@ -0,0 +1,195 @@
# Claude Code with LiteLLM Quickstart
This guide shows how to call Claude models (and any LiteLLM-supported model) through LiteLLM proxy from Claude Code.
> **Note:** This integration is based on [Anthropic's official LiteLLM configuration documentation](https://docs.anthropic.com/en/docs/claude-code/llm-gateway#litellm-configuration). It allows you to use any LiteLLM supported model through Claude Code with centralized authentication, usage tracking, and cost controls.
## Video Walkthrough
Watch the full tutorial: https://www.loom.com/embed/3c17d683cdb74d36a3698763cc558f56
## Prerequisites
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code/overview) installed
- API keys for your chosen providers
## Installation
First, install LiteLLM with proxy support:
```bash
pip install 'litellm[proxy]'
```
## Step 1: Setup config.yaml
Create a secure configuration using environment variables:
```yaml
model_list:
# Claude models
- model_name: claude-3-5-sonnet-20241022
litellm_params:
model: anthropic/claude-3-5-sonnet-20241022
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: claude-3-5-haiku-20241022
litellm_params:
model: anthropic/claude-3-5-haiku-20241022
api_key: os.environ/ANTHROPIC_API_KEY
litellm_settings:
master_key: os.environ/LITELLM_MASTER_KEY
```
Set your environment variables:
```bash
export ANTHROPIC_API_KEY="your-anthropic-api-key"
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
```
## Step 2: Start Proxy
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
## Step 3: Verify Setup
Test that your proxy is working correctly:
```bash
curl -X POST http://0.0.0.0:4000/v1/messages \
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "claude-3-5-sonnet-20241022",
"max_tokens": 1000,
"messages": [{"role": "user", "content": "What is the capital of France?"}]
}'
```
## Step 4: Configure Claude Code
### Method 1: Unified Endpoint (Recommended)
Configure Claude Code to use LiteLLM's unified endpoint. Either a virtual key or master key can be used here:
```bash
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000"
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
```
> **Tip:** LITELLM_MASTER_KEY gives Claude access to all proxy models, whereas a virtual key would be limited to the models set in the UI.
### Method 2: Provider-specific Pass-through Endpoint
Alternatively, use the Anthropic pass-through endpoint:
```bash
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000/anthropic"
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
```
## Step 5: Use Claude Code
Start Claude Code and it will automatically use your configured models:
```bash
# Claude Code will use the models configured in your LiteLLM proxy
claude
# Or specify a model if you have multiple configured
claude --model claude-3-5-sonnet-20241022
claude --model claude-3-5-haiku-20241022
```
## Troubleshooting
Common issues and solutions:
**Claude Code not connecting:**
- Verify your proxy is running: `curl http://0.0.0.0:4000/health`
- Check that `ANTHROPIC_BASE_URL` is set correctly
- Ensure your `ANTHROPIC_AUTH_TOKEN` matches your LiteLLM master key
**Authentication errors:**
- Verify your environment variables are set: `echo $LITELLM_MASTER_KEY`
- Check that your API keys are valid and have sufficient credits
- Ensure the `ANTHROPIC_AUTH_TOKEN` matches your LiteLLM master key
**Model not found:**
- Ensure the model name in Claude Code matches exactly with your `config.yaml`
- Check LiteLLM logs for detailed error messages
## Using Multiple Models and Providers
Expand your configuration to support multiple providers and models:
```yaml
model_list:
# OpenAI models
- model_name: codex-mini
litellm_params:
model: openai/codex-mini
api_key: os.environ/OPENAI_API_KEY
api_base: https://api.openai.com/v1
- model_name: o3-pro
litellm_params:
model: openai/o3-pro
api_key: os.environ/OPENAI_API_KEY
api_base: https://api.openai.com/v1
- model_name: gpt-4o
litellm_params:
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
api_base: https://api.openai.com/v1
# Anthropic models
- model_name: claude-3-5-sonnet-20241022
litellm_params:
model: anthropic/claude-3-5-sonnet-20241022
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: claude-3-5-haiku-20241022
litellm_params:
model: anthropic/claude-3-5-haiku-20241022
api_key: os.environ/ANTHROPIC_API_KEY
# AWS Bedrock
- model_name: claude-bedrock
litellm_params:
model: bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: us-east-1
litellm_settings:
master_key: os.environ/LITELLM_MASTER_KEY
```
Switch between models seamlessly:
```bash
# Use Claude for complex reasoning
claude --model claude-3-5-sonnet-20241022
# Use Haiku for fast responses
claude --model claude-3-5-haiku-20241022
# Use Bedrock deployment
claude --model claude-bedrock
```
## Additional Resources
- [LiteLLM Documentation](https://docs.litellm.ai/)
- [Claude Code Documentation](https://docs.anthropic.com/en/docs/claude-code/overview)
- [Anthropic's LiteLLM Configuration Guide](https://docs.anthropic.com/en/docs/claude-code/llm-gateway#litellm-configuration)

View file

@ -0,0 +1,98 @@
[{
"title": "Claude Code Quickstart",
"description": "This is a quickstart guide to using Claude Code with LiteLLM.",
"url": "https://docs.litellm.ai/docs/tutorials/claude_responses_api",
"date": "2026-01-15",
"version": "1.0.0",
"tags": [
"Claude Code",
"LiteLLM"
]
},
{
"title": "Claude Code with MCPs",
"description": "This is a guide to using Claude Code with MCPs via LiteLLM Proxy.",
"url": "https://docs.litellm.ai/docs/tutorials/claude_mcp",
"date": "2026-01-15",
"version": "1.0.0",
"tags": [
"Claude Code",
"LiteLLM",
"MCP"
]
},
{
"title": "Claude Code with Non-Anthropic Models",
"description": "This is a guide to using Claude Code with non-Anthropic models via LiteLLM Proxy.",
"url": "https://docs.litellm.ai/docs/tutorials/claude_non_anthropic_models",
"date": "2026-01-16",
"version": "1.0.0",
"tags": [
"Claude Code",
"LiteLLM",
"OpenAI",
"Gemini"
]
},
{
"title": "Cursor Quickstart",
"description": "This is a quickstart guide to using Cursor with LiteLLM.",
"url": "https://docs.litellm.ai/docs/tutorials/cursor_integration",
"date": "2026-01-16",
"version": "1.0.0",
"tags": [
"Cursor",
"LiteLLM",
"Quickstart"
]
},
{
"title": "Github Copilot Quickstart",
"description": "This is a quickstart guide to using Github Copilot with LiteLLM.",
"url": "https://docs.litellm.ai/docs/tutorials/github_copilot_integration",
"date": "2026-01-16",
"version": "1.0.0",
"tags": [
"Github Copilot",
"LiteLLM",
"Quickstart"
]
},
{
"title": "LiteLLM Gemini CLI Quickstart",
"description": "This is a quickstart guide to using LiteLLM Gemini CLI.",
"url": "https://docs.litellm.ai/docs/tutorials/litellm_gemini_cli",
"date": "2026-01-16",
"version": "1.0.0",
"tags": [
"Gemini CLI",
"Gemini",
"LiteLLM",
"Quickstart"
]
},
{
"title": "OpenAI Codex CLI Quickstart",
"description": "This is a quickstart guide to using OpenAI Codex CLI.",
"url": "https://docs.litellm.ai/docs/tutorials/openai_codex",
"date": "2026-01-16",
"version": "1.0.0",
"tags": [
"OpenAI Codex CLI",
"OpenAI",
"LiteLLM",
"Quickstart"
]
},
{
"title": "OpenWebUI Quickstart",
"description": "This is a quickstart guide to using OpenWebUI with LiteLLM.",
"url": "https://docs.litellm.ai/docs/tutorials/openweb_ui",
"date": "2026-01-16",
"version": "1.0.0",
"tags": [
"OpenWebUI",
"LiteLLM",
"Quickstart"
]
}]

View file

@ -170,7 +170,8 @@ spec:
{{- toYaml .Values.resources | nindent 12 }}
volumeMounts:
- name: litellm-config
mountPath: /etc/litellm/
mountPath: /etc/litellm/config.yaml
subPath: config.yaml
{{ if .Values.securityContext.readOnlyRootFilesystem }}
- name: tmp
mountPath: /tmp

View file

@ -136,7 +136,8 @@ tests:
path: spec.template.spec.containers[0].volumeMounts
content:
name: litellm-config
mountPath: /etc/litellm/
mountPath: /etc/litellm/config.yaml
subPath: config.yaml
- it: should work with lifecycle hooks
template: deployment.yaml
set:

View file

@ -15,7 +15,7 @@ import TabItem from '@theme/TabItem';
| Fallbacks | ✅ | Works between supported models |
| Loadbalancing | ✅ | Works between supported models |
| Guardrails | ✅ | Applies to input prompts (non-streaming only) |
| Supported Providers | OpenAI, Azure, Google AI Studio, Vertex AI, AWS Bedrock, Recraft, Xinference, Nscale | |
| Supported Providers | OpenAI, Azure, Google AI Studio, Vertex AI, AWS Bedrock, Recraft, OpenRouter, Xinference, Nscale | |
## Quick Start
@ -238,6 +238,27 @@ print(response)
See Recraft usage with LiteLLM [here](./providers/recraft.md#image-generation)
## OpenRouter Image Generation Models
Use this for image generation models available through OpenRouter (e.g., Google Gemini image generation models)
#### Usage
```python showLineNumbers
from litellm import image_generation
import os
os.environ['OPENROUTER_API_KEY'] = "your-api-key"
response = image_generation(
model="openrouter/google/gemini-2.5-flash-image",
prompt="A beautiful sunset over a calm ocean",
size="1024x1024",
quality="high",
)
print(response)
```
## OpenAI Compatible Image Generation Models
Use this for calling `/image_generation` endpoints on OpenAI Compatible Servers, example https://github.com/xorbitsai/inference
@ -301,5 +322,6 @@ print(f"response: {response}")
| Vertex AI | [Vertex AI Image Generation →](./providers/vertex_image) |
| AWS Bedrock | [Bedrock Image Generation →](./providers/bedrock) |
| Recraft | [Recraft Image Generation →](./providers/recraft#image-generation) |
| OpenRouter | [OpenRouter Image Generation →](./providers/openrouter#image-generation) |
| Xinference | [Xinference Image Generation →](./providers/xinference#image-generation) |
| Nscale | [Nscale Image Generation →](./providers/nscale#image-generation) |

View file

@ -40,6 +40,10 @@ import os
# from https://logfire.pydantic.dev/
os.environ["LOGFIRE_TOKEN"] = ""
# Optionally customize the base url
# from https://logfire.pydantic.dev/
os.environ["LOGFIRE_BASE_URL"] = ""
# LLM API Keys
os.environ['OPENAI_API_KEY']=""

View file

@ -93,3 +93,120 @@ response = embedding(
)
print(response)
```
## Image Generation
OpenRouter supports image generation through select models like Google Gemini image generation models. LiteLLM transforms standard image generation requests to OpenRouter's chat completion format.
### Supported Parameters
- `size`: Maps to OpenRouter's `aspect_ratio` format
- `1024x1024``1:1` (square)
- `1536x1024``3:2` (landscape)
- `1024x1536``2:3` (portrait)
- `1792x1024``16:9` (wide landscape)
- `1024x1792``9:16` (tall portrait)
- `quality`: Maps to OpenRouter's `image_size` format (Gemini models)
- `low` or `standard``1K`
- `medium``2K`
- `high` or `hd``4K`
- `n`: Number of images to generate
### Usage
```python
from litellm import image_generation
import os
os.environ["OPENROUTER_API_KEY"] = "your-api-key"
# Basic image generation
response = image_generation(
model="openrouter/google/gemini-2.5-flash-image",
prompt="A beautiful sunset over a calm ocean",
)
print(response)
```
### Advanced Usage with Parameters
```python
from litellm import image_generation
import os
os.environ["OPENROUTER_API_KEY"] = "your-api-key"
# Generate high-quality landscape image
response = image_generation(
model="openrouter/google/gemini-2.5-flash-image",
prompt="A serene mountain landscape with a lake",
size="1536x1024", # Landscape format
quality="high", # High quality (4K)
)
# Access the generated image
image_data = response.data[0]
if image_data.b64_json:
# Base64 encoded image
print(f"Generated base64 image: {image_data.b64_json[:50]}...")
elif image_data.url:
# Image URL
print(f"Generated image URL: {image_data.url}")
```
### Using OpenRouter-Specific Parameters
You can also pass OpenRouter-specific parameters directly using `image_config`:
```python
from litellm import image_generation
import os
os.environ["OPENROUTER_API_KEY"] = "your-api-key"
response = image_generation(
model="openrouter/google/gemini-2.5-flash-image",
prompt="A futuristic cityscape at night",
image_config={
"aspect_ratio": "16:9", # OpenRouter native format
"image_size": "4K" # OpenRouter native format
}
)
print(response)
```
### Response Format
The response follows the standard LiteLLM ImageResponse format:
```python
{
"created": 1703658209,
"data": [{
"b64_json": "iVBORw0KGgoAAAANSUhEUgAA...", # Base64 encoded image
"url": None,
"revised_prompt": None
}],
"usage": {
"input_tokens": 10,
"output_tokens": 1290,
"total_tokens": 1300
}
}
```
### Cost Tracking
OpenRouter provides cost information in the response, which LiteLLM automatically tracks:
```python
response = image_generation(
model="openrouter/google/gemini-2.5-flash-image",
prompt="A cute baby sea otter",
)
# Cost is available in the response metadata
print(f"Request cost: ${response._hidden_params['additional_headers']['llm_provider-x-litellm-response-cost']}")
```

View file

@ -12,100 +12,340 @@ LiteLLM supports SAP Generative AI Hub's Orchestration Service.
| Supported Endpoints | `/chat/completions`, `/embeddings` |
| API Reference | [SAP AI Core Documentation](https://help.sap.com/docs/sap-ai-core) |
## Prerequisites
Before you begin, ensure you have:
1. **SAP BTP Account** with access to SAP AI Core
2. **AI Core Service Instance** provisioned in your subaccount
3. **Service Key** created for your AI Core instance (this contains your credentials)
4. **Resource Group** with deployed AI models (check with your SAP administrator)
:::tip Where to Find Your Credentials
Your credentials come from the **Service Key** you create in SAP BTP Cockpit:
1. Navigate to your **Subaccount** → **Instances and Subscriptions**
2. Find your **AI Core** instance and click on it
3. Go to **Service Keys** and create one (or use existing)
4. The JSON contains all values needed below
The service key JSON looks like this:
```json
{
"clientid": "sb-abc123...",
"clientsecret": "xyz789...",
"url": "https://myinstance.authentication.eu10.hana.ondemand.com",
"serviceurls": {
"AI_API_URL": "https://api.ai.prod.eu-central-1.aws.ml.hana.ondemand.com"
}
}
```
:::info Resource Group
The resource group is typically configured separately in your AI Core deployment, not in the service key itself. You can set it via the `AICORE_RESOURCE_GROUP` environment variable (defaults to "default").
:::
## Quick Start
### Step 1: Install LiteLLM
```bash
pip install litellm
```
### Step 2: Set Your Credentials
Choose **one** of these authentication methods:
<Tabs>
<TabItem value="service-key" label="Service Key JSON (Recommended)">
The simplest approach - paste your entire service key as a single environment variable. The service key must be wrapped in a `credentials` object:
```bash
export AICORE_SERVICE_KEY='{
"credentials": {
"clientid": "your-client-id",
"clientsecret": "your-client-secret",
"url": "https://<your-instance>.authentication.sap.hana.ondemand.com",
"serviceurls": {
"AI_API_URL": "https://api.ai.<your-region>.aws.ml.hana.ondemand.com"
}
}
}'
export AICORE_RESOURCE_GROUP="default"
```
</TabItem>
<TabItem value="individual" label="Individual Variables">
Alternatively, instead of using the service key above, you could set each credential separately:
```bash
export AICORE_AUTH_URL="https://<your-instance>.authentication.sap.hana.ondemand.com/oauth/token"
export AICORE_CLIENT_ID="your-client-id"
export AICORE_CLIENT_SECRET="your-client-secret"
export AICORE_RESOURCE_GROUP="default"
export AICORE_BASE_URL="https://api.ai.<your-region>.aws.ml.hana.ondemand.com/v2"
```
</TabItem>
</Tabs>
### Step 3: Make Your First Request
```python title="test_sap.py"
from litellm import completion
response = completion(
model="sap/gpt-4o",
messages=[{"role": "user", "content": "Hello from LiteLLM!"}]
)
print(response.choices[0].message.content)
```
Run it:
```bash
python test_sap.py
```
**Expected output:**
```text
Hello! How can I assist you today?
```
### Step 4: Verify Your Setup (Optional)
Test that everything is working with this diagnostic script:
```python title="verify_sap_setup.py"
import os
import litellm
# Enable debug logging to see what's happening
import os
os.environ["LITELLM_LOG"] = "DEBUG"
# Either use AICORE_SERVICE_KEY (contains all credentials including resourcegroup)
# OR use individual variables (all required together)
individual_vars = ["AICORE_AUTH_URL", "AICORE_CLIENT_ID", "AICORE_CLIENT_SECRET", "AICORE_BASE_URL", "AICORE_RESOURCE_GROUP"]
print("=== SAP Gen AI Hub Setup Verification ===\n")
# Check for service key method
if os.environ.get("AICORE_SERVICE_KEY"):
print("✓ Using AICORE_SERVICE_KEY authentication (includes resource group)")
else:
# Check individual variables
missing = [v for v in individual_vars if not os.environ.get(v)]
if missing:
print(f"✗ Missing environment variables: {missing}")
else:
print("✓ Using individual variable authentication")
print(f"✓ Resource group: {os.environ.get('AICORE_RESOURCE_GROUP')}")
# Test API connection
print("\n=== Testing API Connection ===\n")
try:
response = litellm.completion(
model="sap/gpt-4o",
messages=[{"role": "user", "content": "Say 'Connection successful!' and nothing else."}],
max_tokens=20
)
print(f"✓ API Response: {response.choices[0].message.content}")
print("\n🎉 Setup complete! You're ready to use SAP Gen AI Hub with LiteLLM.")
except Exception as e:
print(f"✗ API Error: {e}")
print("\nTroubleshooting tips:")
print(" 1. Verify your service key credentials are correct")
print(" 2. Check that 'gpt-4o' is deployed in your resource group")
print(" 3. Ensure your SAP AI Core instance is running")
```
Run the verification:
```bash
python verify_sap_setup.py
```
**Expected output on success:**
```text
=== SAP Gen AI Hub Setup Verification ===
✓ Using AICORE_SERVICE_KEY authentication
✓ Resource group: default
=== Testing API Connection ===
✓ API Response: Connection successful!
🎉 Setup complete! You're ready to use SAP Gen AI Hub with LiteLLM.
```
## Authentication
SAP Generative AI Hub uses service key authentication. You can provide credentials via:
SAP Generative AI Hub uses OAuth2 service keys for authentication. See [Quick Start](#quick-start) for setup instructions.
1. **Environment variable** - Set `AICORE_SERVICE_KEY` with your service key JSON
2. **Direct parameter** - Pass `api_key` with the service key JSON string
### Environment Variables Reference
```python showLineNumbers title="Environment Variable"
import os
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
| Variable | Required | Description |
|----------|----------|-------------|
| `AICORE_SERVICE_KEY` | Yes* | Complete service key JSON (recommended method) |
| `AICORE_RESOURCE_GROUP` | Yes | Your AI Core resource group name |
| `AICORE_AUTH_URL` | Yes* | OAuth token URL (alternative to service key) |
| `AICORE_CLIENT_ID` | Yes* | OAuth client ID (alternative to service key) |
| `AICORE_CLIENT_SECRET` | Yes* | OAuth client secret (alternative to service key) |
| `AICORE_BASE_URL` | Yes* | AI Core API base URL (alternative to service key) |
*Choose either `AICORE_SERVICE_KEY` OR the individual variables (`AICORE_AUTH_URL`, `AICORE_CLIENT_ID`, `AICORE_CLIENT_SECRET`, `AICORE_BASE_URL`).
## Model Naming Conventions
Understanding model naming is crucial for using SAP Gen AI Hub correctly. The naming pattern differs depending on whether you're using the SDK directly or through the proxy.
### Direct SDK Usage
When calling LiteLLM's SDK directly, you **must** include the `sap/` prefix in the model name:
```python
# Correct - includes sap/ prefix
model="sap/gpt-4o"
model="sap/anthropic--claude-4.5-sonnet"
model="sap/gemini-2.5-pro"
# Incorrect - missing prefix
model="gpt-4o" # ❌ Won't work
```
3. **Environment variables** - Set the following list of credentials in .env file
<pre>
AICORE_AUTH_URL = "https://* * * .authentication.sap.hana.ondemand.com/oauth/token",
AICORE_CLIENT_ID = " *** ",
AICORE_CLIENT_SECRET = " *** ",
AICORE_RESOURCE_GROUP = " *** ",
AICORE_BASE_URL = "https://api.ai.***.cfapps.sap.hana.ondemand.com/v2"
</pre>
## Usage - LiteLLM Python SDK
```python showLineNumbers title="SAP Chat Completion"
from litellm import completion
import os
### Proxy Usage
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
When using the LiteLLM Proxy, you use the **friendly `model_name`** defined in your configuration. The proxy automatically handles the `sap/` prefix routing.
response = completion(
model="sap/gpt-4",
messages=[{"role": "user", "content": "Hello from LiteLLM"}]
```yaml
# In config.yaml, define the mapping
model_list:
- model_name: gpt-4o # ← Use this name in client requests
litellm_params:
model: sap/gpt-4o # ← Proxy handles the sap/ prefix
```
```python
# Client request - no sap/ prefix needed
client.chat.completions.create(
model="gpt-4o", # ✓ Correct for proxy usage
messages=[...]
)
print(response)
```
```python showLineNumbers title="SAP Chat Completion - Streaming"
### Anthropic Models Special Syntax
Anthropic models use a double-dash (`--`) prefix convention:
| Provider | Model Example | LiteLLM Format |
|----------|---------------|----------------|
| OpenAI | GPT-4o | `sap/gpt-4o` |
| Anthropic | Claude 4.5 Sonnet | `sap/anthropic--claude-4.5-sonnet` |
| Google | Gemini 2.5 Pro | `sap/gemini-2.5-pro` |
| Mistral | Mistral Large | `sap/mistral-large` |
### Quick Reference Table
| Usage Type | Model Format | Example |
|------------|--------------|---------|
| Direct SDK | `sap/<model-name>` | `sap/gpt-4o` |
| Direct SDK (Anthropic) | `sap/anthropic--<model>` | `sap/anthropic--claude-4.5-sonnet` |
| Proxy Client | `<friendly-name>` | `gpt-4o` or `claude-sonnet` |
## Using the Python SDK
The LiteLLM Python SDK automatically detects your authentication method. Simply set your environment variables and make requests.
```python showLineNumbers title="Basic Completion"
from litellm import completion
import os
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
# Assumes AICORE_AUTH_URL, AICORE_CLIENT_ID, etc. are set
response = completion(
model="sap/gpt-4",
messages=[{"role": "user", "content": "Hello from LiteLLM"}],
stream=True
model="sap/anthropic--claude-4.5-sonnet",
messages=[{"role": "user", "content": "Explain quantum computing"}]
)
for chunk in response:
print(chunk.choices[0].delta.content or "", end="")
print(response.choices[0].message.content)
```
```python showLineNumbers title="SAP Embedding"
from litellm import embedding
import os
Both authentication methods (individual variables or service key JSON) work automatically - no code changes required.
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
## Using the Proxy Server
result = embedding(
model="sap/text-embedding-3-small",
input="Answer to the ultimate question of life, the universe, and everything is 42")
print(result.data[0])
```
The LiteLLM Proxy provides a unified OpenAI-compatible API for your SAP models.
## Usage - LiteLLM Proxy
### Configuration
Add to your LiteLLM Proxy config:
Create a `config.yaml` file in your project directory with your model mappings and credentials:
```yaml showLineNumbers title="config.yaml"
model_list:
- model_name: "sap/*"
# OpenAI models
- model_name: gpt-5
litellm_params:
model: "sap/*"
model: sap/gpt-5
general_settings:
master_key: your-proxy-api-key
# Anthropic models (note the double-dash)
- model_name: claude-sonnet
litellm_params:
model: sap/anthropic--claude-4.5-sonnet
- model_name: claude-opus
litellm_params:
model: sap/anthropic--claude-4.5-opus
# Embeddings
- model_name: text-embedding-3-small
litellm_params:
model: sap/text-embedding-3-small
litellm_settings:
drop_params: true
set_verbose: false
request_timeout: 600
num_retries: 2
forward_client_headers_to_llm_api: ["anthropic-version"]
general_settings:
master_key: "sk-1234" # Enter here your desired master key starting with 'sk-'.
# UI Admin is not required but helpful including the management of keys for your team(s). If you are using a database, these parameters are required:
database_url: "Enter you database URL."
UI_USERNAME: "Your desired UI admin account name"
UI_PASSWORD: "Your desired and strong pwd"
# Authentication
environment_variables:
AICORE_SERVICE_KEY: '{"clientid": "...", "clientsecret": "...", ...}'
AICORE_SERVICE_KEY: '{"credentials": {"clientid": "...", "clientsecret": "...", "url": "...", "serviceurls": {"AI_API_URL": "..."}}}'
AICORE_RESOURCE_GROUP: "default"
```
Start the proxy:
### Starting the Proxy
```bash showLineNumbers title="Start Proxy"
litellm --config config.yaml
```
The proxy will start on `http://localhost:4000` by default.
### Making Requests
<Tabs>
<TabItem value="curl" label="cURL">
```bash showLineNumbers title="Test Request"
curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer your-proxy-api-key" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "sap/gpt-4",
"model": "gpt-4o",
"messages": [{"role": "user", "content": "Hello"}]
}'
```
@ -118,11 +358,11 @@ from openai import OpenAI
client = OpenAI(
base_url="http://localhost:4000",
api_key="your-proxy-api-key"
api_key="sk-1234"
)
response = client.chat.completions.create(
model="sap/gpt-4",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello"}]
)
print(response.choices[0].message.content)
@ -134,12 +374,14 @@ print(response.choices[0].message.content)
```python showLineNumbers title="LiteLLM SDK"
import os
import litellm
os.environ["LITELLM_PROXY_API_KEY"] = "your-proxy-api-key"
litellm.use_litellm_proxy = True # it is important to set this parameter
os.environ["LITELLM_PROXY_API_KEY"] = "sk-1234"
litellm.use_litellm_proxy = True
response = litellm.completion(
model="sap/gpt-4o",
messages=[{ "content": "Hello, how are you?","role": "user"}],
api_base="http://your-proxy-api-base"
model="claude-sonnet",
messages=[{"content": "Hello, how are you?", "role": "user"}],
api_base="http://localhost:4000"
)
print(response)
@ -148,15 +390,170 @@ print(response)
</TabItem>
</Tabs>
## Supported Parameters
## Features
| Parameter | Description |
|-----------|-------------|
| `temperature` | Controls randomness |
| `max_tokens` | Maximum tokens in response |
| `top_p` | Nucleus sampling |
| `tools` | Function calling tools |
| `tool_choice` | Tool selection behavior |
| `response_format` | Output format (json_object, json_schema) |
| `stream` | Enable streaming |
### Streaming Responses
Stream responses in real-time for better user experience:
```python showLineNumbers title="Streaming Chat Completion"
from litellm import completion
response = completion(
model="sap/gpt-4o",
messages=[{"role": "user", "content": "Count from 1 to 10"}],
stream=True
)
for chunk in response:
if chunk.choices[0].delta.content:
print(chunk.choices[0].delta.content, end="", flush=True)
```
### Structured Output
#### JSON Schema (Recommended)
Use JSON Schema for structured output with strict validation:
```python showLineNumbers title="JSON Schema Response"
from litellm import completion
response = completion(
model="sap/gpt-4o",
messages=[{
"role": "user",
"content": "Generate info about Tokyo"
}],
response_format={
"type": "json_schema",
"json_schema": {
"name": "city_info",
"schema": {
"type": "object",
"properties": {
"name": {"type": "string"},
"population": {"type": "number"},
"country": {"type": "string"}
},
"required": ["name", "population", "country"],
"additionalProperties": False
},
"strict": True
}
}
)
print(response.choices[0].message.content)
# Output: {"name":"Tokyo","population":37000000,"country":"Japan"}
```
#### JSON Object Format
For flexible JSON output without schema validation:
```python showLineNumbers title="JSON Object Response"
from litellm import completion
response = completion(
model="sap/gpt-4o",
messages=[{
"role": "user",
"content": "Generate a person object in JSON format with name and age"
}],
response_format={"type": "json_object"}
)
print(response.choices[0].message.content)
```
:::note SAP Platform Requirement
When using `json_object` type, SAP's orchestration service requires the word "json" to appear in your prompt. This ensures explicit intent for JSON formatting. For schema-validated output without this requirement, use `json_schema` instead (recommended).
:::
### Multi-turn Conversations
Maintain conversation context across multiple turns:
```python showLineNumbers title="Multi-turn Conversation"
from litellm import completion
response = completion(
model="sap/gpt-4o",
messages=[
{"role": "user", "content": "My name is Alice"},
{"role": "assistant", "content": "Hello Alice! Nice to meet you."},
{"role": "user", "content": "What is my name?"}
]
)
print(response.choices[0].message.content)
# Output: Your name is Alice.
```
### Embeddings
Generate vector embeddings for semantic search and retrieval:
```python showLineNumbers title="Create Embeddings"
from litellm import embedding
response = embedding(
model="sap/text-embedding-3-small",
input=["Hello world", "Machine learning is fascinating"]
)
print(response.data[0]["embedding"]) # Vector representation
```
## Reference
### Supported Parameters
| Parameter | Type | Description |
|-----------|------|-------------|
| `model` | string | Model identifier (with `sap/` prefix for SDK) |
| `messages` | array | Conversation messages |
| `temperature` | float | Controls randomness (0-2) |
| `max_tokens` | integer | Maximum tokens in response |
| `top_p` | float | Nucleus sampling threshold |
| `stream` | boolean | Enable streaming responses |
| `response_format` | object | Output format (`json_object`, `json_schema`) |
| `tools` | array | Function calling tool definitions |
| `tool_choice` | string/object | Tool selection behavior |
### Supported Models
For the complete and up-to-date list of available models provided by SAP Gen AI Hub, please refer to the [SAP AI Core Generative AI Hub documentation](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/models-and-scenarios-in-generative-ai-hub).
:::info Model Availability
Model availability varies by SAP deployment region and your subscription. Contact your SAP administrator to confirm which models are available in your environment.
:::
### Troubleshooting
**Authentication Errors**
If you receive authentication errors:
1. Verify all required environment variables are set correctly
2. Check that your service key hasn't expired
3. Confirm your resource group has access to the desired models
4. Ensure the `AICORE_AUTH_URL` and `AICORE_BASE_URL` match your SAP region
**Model Not Found**
If a model returns "not found":
1. Verify the model is available in your SAP deployment
2. Check you're using the correct model name format (`sap/` prefix for SDK)
3. Confirm your resource group has access to that specific model
4. For Anthropic models, ensure you're using the `anthropic--` double-dash prefix
**Rate Limiting**
SAP Gen AI Hub enforces rate limits based on your subscription. If you hit limits:
1. Implement exponential backoff retry logic
2. Consider using the proxy's built-in rate limiting features
3. Contact your SAP administrator to review quota allocations

View file

@ -744,6 +744,7 @@ router_settings:
| LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging
| LITELM_ENVIRONMENT | Environment for LiteLLM Instance. This is currently only logged to DeepEval to determine the environment for DeepEval integration.
| LOGFIRE_TOKEN | Token for Logfire logging service
| LOGFIRE_BASE_URL | Base URL for Logfire logging service (useful for self hosted deployments)
| LOGGING_WORKER_CONCURRENCY | Maximum number of concurrent coroutine slots for the logging worker on the asyncio event loop. Default is 100. Setting too high will flood the event loop with logging tasks which will lower the overall latency of the requests.
| LOGGING_WORKER_MAX_QUEUE_SIZE | Maximum size of the logging worker queue. When the queue is full, the worker aggressively clears tasks to make room instead of dropping logs. Default is 50,000
| LOGGING_WORKER_MAX_TIME_PER_COROUTINE | Maximum time in seconds allowed for each coroutine in the logging worker before timing out. Default is 20.0

View file

@ -9,7 +9,6 @@ LiteLLM provides flexible cost tracking and pricing customization for all LLM pr
- **Custom Pricing** - Override default model costs or set pricing for custom models
- **Cost Per Token** - Track costs based on input/output tokens (most common)
- **Cost Per Second** - Track costs based on runtime (e.g., Sagemaker)
- **Zero-Cost Models** - Bypass budget checks for free/on-premises models by setting costs to 0
- **[Provider Discounts](./provider_discounts.md)** - Apply percentage-based discounts to specific providers
- **[Provider Margins](./provider_margins.md)** - Add fees/margins to LLM costs for internal billing
- **Base Model Mapping** - Ensure accurate cost tracking for Azure deployments
@ -107,51 +106,6 @@ There are other keys you can use to specify costs for different scenarios and mo
These keys evolve based on how new models handle multimodality. The latest version can be found at [https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json).
## Zero-Cost Models (Bypass Budget Checks)
**Use Case**: You have on-premises or free models that should be accessible even when users exceed their budget limits.
**Solution** ✅: Set both `input_cost_per_token` and `output_cost_per_token` to `0` (explicitly) to bypass all budget checks for that model.
:::info
When a model is configured with zero cost, LiteLLM will automatically skip ALL budget checks (user, team, team member, end-user, organization, and global proxy budget) for requests to that model.
**Important**: Both costs must be **explicitly set to 0**. If costs are `null` or undefined, the model will be treated as having cost and budget checks will apply.
:::
### Configuration Example
```yaml
model_list:
# On-premises model - free to use
- model_name: on-prem-llama
litellm_params:
model: ollama/llama3
api_base: http://localhost:11434
model_info:
input_cost_per_token: 0 # 👈 Explicitly set to 0
output_cost_per_token: 0 # 👈 Explicitly set to 0
# Paid cloud model - budget checks apply
- model_name: gpt-4
litellm_params:
model: gpt-4
api_key: os.environ/OPENAI_API_KEY
# No model_info - uses default pricing from cost map
```
### Behavior
With the above configuration:
- **User over budget** → Can still use `on-prem-llama` ✅, but blocked from `gpt-4`
- **Team over budget** → Can still use `on-prem-llama` ✅, but blocked from `gpt-4`
- **End-user over budget** → Can still use `on-prem-llama` ✅, but blocked from `gpt-4`
This ensures your free/on-premises models remain accessible regardless of budget constraints, while paid models are still properly governed.
## Set 'base_model' for Cost Tracking (e.g. Azure deployments)
**Problem**: Azure returns `gpt-4` in the response when `azure/gpt-4-1106-preview` is used. This leads to inaccurate cost tracking

View file

@ -22,19 +22,22 @@ Customer Usage enables you to track spend and usage for individual customers (en
## How to Track Spend
Track customer spend by including a `user` field in your API requests. The customer ID will be automatically tracked and associated with all spend from that request.
Track customer spend by including a `user` field in your API requests or by passing a customer ID header. The customer ID will be automatically tracked and associated with all spend from that request.
### Example using cURL
<Tabs>
<TabItem value="body" label="Request Body" default>
### Using Request Body
Make a `/chat/completions` call with the `user` field containing your customer ID:
```bash showLineNumbers title="Track spend with customer ID"
```bash showLineNumbers title="Track spend with customer ID in body"
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \ # 👈 YOUR PROXY KEY
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "gpt-3.5-turbo",
"user": "customer-123", # 👈 CUSTOMER ID
"user": "customer-123",
"messages": [
{
"role": "user",
@ -44,7 +47,49 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
}'
```
The customer ID (`customer-123`) will be automatically upserted into the database with the new spend. If the customer ID already exists, spend will be incremented.
</TabItem>
<TabItem value="header" label="Request Header">
### Using Request Headers
You can also pass the customer ID via HTTP headers. This is useful for tools that support custom headers but don't allow modifying the request body (like Claude Code with `ANTHROPIC_CUSTOM_HEADERS`).
LiteLLM automatically recognizes these standard headers (no configuration required):
- `x-litellm-customer-id`
- `x-litellm-end-user-id`
```bash showLineNumbers title="Track spend with customer ID in header"
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--header 'x-litellm-customer-id: customer-123' \
--data '{
"model": "gpt-3.5-turbo",
"messages": [
{
"role": "user",
"content": "What is the capital of France?"
}
]
}'
```
#### Using with Claude Code
Claude Code supports custom headers via the `ANTHROPIC_CUSTOM_HEADERS` environment variable. Set it to pass your customer ID:
```bash title="Configure Claude Code with customer tracking"
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000/v1/messages"
export ANTHROPIC_API_KEY="sk-1234"
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-customer-id: my-customer-id"
```
Now all requests from Claude Code will automatically track spend under `my-customer-id`.
</TabItem>
</Tabs>
The customer ID will be automatically upserted into the database with the new spend. If the customer ID already exists, spend will be incremented.
### Example using OpenWebUI

View file

@ -1827,6 +1827,64 @@ This approach allows you to:
- Share callbacks across different environments
- Version control callback files in cloud storage
#### Step 2c - Mounting Custom Callbacks in Helm/Kubernetes (Alternative)
When deploying with Helm or Kubernetes, you can mount custom callback Python files alongside your `config.yaml` using `subPath` to avoid overwriting the config directory.
**The Problem:**
Mounting a volume to a directory (e.g., `/app/`) would normally hide all existing files in that directory, including your `config.yaml`.
**The Solution:**
Use `subPath` in your `volumeMounts` to mount individual files without overwriting the entire directory.
**Example - Helm values.yaml:**
```yaml
# values.yaml
volumes:
- name: callback-files
configMap:
name: litellm-callback-files
volumeMounts:
- name: callback-files
mountPath: /app/custom_callbacks.py # Mount to specific FILE path
subPath: custom_callbacks.py # Required to avoid overwriting directory
```
**Create the ConfigMap with your callback file:**
```yaml
apiVersion: v1
kind: ConfigMap
metadata:
name: litellm-callback-files
data:
custom_callbacks.py: |
from litellm.integrations.custom_logger import CustomLogger
class MyCustomHandler(CustomLogger):
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
print(f"Success! Model: {kwargs.get('model')}")
proxy_handler_instance = MyCustomHandler()
```
**Reference in your config.yaml:**
```yaml
litellm_settings:
callbacks: custom_callbacks.proxy_handler_instance
```
**How it works:**
1. The `subPath` parameter tells Kubernetes to mount only the specific file
2. This places `custom_callbacks.py` in `/app/` alongside your existing `config.yaml`
3. LiteLLM automatically finds the callback file in the same directory as the config
4. No files are overwritten or hidden
**Note:** You can mount multiple callback files by adding more `volumeMounts` entries, each with its own `subPath`.
#### Step 3 - Start proxy + test request
```shell

View file

@ -30,6 +30,9 @@ general_settings:
# Optional: set how frequently cleanup should run - default is daily
maximum_spend_logs_retention_interval: "1d" # Run cleanup daily
# Optional: set exact time for cleanup (Cron syntax)
maximum_spend_logs_cleanup_cron: "0 4 * * *" # Run at 04:00 AM daily
litellm_settings:
cache: true
cache_params:
@ -51,6 +54,15 @@ How long logs should be kept before deletion. Supported formats:
How often the cleanup job should run. Uses the same format as above. If not set, cleanup will run every 24 hours if and only if `maximum_spend_logs_retention_period` is set.
#### `maximum_spend_logs_cleanup_cron` (optional)
Schedule the cleanup using standard cron syntax. This takes precedence over `maximum_spend_logs_retention_interval`.
Examples:
- `"0 4 * * *"` Run at 04:00 AM daily
- `"0 0 * * 0"` Run at midnight every Sunday
- `"*/30 * * * *"` Run every 30 minutes
## How it works
### Step 1. Lock Acquisition (Optional with Redis)

View file

@ -0,0 +1,99 @@
# Claude Code - Granular Cost Tracking
Track Claude Code usage by customer or tags using LiteLLM proxy. This enables granular cost attribution for billing, budgeting, and analytics.
## How It Works
Claude Code supports custom headers via `ANTHROPIC_CUSTOM_HEADERS`. LiteLLM automatically tracks requests with specific headers for cost attribution.
## Tracking Options
Choose how you want to attribute costs:
| Track By | Header | Use Case |
|----------|--------|----------|
| Customer | `x-litellm-customer-id` | Bill customers, per-user budgets |
| Tags | `x-litellm-tags` | Project tracking, cost centers, environments |
## Environment Variables
| Variable | Description | Example |
|----------|-------------|---------|
| `ANTHROPIC_BASE_URL` | LiteLLM proxy URL | `http://localhost:4000` |
| `ANTHROPIC_API_KEY` | LiteLLM API key | `sk-1234` |
| `ANTHROPIC_CUSTOM_HEADERS` | Custom headers (`header-name: value` format) | See examples below |
## Option 1: Track by Customer
Use this to attribute costs to specific customers or end-users.
```bash
export ANTHROPIC_BASE_URL=http://localhost:4000
export ANTHROPIC_API_KEY=sk-1234
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-customer-id: claude-ishaan-local"
```
## Option 2: Track by Tags
Use this to attribute costs to projects, cost centers, or environments. Pass comma-separated tags.
```bash
export ANTHROPIC_BASE_URL=http://localhost:4000
export ANTHROPIC_API_KEY=sk-1234
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-tags: project:acme,env:prod,team:backend"
```
## Quick Start
### 1. Set Environment Variables
```bash
export ANTHROPIC_BASE_URL=http://localhost:4000
export ANTHROPIC_API_KEY=sk-1234
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-customer-id: claude-ishaan-local"
```
### 2. Use Claude Code
```bash
claude
```
All requests will now be tracked under the customer ID `claude-ishaan-local`.
![](https://colony-recorder.s3.amazonaws.com/files/2026-01-16/8f45872e-2d00-4d01-bf3d-4d6ae11d1396/ascreenshot_d2a745b8da4f4a56aaf2cac02871ef53_text_export.jpeg)
![](https://colony-recorder.s3.amazonaws.com/files/2026-01-16/dd41eae3-2592-4bc9-a8d2-d6d02614cd2d/ascreenshot_43ec9ee48ad946cca49732f007e786fc_text_export.jpeg)
![](https://colony-recorder.s3.amazonaws.com/files/2026-01-16/0c30309e-7117-4999-a3df-d22a2d5629c1/ascreenshot_d76a48c53b9a4fad8f6727baf4aa6a9c_text_export.jpeg)
### 3. View Usage in LiteLLM UI
Navigate to the **Logs** tab in the LiteLLM UI.
![](https://colony-recorder.s3.amazonaws.com/files/2026-01-16/ff774392-69f5-483e-83e2-fb749c94ee90/ascreenshot_d264fc04c9ee47edb047f61b6eb8c4d7_text_export.jpeg)
Click on a request to see details.
![](https://colony-recorder.s3.amazonaws.com/files/2026-01-16/5f71589b-5fdd-4759-9b6e-e6874be0eb21/ascreenshot_92dd86dadccb4764b1169c29c10dfe65_text_export.jpeg)
Filter by customer ID to see all requests for that customer.
![](https://colony-recorder.s3.amazonaws.com/files/2026-01-16/dd1c8aba-e75b-4714-9eee-c785e9db99af/ascreenshot_36aaec0fe12f4189b64f704a551e6729_text_export.jpeg)
## Supported Headers
| Header | Description |
|--------|-------------|
| `x-litellm-customer-id` | Track by customer/end-user ID |
| `x-litellm-end-user-id` | Alternative customer ID header |
| `x-litellm-tags` | Comma-separated tags for cost attribution |
## Related
- [Claude Code Quickstart](./claude_responses_api.md)
- [Customer Budgets](../proxy/customers.md)
- [Tag Budgets](../proxy/tag_budgets.md)
- [Track Usage for Coding Tools](./cost_tracking_coding.md)

View file

@ -0,0 +1,93 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Use Claude Code with MCPs
This tutorial shows how to connect MCP servers to Claude Code via LiteLLM Proxy.
Note: LiteLLM supports OAuth for MCP servers as well. [Learn more](https://docs.litellm.ai/docs/mcp#mcp-oauth)
## Connecting MCP Servers
You can also connect MCP servers to Claude Code via LiteLLM Proxy.
1. Add the MCP server to your `config.yaml`
<Tabs>
<TabItem value="github" label="GitHub MCP">
In this example, we'll add the Github MCP server to our `config.yaml`
```yaml title="config.yaml" showLineNumbers
mcp_servers:
github_mcp:
url: "https://api.githubcopilot.com/mcp"
auth_type: oauth2
client_id: os.environ/GITHUB_OAUTH_CLIENT_ID
client_secret: os.environ/GITHUB_OAUTH_CLIENT_SECRET
```
</TabItem>
<TabItem value="atlassian" label="Atlassian MCP">
In this example, we'll add the Atlassian MCP server to our `config.yaml`
```yaml title="config.yaml" showLineNumbers
atlassian_mcp:
server_id: atlassian_mcp_id
url: "https://mcp.atlassian.com/v1/sse"
transport: "sse"
auth_type: oauth2
```
</TabItem>
</Tabs>
2. Start LiteLLM Proxy
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
3. Use the MCP server in Claude Code
```bash
claude mcp add --transport http litellm_proxy http://0.0.0.0:4000/github_mcp/mcp --header "Authorization: Bearer sk-LITELLM_VIRTUAL_KEY"
```
For MCP servers that require dynamic client registration (such as Atlassian), please set `x-litellm-api-key: Bearer sk-LITELLM_VIRTUAL_KEY` instead of using `Authorization: Bearer LITELLM_VIRTUAL_KEY`.
4. Authenticate via Claude Code
a. Start Claude Code
```bash
claude
```
b. Authenticate via Claude Code
```bash
/mcp
```
c. Select the MCP server
```bash
> litellm_proxy
```
d. Start Oauth flow via Claude Code
```bash
> 1. Authenticate
2. Reconnect
3. Disable
```
e. Once completed, you should see this success message:
<img src={require('../../img/oauth_2_success.png').default} alt="OAuth 2.0 Success" style={{ width: '500px', height: 'auto' }} />

View file

@ -0,0 +1,316 @@
import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Use Claude Code with Non-Anthropic Models
This tutorial shows how to use Claude Code with non-Anthropic models like OpenAI, Gemini, and other LLM providers through LiteLLM proxy.
:::info
LiteLLM automatically translates between different provider formats, allowing you to use any supported LLM provider with Claude Code while maintaining the Anthropic Messages API format.
:::
## Prerequisites
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code/overview) installed
- API keys for your chosen providers (OpenAI, Vertex AI, etc.)
## Installation
First, install LiteLLM with proxy support:
```bash
pip install 'litellm[proxy]'
```
## Configuration
### 1. Setup config.yaml
Create a configuration file with your preferred non-Anthropic models:
<Tabs>
<TabItem value="openai" label="OpenAI">
```yaml
model_list:
# OpenAI GPT-4o
- model_name: gpt-4o
litellm_params:
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
# OpenAI GPT-4o-mini
- model_name: gpt-4o-mini
litellm_params:
model: openai/gpt-4o-mini
api_key: os.environ/OPENAI_API_KEY
```
Set your environment variables:
```bash
export OPENAI_API_KEY="your-openai-api-key"
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
```
</TabItem>
<TabItem value="gemini" label="Google AI Studio">
```yaml
model_list:
# Google Gemini
- model_name: gemini-3.0-flash-exp
litellm_params:
model: gemini/gemini-3.0-flash-exp
api_key: os.environ/GEMINI_API_KEY
```
Set your environment variables:
```bash
export GEMINI_API_KEY="your-gemini-api-key"
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
```
</TabItem>
<TabItem value="vertex_ai" label="Vertex AI">
```yaml
model_list:
# Google Gemini
- model_name: vertex-gemini-3-flash-preview
litellm_params:
model: vertex_ai/gemini-3-flash-preview
vertex_credentials: os.environ/VERTEX_FILE_PATH_ENV_VAR # os.environ["VERTEX_FILE_PATH_ENV_VAR"] = "/path/to/service_account.json"
vertex_project: "my-test-project"
vertex_location: "us-east-1"
# Anthropic Claude
- model_name: anthropic-vertex
litellm_params:
model: vertex_ai/claude-3-sonnet@20240229
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
vertex_credentials: os.environ/VERTEX_FILE_PATH_ENV_VAR # os.environ["VERTEX_FILE_PATH_ENV_VAR"] = "/path/to/service_account.json"
```
Set your environment variables:
```bash
export VERTEX_FILE_PATH_ENV_VAR="/path/to/service_account.json"
export LITELLM_MASTER_KEY="sk-1234567890"
```
</TabItem>
<TabItem value="multi" label="Azure OpenAI">
```yaml
model_list:
# Azure OpenAI
- model_name: azure-gpt-4
litellm_params:
model: azure/gpt-4
api_key: os.environ/AZURE_API_KEY
api_base: os.environ/AZURE_API_BASE
api_version: "2024-02-01"
```
Set your environment variables:
```bash
export AZURE_API_KEY="your-azure-api-key"
export AZURE_API_BASE="https://your-resource.openai.azure.com"
export LITELLM_MASTER_KEY="sk-1234567890"
```
</TabItem>
</Tabs>
### 2. Start LiteLLM Proxy
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
### 3. Verify Setup
Test that your proxy is working correctly:
<Tabs>
<TabItem value="openai-test" label="OpenAI">
```bash
curl -X POST http://0.0.0.0:4000/v1/messages \
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "gpt-4o",
"max_tokens": 1000,
"messages": [{"role": "user", "content": "What is the capital of France?"}]
}'
```
</TabItem>
<TabItem value="gemini-test" label="Google AI Studio">
```bash
curl -X POST http://0.0.0.0:4000/v1/messages \
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "gemini-3.0-flash-exp",
"max_tokens": 1000,
"messages": [{"role": "user", "content": "What is the capital of France?"}]
}'
```
</TabItem>
<TabItem value="vertex-test" label="Vertex AI">
```bash
curl -X POST http://0.0.0.0:4000/v1/messages \
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "gemini-3.0-flash-exp",
"max_tokens": 1000,
"messages": [{"role": "user", "content": "What is the capital of France?"}]
}'
```
</TabItem>
<TabItem value="azure-test" label="Azure OpenAI">
```bash
curl -X POST http://0.0.0.0:4000/v1/messages \
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "azure-gpt-4",
"max_tokens": 1000,
"messages": [{"role": "user", "content": "What is the capital of France?"}]
}'
```
</TabItem>
</Tabs>
### 4. Configure Claude Code
Configure Claude Code to use your LiteLLM proxy:
```bash
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000"
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
```
:::tip
The `LITELLM_MASTER_KEY` gives Claude Code access to all proxy models. You can also create virtual keys in the LiteLLM UI to limit access to specific models.
:::
### 5. Use Claude Code with Non-Anthropic Models
Start Claude Code and specify which model to use:
```bash
# Use OpenAI GPT-4o
claude --model gpt-4o
# Use OpenAI GPT-4o-mini for faster responses
claude --model gpt-4o-mini
# Use Google Gemini
claude --model gemini-3.0-flash-exp
# Use Vertex AI Gemini
claude --model vertex-gemini-3-flash-preview
# Use Vertex AI Anthropic Claude
claude --model anthropic-vertex
# Use Azure OpenAI
claude --model azure-gpt-4
```
## How It Works
LiteLLM acts as a unified interface that:
1. **Receives requests** from Claude Code in Anthropic Messages API format
2. **Translates** the request to the target provider's format (OpenAI, Gemini, etc.)
3. **Forwards** the request to the actual provider
4. **Translates** the response back to Anthropic Messages API format
5. **Returns** the response to Claude Code
This allows you to use Claude Code's interface with any LLM provider supported by LiteLLM.
## Advanced Features
### Load Balancing and Fallbacks
Configure multiple deployments with automatic fallback:
```yaml
model_list:
- model_name: gpt-4o # virtual model name
litellm_params:
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-4o # same virtual name
litellm_params:
model: azure/gpt-4o
api_key: os.environ/AZURE_API_KEY
api_base: os.environ/AZURE_API_BASE
router_settings:
routing_strategy: simple-shuffle # Load balance between deployments
num_retries: 2
timeout: 30
```
### Usage Tracking and Budgets
Track usage and set budgets through the LiteLLM UI:
```yaml
litellm_settings:
master_key: os.environ/LITELLM_MASTER_KEY
database_url: "postgresql://..." # Enable database for tracking
general_settings:
store_model_in_db: true
```
Start the proxy with the UI:
```bash
litellm --config /path/to/config.yaml --detailed_debug
```
Access the UI at `http://0.0.0.0:4000/ui` to:
- View usage analytics
- Set budget limits per user/key
- Monitor costs across different providers
- Create virtual keys with specific permissions
## Supported Providers
LiteLLM supports 100+ providers. Here are some popular ones for use with Claude Code:
- **OpenAI**: GPT-4o, GPT-4o-mini, o1, o3-mini
- **Google**: Gemini 2.0 Flash, Gemini 1.5 Pro/Flash
- **Azure OpenAI**: All OpenAI models via Azure
- **AWS Bedrock**: Llama, Mistral, and other models
- **Vertex AI**: Gemini, Claude, and other models on Google Cloud
- **Groq**: Fast inference for Llama and Mixtral
- **Together AI**: Llama, Mixtral, and other open source models
- **Deepseek**: Deepseek-chat, Deepseek-coder
[View full list of supported providers →](https://docs.litellm.ai/docs/providers)

View file

@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Claude Code
# Claude Code Quickstart
This tutorial shows how to call Claude models through LiteLLM proxy from Claude Code.
@ -142,7 +142,7 @@ Common issues and solutions:
- Ensure the model name in Claude Code matches exactly with your `config.yaml`
- Check LiteLLM logs for detailed error messages
## Using Multiple Models
## Using Bedrock/Vertex AI/Azure Foundry Models
Expand your configuration to support multiple providers and models:
@ -151,25 +151,6 @@ Expand your configuration to support multiple providers and models:
```yaml
model_list:
# OpenAI models
- model_name: codex-mini
litellm_params:
model: openai/codex-mini
api_key: os.environ/OPENAI_API_KEY
api_base: https://api.openai.com/v1
- model_name: o3-pro
litellm_params:
model: openai/o3-pro
api_key: os.environ/OPENAI_API_KEY
api_base: https://api.openai.com/v1
- model_name: gpt-4o
litellm_params:
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
api_base: https://api.openai.com/v1
# Anthropic models
- model_name: claude-3-5-sonnet-20241022
litellm_params:
@ -189,6 +170,24 @@ model_list:
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: us-east-1
# Azure Foundry
- model_name: claude-4-azure
litellm_params:
model: azure_ai/claude-opus-4-1
api_key: os.environ/AZURE_AI_API_KEY
api_base: os.environ/AZURE_AI_API_BASE # https://my-resource.services.ai.azure.com/anthropic
# Google Vertex AI
- model_name: anthropic-vertex
litellm_params:
model: vertex_ai/claude-haiku-4-5@20251001
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
vertex_credentials: os.environ/VERTEX_FILE_PATH_ENV_VAR # os.environ["VERTEX_FILE_PATH_ENV_VAR"] = "/path/to/service_account.json"
litellm_settings:
master_key: os.environ/LITELLM_MASTER_KEY
```
@ -204,6 +203,12 @@ claude --model claude-3-5-haiku-20241022
# Use Bedrock deployment
claude --model claude-bedrock
# Use Azure Foundry deployment
claude --model claude-4-azure
# Use Vertex AI deployment
claude --model anthropic-vertex
```
</TabItem>
@ -211,96 +216,3 @@ claude --model claude-bedrock
<Image img={require('../../img/release_notes/claude_code_demo.png')} style={{ width: '500px', height: 'auto' }} />
## Connecting MCP Servers
You can also connect MCP servers to Claude Code via LiteLLM Proxy.
:::note
Limitations:
- Currently, only HTTP MCP servers are supported
:::
1. Add the MCP server to your `config.yaml`
<Tabs>
<TabItem value="github" label="GitHub MCP">
In this example, we'll add the Github MCP server to our `config.yaml`
```yaml title="config.yaml" showLineNumbers
mcp_servers:
github_mcp:
url: "https://api.githubcopilot.com/mcp"
auth_type: oauth2
client_id: os.environ/GITHUB_OAUTH_CLIENT_ID
client_secret: os.environ/GITHUB_OAUTH_CLIENT_SECRET
```
</TabItem>
<TabItem value="atlassian" label="Atlassian MCP">
In this example, we'll add the Atlassian MCP server to our `config.yaml`
```yaml title="config.yaml" showLineNumbers
atlassian_mcp:
server_id: atlassian_mcp_id
url: "https://mcp.atlassian.com/v1/sse"
transport: "sse"
auth_type: oauth2
```
</TabItem>
</Tabs>
2. Start LiteLLM Proxy
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
3. Use the MCP server in Claude Code
```bash
claude mcp add --transport http litellm_proxy http://0.0.0.0:4000/github_mcp/mcp --header "Authorization: Bearer sk-LITELLM_VIRTUAL_KEY"
```
For MCP servers that require dynamic client registration (such as Atlassian), please set `x-litellm-api-key: Bearer sk-LITELLM_VIRTUAL_KEY` instead of using `Authorization: Bearer LITELLM_VIRTUAL_KEY`.
4. Authenticate via Claude Code
a. Start Claude Code
```bash
claude
```
b. Authenticate via Claude Code
```bash
/mcp
```
c. Select the MCP server
```bash
> litellm_proxy
```
d. Start Oauth flow via Claude Code
```bash
> 1. Authenticate
2. Reconnect
3. Disable
```
e. Once completed, you should see this success message:
<Image img={require('../../img/oauth_2_success.png')} style={{ width: '500px', height: 'auto' }} />

View file

@ -108,15 +108,30 @@ const sidebars = {
{
type: "category",
label: "AI Tools (OpenWebUI, Claude Code, etc.)",
link: {
type: "generated-index",
title: "AI Tools",
description: "Integrate LiteLLM with AI tools like OpenWebUI, Claude Code, and more",
slug: "/ai_tools"
},
items: [
"tutorials/claude_responses_api",
"tutorials/openweb_ui",
{
type: "category",
label: "Claude Code",
items: [
"tutorials/claude_responses_api",
"tutorials/claude_code_customer_tracking",
"tutorials/claude_mcp",
"tutorials/claude_non_anthropic_models",
]
},
"tutorials/cost_tracking_coding",
"tutorials/cursor_integration",
"tutorials/github_copilot_integration",
"tutorials/litellm_gemini_cli",
"tutorials/litellm_qwen_code_cli",
"tutorials/openai_codex",
"tutorials/openweb_ui"
"tutorials/openai_codex"
]
},
@ -862,10 +877,11 @@ const sidebars = {
type: "category",
label: "Tutorials",
items: [
"tutorials/openweb_ui",
"tutorials/openai_codex",
"tutorials/litellm_gemini_cli",
"tutorials/litellm_qwen_code_cli",
{
type: "link",
label: "AI Coding Tools (OpenWebUI, Claude Code, Gemini CLI, OpenAI Codex, etc.)",
href: "/docs/ai_tools",
},
"tutorials/anthropic_file_usage",
"tutorials/default_team_self_serve",
"tutorials/msft_sso",
@ -875,7 +891,6 @@ const sidebars = {
"tutorials/presidio_pii_masking",
"tutorials/elasticsearch_logging",
"tutorials/gemini_realtime_with_audio",
"tutorials/claude_responses_api",
{
type: "category",
label: "LiteLLM Python SDK Tutorials",

View file

@ -1,19 +0,0 @@
LiteLLM provides a unified interface for calling 100+ different LLM providers.
Key capabilities:
- Translate requests to provider-specific formats
- Consistent OpenAI-compatible responses
- Retry and fallback logic across deployments
- Proxy server with authentication and rate limiting
- Support for streaming, function calling, and embeddings
Popular providers supported:
- OpenAI (GPT-4, GPT-3.5)
- Anthropic (Claude)
- AWS Bedrock
- Azure OpenAI
- Google Vertex AI
- Cohere
- And 95+ more
This allows developers to easily switch between providers without code changes.

View file

@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm-enterprise"
version = "0.1.27"
version = "0.1.28"
description = "Package for LiteLLM Enterprise features"
authors = ["BerriAI"]
readme = "README.md"
@ -22,7 +22,7 @@ requires = ["poetry-core"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "0.1.27"
version = "0.1.28"
version_files = [
"pyproject.toml:version",
"../requirements.txt:litellm-enterprise==",

Binary file not shown.

Before

Width:  |  Height:  |  Size: 172 KiB

View file

@ -1073,6 +1073,13 @@ LITELLM_TRUNCATED_PAYLOAD_FIELD = "litellm_truncated"
########################### LiteLLM Proxy Specific Constants ###########################
########################################################################################
# Standard headers that are always checked for customer/end-user ID (no configuration required)
# These headers work out-of-the-box for tools like Claude Code that support custom headers
STANDARD_CUSTOMER_ID_HEADERS = [
"x-litellm-customer-id",
"x-litellm-end-user-id",
]
MAX_SPENDLOG_ROWS_TO_QUERY = int(
os.getenv("MAX_SPENDLOG_ROWS_TO_QUERY", 1_000_000)
) # if spendLogs has more than 1M rows, do not query the DB

View file

@ -952,7 +952,8 @@ def completion_cost( # noqa: PLR0915
)
potential_model_names = [selected_model, _get_response_model(completion_response)]
if model is not None:
potential_model_names.append(model)
for idx, model in enumerate(potential_model_names):
try:

View file

@ -404,6 +404,7 @@ def image_generation( # noqa: PLR0915
litellm.LlmProviders.STABILITY,
litellm.LlmProviders.RUNWAYML,
litellm.LlmProviders.VERTEX_AI,
litellm.LlmProviders.OPENROUTER
):
if image_generation_config is None:
raise ValueError(

View file

@ -21,7 +21,7 @@ from typing import (
import litellm
from litellm._logging import print_verbose, verbose_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import LiteLLM_TeamTable, UserAPIKeyAuth
from litellm.proxy._types import LiteLLM_TeamTable, LiteLLM_UserTable, UserAPIKeyAuth
from litellm.types.integrations.prometheus import *
from litellm.types.integrations.prometheus import _sanitize_prometheus_label_name
from litellm.types.utils import StandardLoggingPayload
@ -52,7 +52,7 @@ def _get_cached_end_user_id_for_cost_tracking():
class PrometheusLogger(CustomLogger):
# Class variables or attributes
def __init__(
def __init__( # noqa: PLR0915
self,
**kwargs,
):
@ -193,6 +193,30 @@ class PrometheusLogger(CustomLogger):
),
)
# Remaining Budget for User
self.litellm_remaining_user_budget_metric = self._gauge_factory(
"litellm_remaining_user_budget_metric",
"Remaining budget for user",
labelnames=self.get_labels_for_metric(
"litellm_remaining_user_budget_metric"
),
)
# Max Budget for User
self.litellm_user_max_budget_metric = self._gauge_factory(
"litellm_user_max_budget_metric",
"Maximum budget set for user",
labelnames=self.get_labels_for_metric("litellm_user_max_budget_metric"),
)
self.litellm_user_budget_remaining_hours_metric = self._gauge_factory(
"litellm_user_budget_remaining_hours_metric",
"Remaining hours for user budget to be reset",
labelnames=self.get_labels_for_metric(
"litellm_user_budget_remaining_hours_metric"
),
)
########################################
# LiteLLM Virtual API KEY metrics
########################################
@ -960,6 +984,7 @@ class PrometheusLogger(CustomLogger):
user_api_key_alias=user_api_key_alias,
litellm_params=litellm_params,
response_cost=response_cost,
user_id=user_id,
)
# set proxy virtual key rpm/tpm metrics
@ -1120,6 +1145,7 @@ class PrometheusLogger(CustomLogger):
user_api_key_alias: Optional[str],
litellm_params: dict,
response_cost: float,
user_id: Optional[str] = None,
):
_team_spend = litellm_params.get("metadata", {}).get(
"user_api_key_team_spend", None
@ -1134,6 +1160,14 @@ class PrometheusLogger(CustomLogger):
_api_key_max_budget = litellm_params.get("metadata", {}).get(
"user_api_key_max_budget", None
)
_user_spend = litellm_params.get("metadata", {}).get(
"user_api_key_user_spend", None
)
_user_max_budget = litellm_params.get("metadata", {}).get(
"user_api_key_user_max_budget", None
)
await self._set_api_key_budget_metrics_after_api_request(
user_api_key=user_api_key,
user_api_key_alias=user_api_key_alias,
@ -1150,6 +1184,13 @@ class PrometheusLogger(CustomLogger):
response_cost=response_cost,
)
await self._set_user_budget_metrics_after_api_request(
user_id=user_id,
user_spend=_user_spend,
user_max_budget=_user_max_budget,
response_cost=response_cost,
)
def _increment_top_level_request_and_spend_metrics(
self,
end_user_id: Optional[str],
@ -2229,6 +2270,37 @@ class PrometheusLogger(CustomLogger):
data_type="keys",
)
async def _initialize_user_budget_metrics(self):
"""
Initialize user budget metrics by reusing the generic pagination logic.
"""
from litellm.proxy._types import LiteLLM_UserTable
from litellm.proxy.proxy_server import prisma_client
if prisma_client is None:
verbose_logger.debug(
"Prometheus: skipping user metrics initialization, DB not initialized"
)
return
async def fetch_users(
page_size: int, page: int
) -> Tuple[List[LiteLLM_UserTable], Optional[int]]:
skip = (page - 1) * page_size
users = await prisma_client.db.litellm_usertable.find_many(
skip=skip,
take=page_size,
order={"created_at": "desc"},
)
total_count = await prisma_client.db.litellm_usertable.count()
return users, total_count
await self._initialize_budget_metrics(
data_fetch_function=fetch_users,
set_metrics_function=self._set_user_list_budget_metrics,
data_type="users",
)
async def initialize_remaining_budget_metrics(self):
"""
Handler for initializing remaining budget metrics for all teams to avoid metric discrepancies.
@ -2261,11 +2333,12 @@ class PrometheusLogger(CustomLogger):
async def _initialize_remaining_budget_metrics(self):
"""
Helper to initialize remaining budget metrics for all teams and API keys.
Helper to initialize remaining budget metrics for all teams, API keys, and users.
"""
verbose_logger.debug("Emitting key, team budget metrics....")
verbose_logger.debug("Emitting key, team, user budget metrics....")
await self._initialize_team_budget_metrics()
await self._initialize_api_key_budget_metrics()
await self._initialize_user_budget_metrics()
async def _set_key_list_budget_metrics(
self, keys: List[Union[str, UserAPIKeyAuth]]
@ -2280,6 +2353,11 @@ class PrometheusLogger(CustomLogger):
for team in teams:
self._set_team_budget_metrics(team)
async def _set_user_list_budget_metrics(self, users: List[LiteLLM_UserTable]):
"""Helper function to set budget metrics for a list of users"""
for user in users:
self._set_user_budget_metrics(user)
async def _set_team_budget_metrics_after_api_request(
self,
user_api_team: Optional[str],
@ -2497,6 +2575,122 @@ class PrometheusLogger(CustomLogger):
return user_api_key_dict
async def _set_user_budget_metrics_after_api_request(
self,
user_id: Optional[str],
user_spend: Optional[float],
user_max_budget: Optional[float],
response_cost: float,
):
"""
Set user budget metrics after an LLM API request
- Assemble a LiteLLM_UserTable object
- looks up user info from db if not available in metadata
- Set user budget metrics
"""
if user_id:
user_object = await self._assemble_user_object(
user_id=user_id,
spend=user_spend,
max_budget=user_max_budget,
response_cost=response_cost,
)
self._set_user_budget_metrics(user_object)
async def _assemble_user_object(
self,
user_id: str,
spend: Optional[float],
max_budget: Optional[float],
response_cost: float,
) -> LiteLLM_UserTable:
"""
Assemble a LiteLLM_UserTable object
for fields not available in metadata, we fetch from db
Fields not available in metadata:
- `budget_reset_at`
"""
from litellm.proxy.auth.auth_checks import get_user_object
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
_total_user_spend = (spend or 0) + response_cost
user_object = LiteLLM_UserTable(
user_id=user_id,
spend=_total_user_spend,
max_budget=max_budget,
)
try:
user_info = await get_user_object(
user_id=user_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
user_id_upsert=False,
check_db_only=True,
)
except Exception as e:
verbose_logger.debug(
f"[Non-Blocking] Prometheus: Error getting user info: {str(e)}"
)
return user_object
if user_info:
user_object.budget_reset_at = user_info.budget_reset_at
return user_object
def _set_user_budget_metrics(
self,
user: LiteLLM_UserTable,
):
"""
Set user budget metrics for a single user
- Remaining Budget
- Max Budget
- Budget Reset At
"""
enum_values = UserAPIKeyLabelValues(
user=user.user_id,
)
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
metric_name="litellm_remaining_user_budget_metric"
),
enum_values=enum_values,
)
self.litellm_remaining_user_budget_metric.labels(**_labels).set(
self._safe_get_remaining_budget(
max_budget=user.max_budget,
spend=user.spend,
)
)
if user.max_budget is not None:
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
metric_name="litellm_user_max_budget_metric"
),
enum_values=enum_values,
)
self.litellm_user_max_budget_metric.labels(**_labels).set(user.max_budget)
if user.budget_reset_at is not None:
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
metric_name="litellm_user_budget_remaining_hours_metric"
),
enum_values=enum_values,
)
self.litellm_user_budget_remaining_hours_metric.labels(**_labels).set(
self._get_remaining_hours_for_budget_reset(
budget_reset_at=user.budget_reset_at
)
)
def _get_remaining_hours_for_budget_reset(self, budget_reset_at: datetime) -> float:
"""
Get remaining hours for budget reset

View file

@ -3743,10 +3743,10 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
OpenTelemetry,
OpenTelemetryConfig,
)
logfire_base_url = os.getenv("LOGFIRE_BASE_URL", "https://logfire-api.pydantic.dev")
otel_config = OpenTelemetryConfig(
exporter="otlp_http",
endpoint="https://logfire-api.pydantic.dev/v1/traces",
endpoint = f"{logfire_base_url.rstrip('/')}/v1/traces",
headers=f"Authorization={os.getenv('LOGFIRE_TOKEN')}",
)
for callback in _in_memory_loggers:
@ -4488,7 +4488,7 @@ class StandardLoggingPayloadSetup:
@staticmethod
def get_usage_from_response_obj(
response_obj: Optional[Union[dict, BaseModel]], combined_usage_object: Optional[Usage] = None
response_obj: Optional[dict], combined_usage_object: Optional[Usage] = None
) -> Usage:
## BASE CASE ##
if combined_usage_object is not None:
@ -4500,32 +4500,27 @@ class StandardLoggingPayloadSetup:
total_tokens=0,
)
usage = _safe_extract_usage_from_obj(response_obj)
if usage is None:
usage = response_obj.get("usage", None) or {}
if usage is None or (
not isinstance(usage, dict) and not isinstance(usage, Usage)
):
return Usage(
prompt_tokens=0,
completion_tokens=0,
total_tokens=0,
)
if isinstance(usage, Usage):
elif isinstance(usage, Usage):
return usage
transformed_usage = _try_transform_response_api_usage(usage)
if transformed_usage is not None:
return transformed_usage
if isinstance(usage, dict):
created_usage = _try_create_usage_from_dict(usage)
if created_usage is not None:
return created_usage
return Usage(
prompt_tokens=0,
completion_tokens=0,
total_tokens=0,
)
elif isinstance(usage, dict):
if ResponseAPILoggingUtils._is_response_api_usage(usage):
return (
ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
usage
)
)
return Usage(**usage)
raise ValueError(f"usage is required, got={usage} of type {type(usage)}")
@staticmethod
def get_model_cost_information(
@ -4566,18 +4561,13 @@ class StandardLoggingPayloadSetup:
@staticmethod
def get_final_response_obj(
response_obj: Union[dict, BaseModel], init_response_obj: Union[Any, BaseModel, dict], kwargs: dict
response_obj: dict, init_response_obj: Union[Any, BaseModel, dict], kwargs: dict
) -> Optional[Union[dict, str, list]]:
"""
Get final response object after redacting the message input/output from logging
"""
if response_obj:
if isinstance(response_obj, BaseModel):
final_response_obj: Optional[Union[dict, str, list]] = _safe_model_dump(
response_obj, default={}
)
else:
final_response_obj = response_obj
final_response_obj: Optional[Union[dict, str, list]] = response_obj
elif isinstance(init_response_obj, list) or isinstance(init_response_obj, str):
final_response_obj = init_response_obj
else:
@ -4591,7 +4581,7 @@ class StandardLoggingPayloadSetup:
if modified_final_response_obj is not None and isinstance(
modified_final_response_obj, BaseModel
):
final_response_obj = _safe_model_dump(modified_final_response_obj, default={})
final_response_obj = modified_final_response_obj.model_dump()
else:
final_response_obj = modified_final_response_obj
@ -4862,125 +4852,6 @@ class StandardLoggingPayloadSetup:
return request_tags
def _safe_model_dump(
obj: BaseModel, default: Optional[Union[dict, str, list]] = None
) -> Union[dict, str, list]:
"""
Safely call model_dump() on a BaseModel with fallback strategies.
Args:
obj: BaseModel instance to dump
default: Default value to return if all strategies fail
Returns:
Dict representation of the BaseModel, or fallback value
"""
if default is None:
default = {}
try:
return obj.model_dump()
except (AttributeError, TypeError) as e:
verbose_logger.debug(
f"Error calling model_dump() on BaseModel: {e}, type: {type(obj)}"
)
try:
if hasattr(obj, "__dict__"):
return obj.__dict__
else:
return str(obj)
except Exception:
return default
def _safe_get_attribute(
obj: Union[dict, BaseModel, Any], attr_name: str, default: Any = None
) -> Any:
"""
Safely get an attribute from a dict or BaseModel object.
Args:
obj: Object to get attribute from (dict, BaseModel, or any object)
attr_name: Name of the attribute to get
default: Default value to return if attribute doesn't exist
Returns:
Attribute value or default
"""
try:
if isinstance(obj, dict):
return obj.get(attr_name, default)
else:
return getattr(obj, attr_name, default)
except (AttributeError, TypeError) as e:
verbose_logger.debug(
f"Error getting attribute '{attr_name}' from object: {e}, type: {type(obj)}"
)
return default
def _safe_extract_usage_from_obj(
response_obj: Union[dict, BaseModel, Any]
) -> Optional[Union[dict, Usage, Any]]:
"""
Safely extract usage from response_obj (dict or BaseModel).
Args:
response_obj: Response object (dict, BaseModel, or any object)
Returns:
Usage object, dict, or None
"""
return _safe_get_attribute(response_obj, "usage", None)
def _try_transform_response_api_usage(usage: Any) -> Optional[Usage]:
"""
Try to transform ResponseAPIUsage to Usage object.
Args:
usage: Usage object (dict, ResponseAPIUsage, or other)
Returns:
Transformed Usage object, or None if transformation fails
"""
try:
if ResponseAPILoggingUtils._is_response_api_usage(usage):
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage)
except (AttributeError, TypeError, KeyError) as e:
verbose_logger.debug(
f"Error checking/transforming ResponseAPIUsage: {e}, type: {type(usage)}"
)
return None
def _try_create_usage_from_dict(usage: dict) -> Optional[Usage]:
"""
Try to create Usage object from dict.
Args:
usage: Dict containing usage information
Returns:
Usage object, or None if creation fails
"""
try:
return Usage(**usage)
except (TypeError, ValueError) as e:
# Avoid logging full dict contents, which may include sensitive data
try:
usage_keys = list(usage.keys())
except Exception:
usage_keys = None
verbose_logger.debug(
"Error creating Usage from dict: %s, usage keys: %s, usage type: %s",
e,
usage_keys,
type(usage),
)
return None
def _get_status_fields(
status: StandardLoggingPayloadStatus,
guardrail_information: Optional[List[dict]],
@ -5030,21 +4901,17 @@ def _get_status_fields(
def _extract_response_obj_and_hidden_params(
init_response_obj: Union[Any, BaseModel, dict],
original_exception: Optional[Exception],
) -> Tuple[Union[dict, BaseModel], Optional[dict]]:
) -> Tuple[dict, Optional[dict]]:
"""Extract response_obj and hidden_params from init_response_obj."""
hidden_params: Optional[dict] = None
if init_response_obj is None:
response_obj: Union[dict, BaseModel] = {}
response_obj = {}
elif isinstance(init_response_obj, BaseModel):
response_obj = init_response_obj
hidden_params = _safe_get_attribute(init_response_obj, "_hidden_params", None)
response_obj = init_response_obj.model_dump()
hidden_params = getattr(init_response_obj, "_hidden_params", None)
elif isinstance(init_response_obj, dict):
response_obj = init_response_obj
else:
verbose_logger.debug(
f"Unknown init_response_obj type: {type(init_response_obj)}, defaulting to empty dict"
)
response_obj = {}
if original_exception is not None and hidden_params is None:
@ -5104,10 +4971,7 @@ def get_standard_logging_object_payload(
),
)
# Preserve falsy values (0, "", False) if they exist in response_obj
id = _safe_get_attribute(response_obj, "id", None)
if id is None:
id = kwargs.get("litellm_call_id")
id = response_obj.get("id", kwargs.get("litellm_call_id"))
_model_id = metadata.get("model_info", {}).get("id", "")
_model_group = metadata.get("model_group", "")

View file

@ -45,7 +45,6 @@ from .common_utils import (
infer_content_type_from_url_and_content,
is_non_content_values_set,
parse_tool_call_arguments,
unpack_defs,
)
from .image_handling import convert_url_to_base64
@ -1463,56 +1462,6 @@ def convert_to_gemini_tool_call_invoke(
)
def _clean_refs_for_gemini(obj: Any) -> None:
"""
Recursively clean $defs, $ref, and definitions from a dict for Gemini compatibility.
Gemini rejects:
- $defs sections (even after $ref has been inlined)
- Any remaining $ref (circular refs, external URLs)
This function:
1. Removes all $defs/definitions keys
2. Replaces any remaining $ref with a placeholder object
"""
if isinstance(obj, dict):
# Remove $defs and definitions at this level
obj.pop("$defs", None)
obj.pop("definitions", None)
# Check for and handle remaining $ref (circular or external)
if "$ref" in obj:
ref_value = obj.pop("$ref")
# Replace with a generic object type as placeholder
obj["type"] = "object"
obj["description"] = f"(schema reference: {ref_value})"
# Recurse into values
for value in obj.values():
_clean_refs_for_gemini(value)
elif isinstance(obj, list):
for item in obj:
_clean_refs_for_gemini(item)
def _prepare_response_for_gemini(response_data: dict) -> dict:
"""
Prepare a tool response dict for Gemini by inlining $ref and removing $defs.
Gemini rejects JSON schemas with $defs/$ref in function_response content.
This function applies unpack_defs to inline references, then cleans up
any remaining $defs sections and unresolved $refs (circular or external).
Returns a new dict (does not mutate the input).
"""
import copy
result = copy.deepcopy(response_data)
unpack_defs(result, {})
_clean_refs_for_gemini(result)
return result
def convert_to_gemini_tool_call_result(
message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage],
last_message_with_tool_calls: Optional[dict],
@ -1621,11 +1570,6 @@ def convert_to_gemini_tool_call_result(
# Not valid JSON, wrap in content field
response_data = {"content": content_str}
# Gemini rejects JSON schemas with $defs/$ref in function_response content.
# Inline $refs and clean up for Gemini compatibility.
if isinstance(response_data, dict):
response_data = _prepare_response_for_gemini(response_data)
# We can't determine from openai message format whether it's a successful or
# error call result so default to the successful result template
_function_response = VertexFunctionResponse(

View file

@ -132,7 +132,7 @@ class ChunkProcessor:
)
return response
def get_combined_tool_content(
def get_combined_tool_content( # noqa: PLR0915
self, tool_call_chunks: List[Dict[str, Any]]
) -> List[ChatCompletionMessageToolCall]:
tool_calls_list: List[ChatCompletionMessageToolCall] = []
@ -147,10 +147,26 @@ class ChunkProcessor:
tool_calls = delta.get("tool_calls", [])
for tool_call in tool_calls:
if not tool_call or not hasattr(tool_call, "function"):
# Handle both dict and object formats
if not tool_call:
continue
# Check if tool_call has function (either as attribute or dict key)
has_function = False
if isinstance(tool_call, dict):
has_function = "function" in tool_call and tool_call["function"] is not None
else:
has_function = hasattr(tool_call, "function") and tool_call.function is not None
if not has_function:
continue
index = getattr(tool_call, "index", 0)
# Get index (handle both dict and object)
if isinstance(tool_call, dict):
index = tool_call.get("index", 0)
else:
index = getattr(tool_call, "index", 0)
if index not in tool_call_map:
tool_call_map[index] = {
"id": None,
@ -160,30 +176,56 @@ class ChunkProcessor:
"provider_specific_fields": None,
}
if hasattr(tool_call, "id") and tool_call.id:
tool_call_map[index]["id"] = tool_call.id
if hasattr(tool_call, "type") and tool_call.type:
tool_call_map[index]["type"] = tool_call.type
if hasattr(tool_call, "function"):
if (
hasattr(tool_call.function, "name")
and tool_call.function.name
):
tool_call_map[index]["name"] = tool_call.function.name
if (
hasattr(tool_call.function, "arguments")
and tool_call.function.arguments
):
tool_call_map[index]["arguments"].append(
tool_call.function.arguments
)
# Extract id, type, and function data (handle both dict and object)
if isinstance(tool_call, dict):
if tool_call.get("id"):
tool_call_map[index]["id"] = tool_call["id"]
if tool_call.get("type"):
tool_call_map[index]["type"] = tool_call["type"]
function = tool_call.get("function", {})
if isinstance(function, dict):
if function.get("name"):
tool_call_map[index]["name"] = function["name"]
if function.get("arguments"):
tool_call_map[index]["arguments"].append(function["arguments"])
else:
# function is an object
if hasattr(function, "name") and function.name:
tool_call_map[index]["name"] = function.name
if hasattr(function, "arguments") and function.arguments:
tool_call_map[index]["arguments"].append(function.arguments)
else:
# tool_call is an object
if hasattr(tool_call, "id") and tool_call.id:
tool_call_map[index]["id"] = tool_call.id
if hasattr(tool_call, "type") and tool_call.type:
tool_call_map[index]["type"] = tool_call.type
if hasattr(tool_call, "function"):
if (
hasattr(tool_call.function, "name")
and tool_call.function.name
):
tool_call_map[index]["name"] = tool_call.function.name
if (
hasattr(tool_call.function, "arguments")
and tool_call.function.arguments
):
tool_call_map[index]["arguments"].append(
tool_call.function.arguments
)
# Preserve provider_specific_fields from streaming chunks
provider_fields = None
if hasattr(tool_call, "provider_specific_fields") and tool_call.provider_specific_fields:
provider_fields = tool_call.provider_specific_fields
elif hasattr(tool_call, "function") and hasattr(tool_call.function, "provider_specific_fields") and tool_call.function.provider_specific_fields:
provider_fields = tool_call.function.provider_specific_fields
if isinstance(tool_call, dict):
provider_fields = tool_call.get("provider_specific_fields")
if not provider_fields and isinstance(tool_call.get("function"), dict):
provider_fields = tool_call["function"].get("provider_specific_fields")
else:
if hasattr(tool_call, "provider_specific_fields") and tool_call.provider_specific_fields:
provider_fields = tool_call.provider_specific_fields
elif hasattr(tool_call, "function") and hasattr(tool_call.function, "provider_specific_fields") and tool_call.function.provider_specific_fields:
provider_fields = tool_call.function.provider_specific_fields
if provider_fields:
# Merge provider_specific_fields if multiple chunks have them
@ -222,6 +264,7 @@ class ChunkProcessor:
return tool_calls_list
def get_combined_function_call_content(
self, function_call_chunks: List[Dict[str, Any]]
) -> FunctionCall:

View file

@ -2,7 +2,8 @@ from typing import Any, AsyncIterator, Dict, List, Optional, Tuple
import httpx
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj, verbose_logger
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.litellm_core_utils.litellm_logging import verbose_logger
from litellm.llms.base_llm.anthropic_messages.transformation import (
BaseAnthropicMessagesConfig,
)
@ -13,9 +14,10 @@ from litellm.types.llms.anthropic import (
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
from litellm.types.llms.anthropic_tool_search import get_tool_search_beta_header
from litellm.types.router import GenericLiteLLMParams
from ...common_utils import AnthropicError
from ...common_utils import AnthropicError, AnthropicModelInfo
DEFAULT_ANTHROPIC_API_BASE = "https://api.anthropic.com"
DEFAULT_ANTHROPIC_API_VERSION = "2023-06-01"
@ -75,9 +77,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if "content-type" not in headers:
headers["content-type"] = "application/json"
headers = self._update_headers_with_optional_anthropic_beta(
headers = self._update_headers_with_anthropic_beta(
headers=headers,
context_management=optional_params.get("context_management"),
optional_params=optional_params,
)
return headers, api_base
@ -153,16 +155,44 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
)
@staticmethod
def _update_headers_with_optional_anthropic_beta(
headers: dict, context_management: Optional[Dict]
def _update_headers_with_anthropic_beta(
headers: dict,
optional_params: dict,
custom_llm_provider: str = "anthropic",
) -> dict:
if context_management is None:
return headers
"""
Auto-inject anthropic-beta headers based on features used.
Handles:
- context_management: adds 'context-management-2025-06-27'
- tool_search: adds provider-specific tool search header
Args:
headers: Request headers dict
optional_params: Optional parameters including tools, context_management
custom_llm_provider: Provider name for looking up correct tool search header
"""
beta_values: set = set()
# Get existing beta headers if any
existing_beta = headers.get("anthropic-beta")
beta_value = ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value
if existing_beta is None:
headers["anthropic-beta"] = beta_value
elif beta_value not in [beta.strip() for beta in existing_beta.split(",")]:
headers["anthropic-beta"] = f"{existing_beta}, {beta_value}"
if existing_beta:
beta_values.update(b.strip() for b in existing_beta.split(","))
# Check for context management
if optional_params.get("context_management") is not None:
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value)
# Check for tool search tools
tools = optional_params.get("tools")
if tools:
anthropic_model_info = AnthropicModelInfo()
if anthropic_model_info.is_tool_search_used(tools):
# Use provider-specific tool search header
tool_search_header = get_tool_search_beta_header(custom_llm_provider)
beta_values.add(tool_search_header)
if beta_values:
headers["anthropic-beta"] = ",".join(sorted(beta_values))
return headers

View file

@ -664,8 +664,29 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM):
**data, timeout=timeout
)
headers = dict(raw_response.headers)
response = raw_response.parse()
# Convert json.JSONDecodeError to AzureOpenAIError for two critical reasons:
#
# 1. ROUTER BEHAVIOR: The router relies on exception.status_code to determine cooldown logic:
# - JSONDecodeError has no status_code → router skips cooldown evaluation
# - AzureOpenAIError has status_code → router properly evaluates for cooldown
#
# 2. CONNECTION CLEANUP: When response.parse() throws JSONDecodeError, the response
# body may not be fully consumed, preventing httpx from properly returning the
# connection to the pool. By catching the exception and accessing raw_response.status_code,
# we trigger httpx's internal cleanup logic. Without this:
# - parse() fails → JSONDecodeError bubbles up → httpx never knows response was acknowledged → connection leak
# This completely eliminates "Unclosed connection" warnings during high load.
try:
response = raw_response.parse()
except json.JSONDecodeError as json_error:
raise AzureOpenAIError(
status_code=raw_response.status_code or 500,
message=f"Failed to parse raw Azure embedding response: {str(json_error)}"
) from json_error
stringified_response = response.model_dump()
## LOGGING
logging_obj.post_call(
input=input,

View file

@ -62,10 +62,10 @@ class AzureAnthropicMessagesConfig(AnthropicMessagesConfig):
if "content-type" not in headers:
headers["content-type"] = "application/json"
# Update headers with optional anthropic beta features
headers = self._update_headers_with_optional_anthropic_beta(
# Update headers with anthropic beta features (context management, tool search, etc.)
headers = self._update_headers_with_anthropic_beta(
headers=headers,
context_management=optional_params.get("context_management"),
optional_params=optional_params,
)
return headers, api_base

View file

@ -425,6 +425,15 @@ def strip_bedrock_routing_prefix(model: str) -> str:
return model
def strip_bedrock_throughput_suffix(model: str) -> str:
""" Strip throughput tier suffixes from Bedrock model names. """
import re
# Pattern matches model:version:throughput where throughput is like 51k, 18k, etc.
# Keep the model:version part, strip the :throughput suffix
return re.sub(r"(:\d+):\d+k$", r"\1", model)
def get_bedrock_base_model(model: str) -> str:
"""
Get the base model from the given model name.
@ -432,9 +441,11 @@ def get_bedrock_base_model(model: str) -> str:
Handle model names like:
- "us.meta.llama3-2-11b-instruct-v1:0" -> "meta.llama3-2-11b-instruct-v1"
- "bedrock/converse/model" -> "model"
- "anthropic.claude-3-5-sonnet-20241022-v2:0:51k" -> "anthropic.claude-3-5-sonnet-20241022-v2:0"
"""
model = strip_bedrock_routing_prefix(model)
model = extract_model_name_from_bedrock_arn(model)
model = strip_bedrock_throughput_suffix(model)
potential_region = model.split(".", 1)[0]
alt_potential_region = model.split("/", 1)[0]

View file

@ -129,6 +129,37 @@ class AmazonAnthropicClaudeMessagesConfig(
if isinstance(cache_control, dict) and "ttl" in cache_control:
cache_control.pop("ttl", None)
def _get_tool_search_beta_header_for_bedrock(
self,
model: str,
tool_search_used: bool,
programmatic_tool_calling_used: bool,
input_examples_used: bool,
beta_set: set,
) -> None:
"""
Adjust tool search beta header for Bedrock.
Bedrock requires a different beta header for tool search on Opus 4 models
when tool search is used without programmatic tool calling or input examples.
Note: On Amazon Bedrock, server-side tool search is only supported on Claude Opus 4
with the `tool-search-tool-2025-10-19` beta header.
Ref: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool
Args:
model: The model name
tool_search_used: Whether tool search is used
programmatic_tool_calling_used: Whether programmatic tool calling is used
input_examples_used: Whether input examples are used
beta_set: The set of beta headers to modify in-place
"""
if tool_search_used and not (programmatic_tool_calling_used or input_examples_used):
beta_set.discard(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
if "opus-4" in model.lower() or "opus_4" in model.lower():
beta_set.add("tool-search-tool-2025-10-19")
def transform_anthropic_messages_request(
self,
model: str,
@ -189,13 +220,13 @@ class AmazonAnthropicClaudeMessagesConfig(
)
beta_set.update(auto_betas)
if (
tool_search_used
and not (programmatic_tool_calling_used or input_examples_used)
):
beta_set.discard(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
if "opus-4" in model.lower() or "opus_4" in model.lower():
beta_set.add("tool-search-tool-2025-10-19")
self._get_tool_search_beta_header_for_bedrock(
model=model,
tool_search_used=tool_search_used,
programmatic_tool_calling_used=programmatic_tool_calling_used,
input_examples_used=input_examples_used,
beta_set=beta_set,
)
if beta_set:
anthropic_messages_request["anthropic_beta"] = list(beta_set)

View file

@ -57,6 +57,19 @@ class OpenAIRealtime(OpenAIChatCompletion):
try:
ssl_context = get_shared_realtime_ssl_context()
# Log a masked request preview consistent with other endpoints.
logging_obj.pre_call(
input=None,
api_key=api_key,
additional_args={
"api_base": url,
"headers": {
"Authorization": f"Bearer {api_key}",
"OpenAI-Beta": "realtime=v1",
},
"complete_input_dict": {"query_params": query_params},
},
)
async with websockets.connect( # type: ignore
url,
additional_headers={

View file

@ -0,0 +1,13 @@
from litellm.llms.base_llm.image_generation.transformation import (
BaseImageGenerationConfig,
)
from .transformation import OpenRouterImageGenerationConfig
__all__ = [
"OpenRouterImageGenerationConfig",
]
def get_openrouter_image_generation_config(model: str) -> BaseImageGenerationConfig:
return OpenRouterImageGenerationConfig()

View file

@ -0,0 +1,414 @@
"""
OpenRouter Image Generation Support
OpenRouter provides image generation through chat completion endpoints.
Models like google/gemini-2.5-flash-image return images in the message content.
Response format:
{
"choices": [{
"message": {
"content": "Here is a beautiful sunset for you! ",
"role": "assistant",
"images": [{
"image_url": {"url": "data:image/png;base64,..."},
"index": 0,
"type": "image_url"
}]
}
}],
"usage": {
"completion_tokens": 1299,
"prompt_tokens": 6,
"total_tokens": 1305,
"completion_tokens_details": {"image_tokens": 1290},
"cost": 0.0387243
}
}
"""
from typing import TYPE_CHECKING, Any, List, Optional, Union
import httpx
import litellm
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.image_generation.transformation import (
BaseImageGenerationConfig,
)
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams, AllMessageValues
from litellm.types.utils import ImageObject, ImageResponse, ImageUsage, ImageUsageInputTokensDetails
from litellm.llms.openrouter.common_utils import OpenRouterException
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
else:
LiteLLMLoggingObj = Any
class OpenRouterImageGenerationConfig(BaseImageGenerationConfig):
"""
Configuration for OpenRouter image generation via chat completions.
OpenRouter uses chat completion endpoints for image generation,
so we need to transform image generation requests to chat format
and extract images from chat responses.
"""
def get_supported_openai_params(
self, model: str
) -> List[OpenAIImageGenerationOptionalParams]:
"""
Get supported OpenAI parameters for OpenRouter image generation.
Since OpenRouter uses chat completions for image generation,
we support standard image generation params.
"""
return [
"size",
"quality",
"n",
]
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
"""
Map image generation params to OpenRouter chat completion format.
Maps OpenAI parameters to OpenRouter's image_config format:
- size -> image_config.aspect_ratio
- quality -> image_config.image_size
"""
supported_params = self.get_supported_openai_params(model)
for key, value in non_default_params.items():
if key in supported_params:
if key == "size":
# Map OpenAI size to OpenRouter aspect_ratio
aspect_ratio = self._map_size_to_aspect_ratio(value)
if "image_config" not in optional_params:
optional_params["image_config"] = {}
optional_params["image_config"]["aspect_ratio"] = aspect_ratio
elif key == "quality":
# Map OpenAI quality to OpenRouter image_size
image_size = self._map_quality_to_image_size(value)
if image_size:
if "image_config" not in optional_params:
optional_params["image_config"] = {}
optional_params["image_config"]["image_size"] = image_size
else:
# Pass through other supported params (like n)
optional_params[key] = value
elif not drop_params:
# If not supported and drop_params is False, pass through
optional_params[key] = value
return optional_params
def _map_size_to_aspect_ratio(self, size: str) -> str:
"""
Map OpenAI size format to OpenRouter aspect_ratio format.
OpenAI sizes:
- 1024x1024 (square)
- 1536x1024 (landscape)
- 1024x1536 (portrait)
- 1792x1024 (wide landscape, dall-e-3)
- 1024x1792 (tall portrait, dall-e-3)
- 256x256, 512x512 (dall-e-2)
- auto (default)
OpenRouter aspect_ratios:
- 1:1 1024×1024 (default)
- 2:3 832×1248
- 3:2 1248×832
- 3:4 864×1184
- 4:3 1184×864
- 4:5 896×1152
- 5:4 1152×896
- 9:16 768×1344
- 16:9 1344×768
- 21:9 1536×672
"""
size_to_aspect_ratio = {
# Square formats
"256x256": "1:1",
"512x512": "1:1",
"1024x1024": "1:1",
# Landscape formats
"1536x1024": "3:2", # 1.5:1 ratio, closest to 3:2
"1792x1024": "16:9", # 1.75:1 ratio, closest to 16:9
# Portrait formats
"1024x1536": "2:3", # 0.67:1 ratio, closest to 2:3
"1024x1792": "9:16", # 0.57:1 ratio, closest to 9:16
# Default
"auto": "1:1",
}
return size_to_aspect_ratio.get(size, "1:1")
def _map_quality_to_image_size(self, quality: str) -> Optional[str]:
"""
Map OpenAI quality to OpenRouter image_size format.
OpenAI quality values:
- auto (default) - automatically select best quality
- high, medium, low - for GPT image models
- hd, standard - for dall-e-3
OpenRouter image_size values (Gemini only):
- 1K Standard resolution (default)
- 2K Higher resolution
- 4K Highest resolution
"""
quality_to_image_size = {
# OpenAI quality mappings
"low": "1K",
"standard": "1K",
"medium": "2K",
"high": "4K",
"hd": "4K",
# Auto defaults to standard
"auto": "1K",
}
return quality_to_image_size.get(quality)
def _set_usage_and_cost(
self,
model_response: ImageResponse,
response_json: dict,
model: str,
) -> None:
"""
Extract and set usage and cost information from OpenRouter response.
Args:
model_response: ImageResponse object to populate
response_json: Parsed JSON response from OpenRouter
model: The model name
"""
usage_data = response_json.get("usage", {})
if usage_data:
prompt_tokens = usage_data.get("prompt_tokens", 0)
total_tokens = usage_data.get("total_tokens", 0)
completion_tokens_details = usage_data.get("completion_tokens_details", {})
image_tokens = completion_tokens_details.get("image_tokens", 0)
model_response.usage = ImageUsage(
input_tokens=prompt_tokens,
input_tokens_details=ImageUsageInputTokensDetails(
image_tokens=0, # Input doesn't contain images for generation
text_tokens=prompt_tokens,
),
output_tokens=image_tokens,
total_tokens=total_tokens,
)
cost = usage_data.get("cost")
if cost is not None:
if not hasattr(model_response, "_hidden_params"):
model_response._hidden_params = {}
if "additional_headers" not in model_response._hidden_params:
model_response._hidden_params["additional_headers"] = {}
model_response._hidden_params["additional_headers"][
"llm_provider-x-litellm-response-cost"
] = float(cost)
cost_details = usage_data.get("cost_details", {})
if cost_details:
if "response_cost_details" not in model_response._hidden_params:
model_response._hidden_params["response_cost_details"] = {}
model_response._hidden_params["response_cost_details"].update(cost_details)
model_response._hidden_params["model"] = response_json.get("model", model)
def get_complete_url(
self,
api_base: Optional[str],
api_key: Optional[str],
model: str,
optional_params: dict,
litellm_params: dict,
stream: Optional[bool] = None,
) -> str:
"""
Get the complete URL for OpenRouter image generation.
OpenRouter uses chat completions endpoint for image generation.
Default: https://openrouter.ai/api/v1/chat/completions
"""
if api_base:
if not api_base.endswith("/chat/completions"):
api_base = api_base.rstrip("/")
return f"{api_base}/chat/completions"
return api_base
return "https://openrouter.ai/api/v1/chat/completions"
def validate_environment(
self,
headers: dict,
model: str,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
api_key: Optional[str] = None,
api_base: Optional[str] = None,
) -> dict:
api_key = (
api_key
or litellm.api_key
or get_secret_str("OPENROUTER_API_KEY")
)
headers.update(
{
"Authorization": f"Bearer {api_key}",
}
)
return headers
def transform_image_generation_request(
self,
model: str,
prompt: str,
optional_params: dict,
litellm_params: dict,
headers: dict,
) -> dict:
"""
Transform image generation request to OpenRouter chat completion format.
Args:
model: The model name
prompt: The image generation prompt
optional_params: Optional parameters (including image_config)
litellm_params: LiteLLM parameters
headers: Request headers
Returns:
dict: Request body in chat completion format with image_config
"""
request_body = {
"model": model,
"messages": [
{
"role": "user",
"content": prompt
}
]
}
# These will be passed through to OpenRouter
for key, value in optional_params.items():
if key not in ["model", "messages", "modalities"]:
request_body[key] = value
return request_body
def transform_image_generation_response(
self,
model: str,
raw_response: httpx.Response,
model_response: ImageResponse,
logging_obj: LiteLLMLoggingObj,
request_data: dict,
optional_params: dict,
litellm_params: dict,
encoding: Any,
api_key: Optional[str] = None,
json_mode: Optional[bool] = None,
) -> ImageResponse:
"""
Transform OpenRouter chat completion response to ImageResponse format.
Extracts images from the message content and maps usage/cost information.
Args:
model: The model name
raw_response: Raw HTTP response from OpenRouter
model_response: ImageResponse object to populate
logging_obj: Logging object
request_data: Original request data
optional_params: Optional parameters
litellm_params: LiteLLM parameters
encoding: Encoding
api_key: API key
json_mode: JSON mode flag
Returns:
ImageResponse: Populated image response
"""
try:
response_json = raw_response.json()
except Exception as e:
raise OpenRouterException(
message=f"Error parsing OpenRouter response: {str(e)}",
status_code=raw_response.status_code,
headers=raw_response.headers,
)
if not model_response.data:
model_response.data = []
try:
choices = response_json.get("choices", [])
for choice in choices:
message = choice.get("message", {})
images = message.get("images", [])
for image_data in images:
image_url_obj = image_data.get("image_url", {})
image_url = image_url_obj.get("url")
if image_url:
if image_url.startswith("data:"):
# Extract base64 data
# Format: data:image/png;base64,<base64_data>
parts = image_url.split(",", 1)
b64_data = parts[1] if len(parts) > 1 else None
model_response.data.append(
ImageObject(
b64_json=b64_data,
url=None,
revised_prompt=None,
)
)
else:
model_response.data.append(
ImageObject(
b64_json=None,
url=image_url,
revised_prompt=None,
)
)
# Extract and set usage and cost information
self._set_usage_and_cost(model_response, response_json, model)
return model_response
except Exception as e:
raise OpenRouterException(
message=f"Error transforming OpenRouter image generation response: {str(e)}",
status_code=500,
headers={},
)
def get_error_class(
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
) -> BaseLLMException:
"""Get the appropriate error class for OpenRouter errors."""
return OpenRouterException(
message=error_message,
status_code=status_code,
headers=headers,
)

View file

@ -665,11 +665,11 @@ def add_object_type(schema):
if "required" in schema and schema["required"] is None:
schema.pop("required", None)
# Gemini doesn't accept empty properties for object types
# If properties is empty, remove it and the type field
# If properties is empty, remove it but keep type as object
if not properties:
schema.pop("properties", None)
schema.pop("type", None)
schema.pop("required", None)
schema["type"] = "object"
else:
schema["type"] = "object"
for name, value in properties.items():
@ -776,6 +776,16 @@ def get_vertex_location_from_url(url: str) -> Optional[str]:
return match.group(1) if match else None
def get_vertex_model_id_from_url(url: str) -> Optional[str]:
"""
Get the vertex model id from the url
`https://${LOCATION}-aiplatform.googleapis.com/v1/projects/${PROJECT_ID}/locations/${LOCATION}/publishers/google/models/${MODEL_ID}:streamGenerateContent`
"""
match = re.search(r"/models/([^/:]+)", url)
return match.group(1) if match else None
def replace_project_and_location_in_route(
requested_route: str, vertex_project: str, vertex_location: str
) -> str:
@ -825,6 +835,15 @@ def construct_target_url(
if "cachedContent" in requested_route:
vertex_version = "v1beta1"
# Check if the requested route starts with a version
# e.g. /v1beta1/publishers/google/models/gemini-3-pro-preview:streamGenerateContent
if requested_route.startswith("/v1/"):
vertex_version = "v1"
requested_route = requested_route.replace("/v1/", "/", 1)
elif requested_route.startswith("/v1beta1/"):
vertex_version = "v1beta1"
requested_route = requested_route.replace("/v1beta1/", "/", 1)
base_requested_route = "{}/projects/{}/locations/{}".format(
vertex_version, vertex_project, vertex_location
)

View file

@ -1,11 +1,16 @@
from typing import Any, Dict, List, Optional, Tuple
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
from litellm.types.llms.anthropic import (
ANTHROPIC_BETA_HEADER_VALUES,
ANTHROPIC_HOSTED_TOOLS,
)
from litellm.types.llms.anthropic_tool_search import get_tool_search_beta_header
from litellm.types.llms.vertex_ai import VertexPartnerProvider
from litellm.types.router import GenericLiteLLMParams
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES, ANTHROPIC_HOSTED_TOOLS
from ....vertex_llm_base import VertexBase
@ -51,13 +56,28 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
headers["content-type"] = "application/json"
# Add web search beta header for Vertex AI only if not already set
if "anthropic-beta" not in headers:
tools = optional_params.get("tools", [])
for tool in tools:
if isinstance(tool, dict) and tool.get("type", "").startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value):
headers["anthropic-beta"] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
break
# Add beta headers for Vertex AI
tools = optional_params.get("tools", [])
beta_values: set[str] = set()
# Get existing beta headers if any
existing_beta = headers.get("anthropic-beta")
if existing_beta:
beta_values.update(b.strip() for b in existing_beta.split(","))
# Check for web search tool
for tool in tools:
if isinstance(tool, dict) and tool.get("type", "").startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value):
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value)
break
# Check for tool search tools - Vertex AI uses different beta header
anthropic_model_info = AnthropicModelInfo()
if anthropic_model_info.is_tool_search_used(tools):
beta_values.add(get_tool_search_beta_header("vertex_ai"))
if beta_values:
headers["anthropic-beta"] = ",".join(beta_values)
return headers, api_base

View file

@ -28,6 +28,7 @@ from typing import (
Callable,
Coroutine,
Dict,
Iterable,
List,
Literal,
Mapping,
@ -1094,23 +1095,68 @@ def completion( # type: ignore # noqa: PLR0915
# validate tool_choice
tool_choice = validate_chat_completion_tool_choice(tool_choice=tool_choice)
######### unpacking kwargs #####################
args = locals()
skip_mcp_handler = kwargs.pop("_skip_mcp_handler", False)
if not skip_mcp_handler and tools:
from litellm.responses.mcp.chat_completions_handler import (
handle_chat_completion_with_mcp,
acompletion_with_mcp,
)
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
LiteLLM_Proxy_MCP_Handler,
)
from litellm.types.llms.openai import ToolParam
mcp_handler_context = locals().copy()
completion_callable = globals().get("acompletion")
mcp_result = run_async_function(
handle_chat_completion_with_mcp,
mcp_handler_context,
completion_callable,
)
if mcp_result is not None:
return mcp_result
######### unpacking kwargs #####################
args = locals()
# Check if MCP tools are present (following responses pattern)
# Cast tools to Optional[Iterable[ToolParam]] for type checking
tools_for_mcp = cast(Optional[Iterable[ToolParam]], tools)
if LiteLLM_Proxy_MCP_Handler._should_use_litellm_mcp_gateway(tools=tools_for_mcp):
# Return coroutine - acompletion will await it
# completion() can return a coroutine when MCP tools are present, which acompletion() awaits
return acompletion_with_mcp( # type: ignore[return-value]
model=model,
messages=messages,
functions=functions,
function_call=function_call,
timeout=timeout,
temperature=temperature,
top_p=top_p,
n=n,
stream=stream,
stream_options=stream_options,
stop=stop,
max_tokens=max_tokens,
max_completion_tokens=max_completion_tokens,
modalities=modalities,
prediction=prediction,
audio=audio,
presence_penalty=presence_penalty,
frequency_penalty=frequency_penalty,
logit_bias=logit_bias,
user=user,
response_format=response_format,
seed=seed,
tools=tools,
tool_choice=tool_choice,
parallel_tool_calls=parallel_tool_calls,
logprobs=logprobs,
top_logprobs=top_logprobs,
deployment_id=deployment_id,
reasoning_effort=reasoning_effort,
verbosity=verbosity,
safety_identifier=safety_identifier,
service_tier=service_tier,
base_url=base_url,
api_version=api_version,
api_key=api_key,
model_list=model_list,
extra_headers=extra_headers,
thinking=thinking,
web_search_options=web_search_options,
shared_session=shared_session,
**kwargs,
)
api_base = kwargs.get("api_base", None)
mock_response: Optional[MOCK_RESPONSE_TYPE] = kwargs.get("mock_response", None)
mock_tool_calls = kwargs.get("mock_tool_calls", None)

View file

@ -28782,13 +28782,13 @@
"supports_web_search": true
},
"vertex_ai/zai-org/glm-4.7-maas": {
"input_cost_per_token": 3e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "vertex_ai-zai_models",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"output_cost_per_token": 2.2e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
"supports_function_calling": true,
"supports_reasoning": true,

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View file

@ -14,76 +14,3 @@ model_list:
litellm_params:
model: openai/gpt-4.1-mini
# guardrails:
# - guardrail_name: generic-guardrail
# litellm_params:
# guardrail: generic_guardrail_api
# mode: ["pre_call"]
# headers:
# Authorization: Bearer mock-bedrock-token-12345
# api_base: http://localhost:8080
# default_on: true
guardrails:
- guardrail_name: "harmful-content-filter"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
default_on: true
# Model configuration
image_model: "claude-sonnet-4-5-20250929"
categories:
- category: "harmful_self_harm"
enabled: true
action: "BLOCK"
severity_threshold: "medium" # Block medium+
- category: "harmful_violence"
enabled: true
action: "BLOCK"
severity_threshold: "high" # Only explicit
- category: "harmful_illegal_weapons"
enabled: true
action: "BLOCK"
severity_threshold: "low" # Strictest
- category: "bias_gender"
enabled: true
action: "BLOCK"
severity_threshold: "high" # Only explicit to reduce false positives
- category: "bias_sexual_orientation"
enabled: true
action: "BLOCK"
severity_threshold: "high" # Only explicit to reduce false positives
- category: "denied_medical_advice"
enabled: true
action: "BLOCK"
severity_threshold: "high" # Only explicit to reduce false positives
- category: "denied_legal_advice"
enabled: true
action: "BLOCK"
severity_threshold: "high" # Only explicit to reduce false positives
- category: "denied_financial_advice"
enabled: true
action: "BLOCK"
severity_threshold: "high" # Only explicit to reduce false positives
prompts:
- prompt_id: "simple_prompt"
litellm_params:
guardrail: generic_guardrail_api
mode: ["post_call"]
headers:
Authorization: Bearer mock-bedrock-token-12345
api_base: http://localhost:8080
api_key: os.environ/BRAINTRUST_API_KEY
ignore_prompt_manager_model: true
ignore_prompt_manager_optional_params: true

View file

@ -2189,6 +2189,8 @@ class UserAPIKeyAuth(
user_tpm_limit: Optional[int] = None
user_rpm_limit: Optional[int] = None
user_email: Optional[str] = None
user_spend: Optional[float] = None
user_max_budget: Optional[float] = None
request_route: Optional[str] = None
user: Optional[Any] = None # Expanded user object when expand=user is used

View file

@ -74,75 +74,6 @@ db_cache_expiry = DEFAULT_IN_MEMORY_TTL # refresh every 5s
all_routes = LiteLLMRoutes.openai_routes.value + LiteLLMRoutes.management_routes.value
def _is_model_cost_zero(
model: Optional[Union[str, List[str]]], llm_router: Optional[Router]
) -> bool:
"""
Check if a model has zero cost (no configured pricing).
Uses the router's get_model_group_info method to get pricing information.
Args:
model: The model name or list of model names
llm_router: The LiteLLM router instance
Returns:
bool: True if all costs for the model are zero, False otherwise
"""
if model is None or llm_router is None:
return False
# Handle list of models
model_list = [model] if isinstance(model, str) else model
for model_name in model_list:
try:
# Use router's get_model_group_info method directly for better reliability
model_group_info = llm_router.get_model_group_info(model_group=model_name)
if model_group_info is None:
# Model not found or no pricing info available
# Conservative approach: assume it has cost
verbose_proxy_logger.debug(
f"No model group info found for {model_name}, assuming it has cost"
)
return False
# Check costs for this model
# Only allow bypass if BOTH costs are explicitly set to 0 (not None)
input_cost = model_group_info.input_cost_per_token
output_cost = model_group_info.output_cost_per_token
# If costs are not explicitly configured (None), assume it has cost
if input_cost is None or output_cost is None:
verbose_proxy_logger.debug(
f"Model {model_name} has undefined cost (input: {input_cost}, output: {output_cost}), assuming it has cost"
)
return False
# If either cost is non-zero, return False
if input_cost > 0 or output_cost > 0:
verbose_proxy_logger.debug(
f"Model {model_name} has non-zero cost (input: {input_cost}, output: {output_cost})"
)
return False
# This model has zero cost explicitly configured
verbose_proxy_logger.debug(
f"Model {model_name} has zero cost explicitly configured (input: {input_cost}, output: {output_cost})"
)
except Exception as e:
# If we can't determine the cost, assume it has cost (conservative approach)
verbose_proxy_logger.debug(
f"Error checking cost for model {model_name}: {str(e)}, assuming it has cost"
)
return False
# All models checked have zero cost
return True
async def common_checks(
request_body: dict,
team_object: Optional[LiteLLM_TeamTable],
@ -155,7 +86,6 @@ async def common_checks(
proxy_logging_obj: ProxyLogging,
valid_token: Optional[UserAPIKeyAuth],
request: Request,
skip_budget_checks: bool = False,
) -> bool:
"""
Common checks across jwt + key-based auth.
@ -207,66 +137,64 @@ async def common_checks(
user_object=user_object,
)
# If this is a free model, skip all budget checks
if not skip_budget_checks:
# 3. If team is in budget
await _team_max_budget_check(
team_object=team_object,
proxy_logging_obj=proxy_logging_obj,
valid_token=valid_token,
)
# 3. If team is in budget
await _team_max_budget_check(
team_object=team_object,
proxy_logging_obj=proxy_logging_obj,
valid_token=valid_token,
)
# 3.1. If organization is in budget
await _organization_max_budget_check(
valid_token=valid_token,
team_object=team_object,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
)
# 3.1. If organization is in budget
await _organization_max_budget_check(
valid_token=valid_token,
team_object=team_object,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
)
await _tag_max_budget_check(
request_body=request_body,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
valid_token=valid_token,
)
await _tag_max_budget_check(
request_body=request_body,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
valid_token=valid_token,
)
# 4. If user is in budget
## 4.1 check personal budget, if personal key
if (
(team_object is None or team_object.team_id is None)
and user_object is not None
and user_object.max_budget is not None
):
user_budget = user_object.max_budget
if user_budget < user_object.spend:
raise litellm.BudgetExceededError(
current_cost=user_object.spend,
max_budget=user_budget,
message=f"ExceededBudget: User={user_object.user_id} over budget. Spend={user_object.spend}, Budget={user_budget}",
)
# 4. If user is in budget
## 4.1 check personal budget, if personal key
if (
(team_object is None or team_object.team_id is None)
and user_object is not None
and user_object.max_budget is not None
):
user_budget = user_object.max_budget
if user_budget < user_object.spend:
raise litellm.BudgetExceededError(
current_cost=user_object.spend,
max_budget=user_budget,
message=f"ExceededBudget: User={user_object.user_id} over budget. Spend={user_object.spend}, Budget={user_budget}",
)
## 4.2 check team member budget, if team key
await _check_team_member_budget(
team_object=team_object,
user_object=user_object,
valid_token=valid_token,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
)
## 4.2 check team member budget, if team key
await _check_team_member_budget(
team_object=team_object,
user_object=user_object,
valid_token=valid_token,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
)
# 5. If end_user ('user' passed to /chat/completions, /embeddings endpoint) is in budget
if end_user_object is not None and end_user_object.litellm_budget_table is not None:
end_user_budget = end_user_object.litellm_budget_table.max_budget
if end_user_budget is not None and end_user_object.spend > end_user_budget:
raise litellm.BudgetExceededError(
current_cost=end_user_object.spend,
max_budget=end_user_budget,
message=f"ExceededBudget: End User={end_user_object.user_id} over budget. Spend={end_user_object.spend}, Budget={end_user_budget}",
)
# 5. If end_user ('user' passed to /chat/completions, /embeddings endpoint) is in budget
if end_user_object is not None and end_user_object.litellm_budget_table is not None:
end_user_budget = end_user_object.litellm_budget_table.max_budget
if end_user_budget is not None and end_user_object.spend > end_user_budget:
raise litellm.BudgetExceededError(
current_cost=end_user_object.spend,
max_budget=end_user_budget,
message=f"ExceededBudget: End User={end_user_object.user_id} over budget. Spend={end_user_object.spend}, Budget={end_user_budget}",
)
# 6. [OPTIONAL] If 'enforce_user_param' enabled - did developer pass in 'user' param for openai endpoints
if (
@ -309,7 +237,6 @@ async def common_checks(
# 7. [OPTIONAL] If 'litellm.max_budget' is set (>0), is proxy under budget
if (
litellm.max_budget > 0
and not skip_budget_checks
and global_proxy_spend is not None
# only run global budget checks for OpenAI routes
# Reason - the Admin UI should continue working if the proxy crosses it's global budget

View file

@ -7,6 +7,7 @@ from fastapi import HTTPException, Request, status
from litellm import Router, provider_list
from litellm._logging import verbose_proxy_logger
from litellm.constants import STANDARD_CUSTOMER_ID_HEADERS
from litellm.proxy._types import *
from litellm.types.router import CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS
@ -561,6 +562,32 @@ def get_customer_user_header_from_mapping(user_id_mapping) -> Optional[str]:
return header_name
return None
def _get_customer_id_from_standard_headers(
request_headers: Optional[dict],
) -> Optional[str]:
"""
Check standard customer ID headers for a customer/end-user ID.
This enables tools like Claude Code to pass customer IDs via ANTHROPIC_CUSTOM_HEADERS.
No configuration required - these headers are always checked.
Args:
request_headers: The request headers dict
Returns:
The customer ID if found in standard headers, None otherwise
"""
if request_headers is None:
return None
for standard_header in STANDARD_CUSTOMER_ID_HEADERS:
for header_name, header_value in request_headers.items():
if header_name.lower() == standard_header.lower():
user_id_str = str(header_value) if header_value is not None else ""
if user_id_str.strip():
return user_id_str
return None
def get_end_user_id_from_request_body(
request_body: dict, request_headers: Optional[dict] = None
@ -569,7 +596,12 @@ def get_end_user_id_from_request_body(
# and to ensure it's fetched at runtime.
from litellm.proxy.proxy_server import general_settings
# Check 1 : Follow the user header mappings feature, if not found, then check for deprecated user_header_name (only if request_headers is provided)
# Check 1: Standard customer ID headers (always checked, no configuration required)
customer_id = _get_customer_id_from_standard_headers(request_headers=request_headers)
if customer_id is not None:
return customer_id
# Check 2: Follow the user header mappings feature, if not found, then check for deprecated user_header_name (only if request_headers is provided)
# User query: "system not respecting user_header_name property"
# This implies the key in general_settings is 'user_header_name'.
if request_headers is not None:
@ -602,19 +634,19 @@ def get_end_user_id_from_request_body(
if user_id_str.strip():
return user_id_str
# Check 2: 'user' field in request_body (commonly OpenAI)
# Check 3: 'user' field in request_body (commonly OpenAI)
if "user" in request_body and request_body["user"] is not None:
user_from_body_user_field = request_body["user"]
return str(user_from_body_user_field)
# Check 3: 'litellm_metadata.user' in request_body (commonly Anthropic)
# Check 4: 'litellm_metadata.user' in request_body (commonly Anthropic)
litellm_metadata = request_body.get("litellm_metadata")
if isinstance(litellm_metadata, dict):
user_from_litellm_metadata = litellm_metadata.get("user")
if user_from_litellm_metadata is not None:
return str(user_from_litellm_metadata)
# Check 4: 'metadata.user_id' in request_body (another common pattern)
# Check 5: 'metadata.user_id' in request_body (another common pattern)
metadata_dict = request_body.get("metadata")
if isinstance(metadata_dict, dict):
user_id_from_metadata_field = metadata_dict.get("user_id")

View file

@ -586,21 +586,6 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
if team_object is not None
else None,
)
# Check if model has zero cost - if so, skip all budget checks
model = get_model_from_request(request_data, route)
skip_budget_checks = False
if model is not None and llm_router is not None:
from litellm.proxy.auth.auth_checks import _is_model_cost_zero
skip_budget_checks = _is_model_cost_zero(
model=model, llm_router=llm_router
)
if skip_budget_checks:
verbose_proxy_logger.info(
f"Skipping all budget checks for zero-cost model: {model}"
)
# run through common checks
_ = await common_checks(
request=request,
@ -614,7 +599,6 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
llm_router=llm_router,
proxy_logging_obj=proxy_logging_obj,
valid_token=valid_token,
skip_budget_checks=skip_budget_checks,
)
# return UserAPIKeyAuth object
@ -1006,22 +990,8 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
)
user_obj = None
# Check 2a. Check if model has zero cost - if so, skip all budget checks
model = get_model_from_request(request_data, route)
skip_budget_checks = False
if model is not None and llm_router is not None:
from litellm.proxy.auth.auth_checks import _is_model_cost_zero
skip_budget_checks = _is_model_cost_zero(
model=model, llm_router=llm_router
)
if skip_budget_checks:
verbose_proxy_logger.info(
f"Skipping all budget checks for zero-cost model: {model}"
)
# Check 3. Check if user is in their team budget
if not skip_budget_checks and valid_token.team_member_spend is not None:
if valid_token.team_member_spend is not None:
if prisma_client is not None:
_cache_key = f"{valid_token.team_id}_{valid_token.user_id}"
@ -1085,47 +1055,46 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
param=abbreviate_api_key(api_key=api_key),
)
if not skip_budget_checks:
# Check 4. Token Spend is under budget
if RouteChecks.is_llm_api_route(route=route):
await _virtual_key_max_budget_check(
valid_token=valid_token,
proxy_logging_obj=proxy_logging_obj,
user_obj=user_obj,
)
# Check 5. Max Budget Alert Check
await _virtual_key_max_budget_alert_check(
# Check 4. Token Spend is under budget
if RouteChecks.is_llm_api_route(route=route):
await _virtual_key_max_budget_check(
valid_token=valid_token,
proxy_logging_obj=proxy_logging_obj,
user_obj=user_obj,
)
# Check 6. Soft Budget Check
await _virtual_key_soft_budget_check(
valid_token=valid_token,
proxy_logging_obj=proxy_logging_obj,
user_obj=user_obj,
# Check 5. Max Budget Alert Check
await _virtual_key_max_budget_alert_check(
valid_token=valid_token,
proxy_logging_obj=proxy_logging_obj,
user_obj=user_obj,
)
# Check 6. Soft Budget Check
await _virtual_key_soft_budget_check(
valid_token=valid_token,
proxy_logging_obj=proxy_logging_obj,
user_obj=user_obj,
)
# Check 5. Token Model Spend is under Model budget
max_budget_per_model = valid_token.model_max_budget
current_model = request_data.get("model", None)
if (
max_budget_per_model is not None
and isinstance(max_budget_per_model, dict)
and len(max_budget_per_model) > 0
and prisma_client is not None
and current_model is not None
and valid_token.token is not None
):
## GET THE SPEND FOR THIS MODEL
await model_max_budget_limiter.is_key_within_model_budget(
user_api_key_dict=valid_token,
model=current_model,
)
# Check 5. Token Model Spend is under Model budget
max_budget_per_model = valid_token.model_max_budget
current_model = request_data.get("model", None)
if (
max_budget_per_model is not None
and isinstance(max_budget_per_model, dict)
and len(max_budget_per_model) > 0
and prisma_client is not None
and current_model is not None
and valid_token.token is not None
):
## GET THE SPEND FOR THIS MODEL
await model_max_budget_limiter.is_key_within_model_budget(
user_api_key_dict=valid_token,
model=current_model,
)
# Check 6: Additional Common Checks across jwt + key auth
if valid_token.team_id is not None:
_team_obj: Optional[LiteLLM_TeamTable] = LiteLLM_TeamTable(
@ -1193,7 +1162,6 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
llm_router=llm_router,
proxy_logging_obj=proxy_logging_obj,
valid_token=valid_token,
skip_budget_checks=skip_budget_checks,
)
# Token passed all checks
if valid_token is None:
@ -1335,6 +1303,8 @@ async def _return_user_api_key_auth_obj(
user_tpm_limit=user_obj.tpm_limit,
user_rpm_limit=user_obj.rpm_limit,
user_email=user_obj.user_email,
user_spend=getattr(user_obj, "spend", None),
user_max_budget=getattr(user_obj, "max_budget", None),
)
if user_obj is not None and _is_user_proxy_admin(user_obj=user_obj):
user_api_key_kwargs.update(

View file

@ -50,8 +50,32 @@ from litellm.types.proxy.guardrails.guardrail_hooks.litellm_content_filter impor
ContentFilterDetection,
PatternDetection,
)
from .patterns import PATTERN_EXTRA_CONFIG, get_compiled_pattern
from .patterns import get_compiled_pattern
MAX_KEYWORD_VALUE_GAP_WORDS = 1
GAP_WORD_TOKENIZER = re.compile(r"\b\w+\b")
WORD_NUMBER_MAP = {
"zero": "0",
"oh": "0",
"one": "1",
"two": "2",
"three": "3",
"four": "4",
"five": "5",
"six": "6",
"seven": "7",
"eight": "8",
"nine": "9",
}
WORD_NUMBER_TOKEN_REGEX = "|".join(WORD_NUMBER_MAP.keys())
WORD_NUMBER_SEQUENCE_PATTERN = re.compile(
rf"(?<![A-Za-z])(?:{WORD_NUMBER_TOKEN_REGEX})(?:[\s\-]+(?:{WORD_NUMBER_TOKEN_REGEX}))+(?![A-Za-z])",
re.IGNORECASE,
)
WORD_NUMBER_TOKEN_FINDER = re.compile(rf"(?:{WORD_NUMBER_TOKEN_REGEX})", re.IGNORECASE)
# Helper data structure for category-based detection
@ -144,9 +168,9 @@ class ContentFilterGuardrail(CustomGuardrail):
self.image_model = image_model
# Store loaded categories
self.loaded_categories: Dict[str, CategoryConfig] = {}
self.category_keywords: Dict[str, Tuple[str, str, ContentFilterAction]] = (
{}
) # keyword -> (category, severity, action)
self.category_keywords: Dict[
str, Tuple[str, str, ContentFilterAction]
] = {} # keyword -> (category, severity, action)
# Load categories if provided
if categories:
@ -170,7 +194,7 @@ class ContentFilterGuardrail(CustomGuardrail):
normalized_blocked_words.append(word)
# Compile regex patterns
self.compiled_patterns: List[Tuple[Pattern, str, ContentFilterAction]] = []
self.compiled_patterns: List[Dict[str, Any]] = []
for pattern_config in normalized_patterns:
self._add_pattern(pattern_config)
@ -323,11 +347,13 @@ class ContentFilterGuardrail(CustomGuardrail):
pattern_config: ContentFilterPattern configuration
"""
try:
extra_config: Dict[str, Any] = {}
if pattern_config.pattern_type == "prebuilt":
if not pattern_config.pattern_name:
raise ValueError("pattern_name is required for prebuilt patterns")
compiled = get_compiled_pattern(pattern_config.pattern_name)
pattern_name = pattern_config.pattern_name
extra_config = PATTERN_EXTRA_CONFIG.get(pattern_name, {}) or {}
elif pattern_config.pattern_type == "regex":
if not pattern_config.pattern:
raise ValueError("pattern is required for regex patterns")
@ -336,8 +362,20 @@ class ContentFilterGuardrail(CustomGuardrail):
else:
raise ValueError(f"Unknown pattern_type: {pattern_config.pattern_type}")
keyword_regex: Optional[Pattern] = None
if extra_config.get("keyword_pattern"):
keyword_regex = re.compile(
extra_config["keyword_pattern"], re.IGNORECASE
)
self.compiled_patterns.append(
(compiled, pattern_name, pattern_config.action)
{
"regex": compiled,
"pattern_name": pattern_name,
"action": pattern_config.action,
"keyword_regex": keyword_regex,
"allow_word_numbers": bool(extra_config.get("allow_word_numbers")),
}
)
verbose_proxy_logger.debug(
f"Added pattern: {pattern_name} with action {pattern_config.action}"
@ -395,6 +433,130 @@ class ContentFilterGuardrail(CustomGuardrail):
except Exception as e:
raise Exception(f"Error loading blocked words file {file_path}: {str(e)}")
def _find_pattern_spans(
self, text: str, pattern_entry: Dict[str, Any]
) -> List[Tuple[int, int]]:
"""Return all match spans for a pattern, applying contextual rules if required."""
regex: Pattern = pattern_entry["regex"]
keyword_regex: Optional[Pattern] = pattern_entry.get("keyword_regex")
allow_word_numbers: bool = pattern_entry.get("allow_word_numbers", False)
keyword_matches: Optional[List[re.Match]] = None
if keyword_regex is not None:
keyword_matches = list(keyword_regex.finditer(text))
if not keyword_matches:
return []
match_spans: List[Tuple[int, int]] = []
for match in regex.finditer(text):
if keyword_matches is not None and not self._match_near_keyword(
match.start(), match.end(), keyword_matches, text
):
continue
match_spans.append((match.start(), match.end()))
if allow_word_numbers:
for word_match in WORD_NUMBER_SEQUENCE_PATTERN.finditer(text):
digits = self._convert_word_number_sequence(word_match.group())
if not digits:
continue
if not regex.fullmatch(digits):
continue
if keyword_matches is not None and not self._match_near_keyword(
word_match.start(), word_match.end(), keyword_matches, text
):
continue
match_spans.append((word_match.start(), word_match.end()))
return self._merge_spans(match_spans)
def _match_near_keyword(
self,
value_start: int,
value_end: int,
keyword_matches: List[re.Match],
text: str,
) -> bool:
"""Check if a value is separated from a keyword by an allowed gap."""
for keyword_match in keyword_matches:
keyword_start = keyword_match.start()
keyword_end = keyword_match.end()
if value_start >= keyword_end:
gap_text = text[keyword_end:value_start]
elif keyword_start >= value_end:
gap_text = text[value_end:keyword_start]
else:
return True # overlapping
if self._gap_text_allowed(gap_text):
return True
return False
def _gap_text_allowed(self, gap_text: str) -> bool:
"""Return True if the gap between keyword and value meets word-count rules."""
if not gap_text.strip():
return True
if any(char.isdigit() for char in gap_text):
return False
words = GAP_WORD_TOKENIZER.findall(gap_text)
return len(words) <= MAX_KEYWORD_VALUE_GAP_WORDS
def _merge_spans(self, spans: List[Tuple[int, int]]) -> List[Tuple[int, int]]:
"""Merge overlapping spans to avoid double-masking."""
if not spans:
return []
spans.sort(key=lambda item: item[0])
merged: List[Tuple[int, int]] = [spans[0]]
for start, end in spans[1:]:
last_start, last_end = merged[-1]
if start <= last_end:
merged[-1] = (last_start, max(last_end, end))
else:
merged.append((start, end))
return merged
def _mask_spans(
self, text: str, spans: List[Tuple[int, int]], redaction: str
) -> str:
"""Apply masking for the provided spans using the given redaction tag."""
if not spans:
return text
result_parts: List[str] = []
previous_end = 0
for start, end in spans:
result_parts.append(text[previous_end:start])
result_parts.append(redaction)
previous_end = end
result_parts.append(text[previous_end:])
return "".join(result_parts)
def _convert_word_number_sequence(self, sequence: str) -> Optional[str]:
"""Convert a spelled-out digit sequence (e.g., 'One-Two') into digits."""
tokens = WORD_NUMBER_TOKEN_FINDER.findall(sequence)
if not tokens:
return None
digits: List[str] = []
for token in tokens:
digit = WORD_NUMBER_MAP.get(token.lower())
if digit is None:
return None
digits.append(digit)
return "".join(digits) if digits else None
def _check_patterns(
self, text: str
) -> Optional[Tuple[str, str, ContentFilterAction]]:
@ -407,10 +569,13 @@ class ContentFilterGuardrail(CustomGuardrail):
Returns:
Tuple of (matched_text, pattern_name, action) if match found, None otherwise
"""
for compiled_pattern, pattern_name, action in self.compiled_patterns:
match = compiled_pattern.search(text)
if match:
matched_text = match.group(0)
for pattern_entry in self.compiled_patterns:
spans = self._find_pattern_spans(text, pattern_entry)
if spans:
start, end = spans[0]
matched_text = text[start:end]
pattern_name = pattern_entry["pattern_name"]
action = pattern_entry["action"]
verbose_proxy_logger.debug(
f"Pattern '{pattern_name}' matched: {matched_text[:20]}..."
)
@ -582,11 +747,13 @@ class ContentFilterGuardrail(CustomGuardrail):
)
# Check regex patterns - process ALL patterns, not just first match
for compiled_pattern, pattern_name, action in self.compiled_patterns:
match = compiled_pattern.search(text)
if not match:
for pattern_entry in self.compiled_patterns:
spans = self._find_pattern_spans(text, pattern_entry)
if not spans:
continue
pattern_name = pattern_entry["pattern_name"]
action = pattern_entry["action"]
if detections is not None:
# Don't log matched_text to avoid exposing sensitive content (emails, credit cards, etc.)
pattern_detection: PatternDetection = {
@ -604,11 +771,10 @@ class ContentFilterGuardrail(CustomGuardrail):
detail={"error": error_msg, "pattern": pattern_name},
)
elif action == ContentFilterAction.MASK:
# Replace ALL matches of this pattern with redaction tag
redaction_tag = self.pattern_redaction_format.format(
pattern_name=pattern_name.upper()
)
text = compiled_pattern.sub(redaction_tag, text)
text = self._mask_spans(text, spans, redaction_tag)
verbose_proxy_logger.info(
f"Masked all {pattern_name} matches in content"
)
@ -924,19 +1090,28 @@ class ContentFilterGuardrail(CustomGuardrail):
if pattern_match:
matched_text, pattern_name, action = pattern_match
if action == ContentFilterAction.BLOCK:
error_msg = f"Content blocked: {pattern_name} pattern detected"
error_msg = (
f"Content blocked: {pattern_name} pattern detected"
)
verbose_proxy_logger.warning(error_msg)
raise HTTPException(
status_code=403,
detail={"error": error_msg, "pattern": pattern_name},
detail={
"error": error_msg,
"pattern": pattern_name,
},
)
# Check blocked words
blocked_word_match = self._check_blocked_words(accumulated_content)
blocked_word_match = self._check_blocked_words(
accumulated_content
)
if blocked_word_match:
keyword, action, description = blocked_word_match
if action == ContentFilterAction.BLOCK:
error_msg = f"Content blocked: keyword '{keyword}' detected"
error_msg = (
f"Content blocked: keyword '{keyword}' detected"
)
if description:
error_msg += f" ({description})"
verbose_proxy_logger.warning(error_msg)

View file

@ -120,11 +120,11 @@
"description": "Detects URLs (http/https)"
},
{
"name": "passport_us",
"display_name": "Passport (US)",
"pattern": "\\b[0-9]{9}\\b",
"category": "PII Patterns",
"description": "US passport numbers (9 digits)"
"name": "passport_us",
"display_name": "Passport (US)",
"pattern": "\\b[0-9]{9}\\b",
"category": "PII Patterns",
"description": "US passport numbers (9 digits)"
},
{
"name": "passport_uk",
@ -203,7 +203,6 @@
"category": "Protected Class - Fair Lending",
"description": "Detects race, ethnicity and national origin terms - protected under ECOA and Fair Housing Act"
},
{
"name": "religion",
"display_name": "Religion & Creed (Protected Class)",
@ -236,7 +235,7 @@
"name": "military_status",
"display_name": "Military Status (Protected Class)",
"pattern": "\\b(veteran|military|armed\\s+forces|army|navy|air\\s+force|marine(s|\\s+corps)?|coast\\s+guard|national\\s+guard|reserve(s|ist)?|active\\s+duty|deployment|deployed|enlisted|commissioned|honorable\\s+discharge|dishonorable\\s+discharge|VA\\s+benefits|GI\\s+bill|military\\s+service|service\\s+member|servicemember|SCRA|MLA|military\\s+lending)\\b",
"category": "Protected Class - Fair Lending",
"category": "Protected Class - Fair Lending",
"description": "Detects military status terms - protected under SCRA and MLA"
},
{
@ -245,7 +244,7 @@
"pattern": "\\b(welfare|public\\s+assistance|food\\s+stamps|SNAP|WIC|TANF|medicaid|section\\s+8|housing\\s+voucher|subsidized\\s+housing|public\\s+housing|government\\s+benefits|social\\s+services|unemployment\\s+(benefits|insurance)|UI\\s+benefits|EBT|benefit\\s+recipient)\\b",
"category": "Protected Class - Fair Lending",
"description": "Detects public assistance terms - protected under ECOA"
} ,
},
{
"name": "weapons_firearms",
"display_name": "Weapons & Firearms",
@ -313,10 +312,12 @@
{
"name": "nl_bsn_contextual",
"display_name": "BSN (Dutch Citizen Service Number)",
"pattern": "\\b(?:BSN|B\\.S\\.N\\.|burgerservicenummer|burger\\s*service\\s*nummer|sofi\\s*nummer|sofinummer|persoonsnummer|identificatienummer|citizen\\s*service\\s*number)[:\\s]*[0-9]{9}\\b|\\b[0-9]{9}\\b(?=\\s*(?:BSN|burgerservicenummer|sofinummer))",
"pattern": "\\b[0-9]{9}\\b",
"category": "PII Patterns",
"action": "MASK",
"description": "Detects Dutch BSN numbers with contextual keywords"
"description": "Detects Dutch BSN numbers with contextual keywords",
"keyword_pattern": "(?:\\b(?:BSN|B\\.S\\.N\\.|burgerservicenummer|burger\\s*service\\s*nummer|sofi\\s*nummer|sofinummer|persoonsnummer|identificatienummer|citizen\\s*service\\s*number)\\b|8\\s*5\\s*\\|\\\\\\|)",
"allow_word_numbers": true
},
{
"name": "br_cpf",
@ -369,5 +370,3 @@
}
]
}

View file

@ -9,7 +9,7 @@ import json
import os
import re
from enum import Enum
from typing import Dict, List, Pattern
from typing import Any, Dict, List, Pattern
def _load_patterns_from_json() -> Dict:
@ -41,6 +41,26 @@ PREBUILT_PATTERNS: Dict[str, str] = {
}
# Capture any extra configuration declared per pattern (e.g., contextual keywords)
KNOWN_PATTERN_KEYS = {
"name",
"display_name",
"pattern",
"category",
"action",
"description",
}
PATTERN_EXTRA_CONFIG: Dict[str, Dict[str, Any]] = {}
for pattern_data in _PATTERNS_DATA["patterns"]:
extra_config = {
key: value
for key, value in pattern_data.items()
if key not in KNOWN_PATTERN_KEYS
}
PATTERN_EXTRA_CONFIG[pattern_data["name"]] = extra_config
def get_compiled_pattern(pattern_name: str) -> Pattern:
"""
Get a compiled regex pattern by name.

View file

@ -79,8 +79,12 @@ class UnifiedLLMGuardrails(CustomLogger):
endpoint_guardrail_translation_mappings = (
load_guardrail_translation_mappings()
)
if CallTypes(call_type) not in endpoint_guardrail_translation_mappings:
return data
try:
if CallTypes(call_type) not in endpoint_guardrail_translation_mappings:
return data
except ValueError:
return data # handle unmapped call types
endpoint_translation = endpoint_guardrail_translation_mappings[
CallTypes(call_type)

View file

@ -114,25 +114,25 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
) -> Optional[str]:
"""
Get priority from user_api_key_dict.
Checks team metadata first (takes precedence), then falls back to key metadata.
Args:
user_api_key_dict: User authentication info
Returns:
Priority string if found, None otherwise
"""
priority: Optional[str] = None
# Check team metadata first (takes precedence)
if user_api_key_dict.team_metadata is not None:
priority = user_api_key_dict.team_metadata.get("priority", None)
# Fall back to key metadata
if priority is None:
priority = user_api_key_dict.metadata.get("priority", None)
return priority
def _normalize_priority_weights(
@ -299,10 +299,13 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
"""
descriptors: List[RateLimitDescriptor] = []
if litellm.priority_reservation is None:
return descriptors
# Get model group info
model_group_info: Optional[ModelGroupInfo] = (
self.llm_router.get_model_group_info(model_group=model)
)
model_group_info: Optional[
ModelGroupInfo
] = self.llm_router.get_model_group_info(model_group=model)
if model_group_info is None:
return descriptors
@ -577,9 +580,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
)
# Get model configuration
model_group_info: Optional[ModelGroupInfo] = (
self.llm_router.get_model_group_info(model_group=model)
)
model_group_info: Optional[
ModelGroupInfo
] = self.llm_router.get_model_group_info(model_group=model)
if model_group_info is None:
verbose_proxy_logger.debug(
f"No model group info for {model}, allowing request"
@ -703,7 +706,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
# Get priority from user_api_key_auth_metadata in standard_logging_metadata
# This is where user_api_key_dict.metadata is stored during pre-call
user_api_key_auth_metadata = standard_logging_metadata.get("user_api_key_auth_metadata") or {}
user_api_key_auth_metadata = (
standard_logging_metadata.get("user_api_key_auth_metadata") or {}
)
key_priority: Optional[str] = user_api_key_auth_metadata.get("priority")
# Get total tokens from response
@ -775,7 +780,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
# Only log 'priority' if it's known safe; otherwise, redact.
SAFE_PRIORITIES = {"low", "medium", "high", "default"}
logged_priority = key_priority if key_priority in SAFE_PRIORITIES else "REDACTED"
logged_priority = (
key_priority if key_priority in SAFE_PRIORITIES else "REDACTED"
)
verbose_proxy_logger.debug(
f"[Dynamic Rate Limiter] Incremented tokens by {total_tokens} for "
f"model={model_group}, priority={logged_priority}"

View file

@ -1236,7 +1236,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
return pipeline_operations
def _get_total_tokens_from_usage(
self, usage: Any | None, rate_limit_type: Literal["output", "input", "total"]
self, usage: Optional[Any], rate_limit_type: Literal["output", "input", "total"]
) -> int:
"""
Get total tokens from response usage for rate limiting.

View file

@ -1000,6 +1000,13 @@ async def add_litellm_data_to_request( # noqa: PLR0915
"user_api_key_model_max_budget"
] = user_api_key_dict.model_max_budget
# User spend, budget - used by prometheus.py
# Follow same pattern as team and API key budgets
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
data[_metadata_variable_name][
"user_api_key_user_max_budget"
] = user_api_key_dict.user_max_budget
data[_metadata_variable_name]["user_api_key_metadata"] = user_api_key_dict.metadata
_headers = dict(request.headers)
_headers.pop(

View file

@ -343,7 +343,7 @@ def _build_where_conditions(
start_date: str,
end_date: str,
model: Optional[str],
api_key: Optional[Union[str, List[str]]],
api_key: Optional[str],
exclude_entity_ids: Optional[List[str]] = None,
) -> Dict[str, Any]:
"""Build prisma where clause for daily activity queries."""
@ -357,10 +357,7 @@ def _build_where_conditions(
if model:
where_conditions["model"] = model
if api_key:
if isinstance(api_key, list):
where_conditions["api_key"] = {"in": api_key}
else:
where_conditions["api_key"] = api_key
where_conditions["api_key"] = api_key
if entity_id is not None:
if isinstance(entity_id, list):
@ -448,7 +445,7 @@ async def get_daily_activity(
start_date: Optional[str],
end_date: Optional[str],
model: Optional[str],
api_key: Optional[Union[str, List[str]]],
api_key: Optional[str],
page: int,
page_size: int,
exclude_entity_ids: Optional[List[str]] = None,

View file

@ -412,6 +412,13 @@ async def new_user(
status_code=403,
detail="License is over limit. Please contact support@berri.ai to upgrade your license.",
)
# Only proxy admins can create administrative users
if data.user_role in [LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY] and user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN:
raise HTTPException(
status_code=403,
detail=f"Only proxy admins can create administrative users (proxy_admin, proxy_admin_viewer). Attempted to create user with role: {data.user_role}. Your role: {user_api_key_dict.user_role}"
)
data_json = data.json() # type: ignore
data_json = _update_internal_new_user_params(data_json, data)

View file

@ -3601,7 +3601,7 @@ async def get_team_daily_activity(
},
)
## Fetch team aliases and check team admin status
## Fetch team aliases
where_condition = {}
if team_ids_list:
where_condition["team_id"] = {"in": list(team_ids_list)}
@ -3612,36 +3612,6 @@ async def get_team_daily_activity(
t.team_id: {"team_alias": t.team_alias} for t in team_aliases
}
# Check if user is team admin for any requested teams
# If not, filter by user's API keys
user_api_keys: Optional[List[str]] = None
if not _user_has_admin_view(user_api_key_dict) and team_ids_list and team_aliases:
# Check if user is team admin for any of the teams
is_team_admin_for_any = False
for team_alias in team_aliases:
team_obj = LiteLLM_TeamTable(**team_alias.model_dump())
if _is_user_team_admin(
user_api_key_dict=user_api_key_dict, team_obj=team_obj
):
is_team_admin_for_any = True
break
# If user is not a team admin for any team, filter by their API keys
if not is_team_admin_for_any:
# Get all API keys for this user
user_keys = await prisma_client.db.litellm_verificationtoken.find_many(
where={"user_id": user_api_key_dict.user_id}
)
user_api_keys = [key.token for key in user_keys if key.token]
# If user has no API keys, return empty result
if not user_api_keys:
user_api_keys = [""] # Use empty string to ensure no matches
# If api_key parameter is provided, use it; otherwise use user_api_keys if set
final_api_key_filter: Optional[Union[str, List[str]]] = api_key
if final_api_key_filter is None and user_api_keys is not None:
final_api_key_filter = user_api_keys
return await get_daily_activity(
prisma_client=prisma_client,
table_name="litellm_dailyteamspend",
@ -3652,7 +3622,7 @@ async def get_team_daily_activity(
start_date=start_date,
end_date=end_date,
model=model,
api_key=final_api_key_filter,
api_key=api_key,
page=page,
page_size=page_size,
)

View file

@ -761,7 +761,6 @@ async def handle_bedrock_passthrough_router_model(
proxy_logging_obj=proxy_logging_obj,
)
async def handle_bedrock_count_tokens(
endpoint: str,
request: Request,
@ -1555,6 +1554,7 @@ async def _base_vertex_proxy_route(
from litellm.llms.vertex_ai.common_utils import (
construct_target_url,
get_vertex_location_from_url,
get_vertex_model_id_from_url,
get_vertex_project_id_from_url,
)
@ -1584,6 +1584,25 @@ async def _base_vertex_proxy_route(
vertex_location=vertex_location,
)
if vertex_project is None or vertex_location is None:
# Check if model is in router config
model_id = get_vertex_model_id_from_url(endpoint)
if model_id:
from litellm.proxy.proxy_server import llm_router
if llm_router:
try:
# Use the dedicated pass-through deployment selection method to automatically filter use_in_pass_through=True
deployment = llm_router.get_available_deployment_for_pass_through(model=model_id)
if deployment:
litellm_params = deployment.get("litellm_params", {})
vertex_project = litellm_params.get("vertex_project")
vertex_location = litellm_params.get("vertex_location")
except Exception as e:
verbose_proxy_logger.debug(
f"Error getting available deployment for model {model_id}: {e}"
)
vertex_credentials = passthrough_endpoint_router.get_vertex_credentials(
project_id=vertex_project,
location=vertex_location,

View file

@ -37,6 +37,12 @@ model_list:
model_info:
litellm_provider: bedrock_converse
mode: chat
- model_name: azure-claude-opus-4-5
litellm_params:
model: azure_ai/claude-opus-4-5
api_base: https://krish-mh44t553-eastus2.services.ai.azure.com
api_key: os.environ/AZURE_ANTHROPIC_API_KEY
general_settings:
store_prompts_in_spend_logs: true

View file

@ -3253,20 +3253,22 @@ class ProxyConfig:
) -> Optional[dict]:
"""
Get router_settings in priority order: Key > Team > Global
Returns:
dict: Combined router_settings, or None if no settings found
"""
if prisma_client is None:
return None
import json
import yaml
# 1. Try key-level router_settings
if user_api_key_dict is not None:
# Check if router_settings is available on the key object
key_router_settings_value = getattr(user_api_key_dict, "router_settings", None)
key_router_settings_value = getattr(
user_api_key_dict, "router_settings", None
)
if key_router_settings_value is not None:
key_router_settings = None
if isinstance(key_router_settings_value, str):
@ -3279,11 +3281,15 @@ class ProxyConfig:
pass
elif isinstance(key_router_settings_value, dict):
key_router_settings = key_router_settings_value
# If key has router_settings (non-empty dict), use it
if key_router_settings is not None and isinstance(key_router_settings, dict) and key_router_settings:
if (
key_router_settings is not None
and isinstance(key_router_settings, dict)
and key_router_settings
):
return key_router_settings
# 2. Try team-level router_settings
if user_api_key_dict is not None and user_api_key_dict.team_id is not None:
try:
@ -3291,37 +3297,51 @@ class ProxyConfig:
where={"team_id": user_api_key_dict.team_id}
)
if team_obj is not None:
team_router_settings_value = getattr(team_obj, "router_settings", None)
team_router_settings_value = getattr(
team_obj, "router_settings", None
)
if team_router_settings_value is not None:
team_router_settings = None
if isinstance(team_router_settings_value, str):
try:
team_router_settings = yaml.safe_load(team_router_settings_value)
team_router_settings = yaml.safe_load(
team_router_settings_value
)
except (yaml.YAMLError, json.JSONDecodeError):
try:
team_router_settings = json.loads(team_router_settings_value)
team_router_settings = json.loads(
team_router_settings_value
)
except json.JSONDecodeError:
pass
elif isinstance(team_router_settings_value, dict):
team_router_settings = team_router_settings_value
# If team has router_settings (non-empty dict), use it
if team_router_settings is not None and isinstance(team_router_settings, dict) and team_router_settings:
if (
team_router_settings is not None
and isinstance(team_router_settings, dict)
and team_router_settings
):
return team_router_settings
except Exception:
# If team lookup fails, continue to global settings
pass
# 3. Try global router_settings
try:
db_router_settings = await prisma_client.db.litellm_config.find_first(
where={"param_name": "router_settings"}
)
if db_router_settings is not None and isinstance(db_router_settings.param_value, dict) and db_router_settings.param_value:
if (
db_router_settings is not None
and isinstance(db_router_settings.param_value, dict)
and db_router_settings.param_value
):
return db_router_settings.param_value
except Exception:
pass
return None
async def _add_router_settings_from_db_config(
@ -4688,27 +4708,48 @@ class ProxyStartupEvent:
### SPEND LOG CLEANUP ###
if general_settings.get("maximum_spend_logs_retention_period") is not None:
spend_log_cleanup = SpendLogCleanup()
# Get the interval from config or default to 1 day
retention_interval = general_settings.get(
"maximum_spend_logs_retention_interval", "1d"
)
try:
interval_seconds = duration_in_seconds(retention_interval)
scheduler.add_job(
spend_log_cleanup.cleanup_old_spend_logs,
"interval",
seconds=interval_seconds
+ random.randint(0, 60), # Add small random offset
# REMOVED jitter parameter - major cause of memory leak
args=[prisma_client],
id="spend_log_cleanup_job",
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
except ValueError:
verbose_proxy_logger.error(
"Invalid maximum_spend_logs_retention_interval value"
cleanup_cron = general_settings.get("maximum_spend_logs_cleanup_cron")
if cleanup_cron:
from apscheduler.triggers.cron import CronTrigger
try:
cron_trigger = CronTrigger.from_crontab(cleanup_cron)
scheduler.add_job(
spend_log_cleanup.cleanup_old_spend_logs,
cron_trigger,
args=[prisma_client],
id="spend_log_cleanup_job",
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
verbose_proxy_logger.info(
f"Spend log cleanup scheduled with cron: {cleanup_cron}"
)
except ValueError:
verbose_proxy_logger.error(
f"Invalid maximum_spend_logs_cleanup_cron value: {cleanup_cron}"
)
else:
# Interval-based scheduling (existing behavior)
retention_interval = general_settings.get(
"maximum_spend_logs_retention_interval", "1d"
)
try:
interval_seconds = duration_in_seconds(retention_interval)
scheduler.add_job(
spend_log_cleanup.cleanup_old_spend_logs,
"interval",
seconds=interval_seconds + random.randint(0, 60),
args=[prisma_client],
id="spend_log_cleanup_job",
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
except ValueError:
verbose_proxy_logger.error(
"Invalid maximum_spend_logs_retention_interval value"
)
### CHECK BATCH COST ###
if llm_router is not None:
try:
@ -9922,7 +9963,9 @@ async def get_config(): # noqa: PLR0915
_success_callbacks = normalize_callback(_success_callbacks)
_failure_callbacks = normalize_callback(_failure_callbacks)
_success_and_failure_callbacks = normalize_callback(_success_and_failure_callbacks)
_success_and_failure_callbacks = normalize_callback(
_success_and_failure_callbacks
)
_data_to_return = []
"""

View file

@ -72,6 +72,11 @@ class UISettings(BaseModel):
description="If true, internal users cannot add models from the UI",
)
disable_team_admin_delete_team_user: bool = Field(
default=False,
description="Prevents Team Admins from deleting users from the teams they manage. Useful for SCIM provisioning where team membership is defined externally.",
)
class UISettingsResponse(SettingsResponse):
"""Response model for UI settings"""
@ -80,7 +85,7 @@ class UISettingsResponse(SettingsResponse):
# Allowlist of UI settings that can be stored
ALLOWED_UI_SETTINGS_FIELDS = {"disable_model_add_for_internal_users"}
ALLOWED_UI_SETTINGS_FIELDS = {"disable_model_add_for_internal_users", "disable_team_admin_delete_team_user"}
@router.get(

View file

@ -61,6 +61,10 @@ async def _arealtime(
api_key=api_key,
)
# Ensure query params use the normalized provider model (no proxy aliases).
if query_params is not None:
query_params = {**query_params, "model": model}
litellm_logging_obj.update_environment_variables(
model=model,
user=user,

View file

@ -2,127 +2,67 @@
from typing import (
Any,
Awaitable,
Callable,
Dict,
Iterable,
List,
Optional,
Union,
cast,
)
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
LiteLLM_Proxy_MCP_Handler,
)
from litellm.responses.utils import ResponsesAPIRequestUtils
from litellm.types.llms.openai import ToolParam
from litellm.types.utils import ModelResponse
from litellm.utils import CustomStreamWrapper
CompletionCallable = Callable[..., Awaitable[Union[ModelResponse, CustomStreamWrapper]]]
_CHAT_COMPLETION_CALL_ARG_KEYS = [
"model",
"messages",
"functions",
"function_call",
"timeout",
"temperature",
"top_p",
"n",
"stream",
"stream_options",
"stop",
"max_tokens",
"max_completion_tokens",
"modalities",
"prediction",
"audio",
"presence_penalty",
"frequency_penalty",
"logit_bias",
"user",
"response_format",
"seed",
"tools",
"tool_choice",
"parallel_tool_calls",
"logprobs",
"top_logprobs",
"deployment_id",
"reasoning_effort",
"verbosity",
"safety_identifier",
"service_tier",
"base_url",
"api_version",
"api_key",
"model_list",
"extra_headers",
"thinking",
"web_search_options",
"shared_session",
]
def _build_call_args_from_context(call_context: Dict[str, Any]) -> Dict[str, Any]:
"""Build kwargs for `acompletion` from the `completion` call context."""
call_args = {
key: call_context.get(key)
for key in _CHAT_COMPLETION_CALL_ARG_KEYS
if key in call_context
}
additional_kwargs = dict(call_context.get("kwargs") or {})
call_args.update(additional_kwargs)
return call_args
async def _call_acompletion_internal(
completion_callable: CompletionCallable, **call_args: Any
async def acompletion_with_mcp(
model: str,
messages: List,
tools: Optional[List] = None,
**kwargs: Any,
) -> Union[ModelResponse, CustomStreamWrapper]:
"""Invoke `acompletion` while skipping MCP interception to avoid recursion."""
"""
Async completion with MCP integration.
safe_args = dict(call_args)
safe_args["_skip_mcp_handler"] = True
safe_args.pop("acompletion", None)
return await completion_callable(**safe_args)
This function handles MCP tool integration following the same pattern as aresponses_api_with_mcp.
It's designed to be called from the synchronous completion() function and return a coroutine.
When MCP tools with server_url="litellm_proxy" are provided, this function will:
1. Get available tools from the MCP server manager
2. Transform them to OpenAI format
3. Call acompletion with the transformed tools
4. If require_approval="never" and tool calls are returned, automatically execute them
5. Make a follow-up call with the tool results
"""
from litellm import acompletion as litellm_acompletion
async def handle_chat_completion_with_mcp(
call_context: Dict[str, Any],
completion_callable: CompletionCallable,
) -> Optional[Union[ModelResponse, CustomStreamWrapper]]:
"""Handle MCP-enabled tool execution for chat completion requests."""
# Parse MCP tools and separate from other tools
(
mcp_tools_with_litellm_proxy,
other_tools,
) = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools)
call_args = _build_call_args_from_context(call_context)
if not mcp_tools_with_litellm_proxy:
# No MCP tools, proceed with regular completion
return await litellm_acompletion(
model=model,
messages=messages,
tools=tools,
**kwargs,
)
tools = call_args.get("tools")
if not tools:
return None
tools_for_mcp = cast(Optional[Iterable[ToolParam]], tools)
if not LiteLLM_Proxy_MCP_Handler._should_use_litellm_mcp_gateway(
tools=tools_for_mcp
):
return None
mcp_tools, _ = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools)
if not mcp_tools:
return None
base_call_args = dict(call_args)
user_api_key_auth = call_args.get("user_api_key_auth") or (
(call_args.get("metadata", {}) or {}).get("user_api_key_auth")
# Extract user_api_key_auth from metadata or kwargs
user_api_key_auth = kwargs.get("user_api_key_auth") or (
(kwargs.get("metadata", {}) or {}).get("user_api_key_auth")
)
# Process MCP tools
(
deduplicated_mcp_tools,
tool_server_map,
) = await LiteLLM_Proxy_MCP_Handler._process_mcp_tools_without_openai_transform(
user_api_key_auth=user_api_key_auth,
mcp_tools_with_litellm_proxy=mcp_tools,
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
)
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(
@ -130,25 +70,43 @@ async def handle_chat_completion_with_mcp(
target_format="chat",
)
base_call_args["tools"] = openai_tools or None
# Combine with other tools
all_tools = openai_tools + other_tools if (openai_tools or other_tools) else None
# Determine if we should auto-execute tools
should_auto_execute = LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
mcp_tools_with_litellm_proxy=mcp_tools
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
)
# Extract MCP auth headers
(
mcp_auth_header,
mcp_server_auth_headers,
oauth2_headers,
raw_headers,
) = ResponsesAPIRequestUtils.extract_mcp_headers_from_request(
secret_fields=base_call_args.get("secret_fields"),
secret_fields=kwargs.get("secret_fields"),
tools=tools,
)
if not should_auto_execute:
return await _call_acompletion_internal(completion_callable, **base_call_args)
# Prepare call parameters
# Remove keys that shouldn't be passed to acompletion
clean_kwargs = {k: v for k, v in kwargs.items() if k not in ["acompletion"]}
base_call_args = {
"model": model,
"messages": messages,
"tools": all_tools,
"_skip_mcp_handler": True, # Prevent recursion
**clean_kwargs,
}
# If not auto-executing, just make the call with transformed tools
if not should_auto_execute:
return await litellm_acompletion(**base_call_args)
# For auto-execute: disable streaming for initial call
stream = kwargs.get("stream", False)
mock_tool_calls = base_call_args.pop("mock_tool_calls", None)
initial_call_args = dict(base_call_args)
@ -156,23 +114,26 @@ async def handle_chat_completion_with_mcp(
if mock_tool_calls is not None:
initial_call_args["mock_tool_calls"] = mock_tool_calls
initial_response = await _call_acompletion_internal(
completion_callable, **initial_call_args
)
# Make initial call
initial_response = await litellm_acompletion(**initial_call_args)
if not isinstance(initial_response, ModelResponse):
return initial_response
# Extract tool calls from response
tool_calls = LiteLLM_Proxy_MCP_Handler._extract_tool_calls_from_chat_response(
response=initial_response
)
if not tool_calls:
if base_call_args.get("stream"):
# No tool calls, return response or retry with streaming if needed
if stream:
retry_args = dict(base_call_args)
retry_args["stream"] = call_args.get("stream")
return await _call_acompletion_internal(completion_callable, **retry_args)
retry_args["stream"] = stream
return await litellm_acompletion(**retry_args)
return initial_response
# Execute tool calls
tool_results = await LiteLLM_Proxy_MCP_Handler._execute_tool_calls(
tool_server_map=tool_server_map,
tool_calls=tool_calls,
@ -186,14 +147,16 @@ async def handle_chat_completion_with_mcp(
if not tool_results:
return initial_response
# Create follow-up messages with tool results
follow_up_messages = LiteLLM_Proxy_MCP_Handler._create_follow_up_messages_for_chat(
original_messages=call_args.get("messages", []),
original_messages=messages,
response=initial_response,
tool_results=tool_results,
)
# Make follow-up call with original stream setting
follow_up_call_args = dict(base_call_args)
follow_up_call_args["messages"] = follow_up_messages
follow_up_call_args["stream"] = call_args.get("stream")
follow_up_call_args["stream"] = stream
return await _call_acompletion_internal(completion_callable, **follow_up_call_args)
return await litellm_acompletion(**follow_up_call_args)

View file

@ -8032,6 +8032,154 @@ class Router:
)
raise e
async def async_get_available_deployment_for_pass_through(
self,
model: str,
request_kwargs: Dict,
messages: Optional[List[Dict[str, str]]] = None,
input: Optional[Union[str, List]] = None,
specific_deployment: Optional[bool] = False,
):
"""
Async version of get_available_deployment_for_pass_through
Only returns deployments configured with use_in_pass_through=True
"""
try:
parent_otel_span = _get_parent_otel_span_from_kwargs(request_kwargs)
# 1. Execute pre-routing hook
pre_routing_hook_response = await self.async_pre_routing_hook(
model=model,
request_kwargs=request_kwargs,
messages=messages,
input=input,
specific_deployment=specific_deployment,
)
if pre_routing_hook_response is not None:
model = pre_routing_hook_response.model
messages = pre_routing_hook_response.messages
# 2. Get healthy deployments
healthy_deployments = await self.async_get_healthy_deployments(
model=model,
request_kwargs=request_kwargs,
messages=messages,
input=input,
specific_deployment=specific_deployment,
parent_otel_span=parent_otel_span,
)
# 3. If specific deployment returned, verify if it supports pass-through
if isinstance(healthy_deployments, dict):
litellm_params = healthy_deployments.get("litellm_params", {})
if litellm_params.get("use_in_pass_through"):
return healthy_deployments
else:
raise litellm.BadRequestError(
message=f"Deployment {healthy_deployments.get('model_info', {}).get('id')} does not support pass-through endpoint (use_in_pass_through=False)",
model=model,
llm_provider="",
)
# 4. Filter deployments that support pass-through
pass_through_deployments = self._filter_pass_through_deployments(
healthy_deployments=healthy_deployments
)
if len(pass_through_deployments) == 0:
raise litellm.BadRequestError(
message=f"Model {model} has no deployments configured with use_in_pass_through=True. Please add use_in_pass_through: true to the deployment configuration",
model=model,
llm_provider="",
)
# 5. Apply load balancing strategy
start_time = time.perf_counter()
if (
self.routing_strategy == "usage-based-routing-v2"
and self.lowesttpm_logger_v2 is not None
):
deployment = (
await self.lowesttpm_logger_v2.async_get_available_deployments(
model_group=model,
healthy_deployments=pass_through_deployments, # type: ignore
messages=messages,
input=input,
)
)
elif (
self.routing_strategy == "latency-based-routing"
and self.lowestlatency_logger is not None
):
deployment = (
await self.lowestlatency_logger.async_get_available_deployments(
model_group=model,
healthy_deployments=pass_through_deployments, # type: ignore
messages=messages,
input=input,
request_kwargs=request_kwargs,
)
)
elif self.routing_strategy == "simple-shuffle":
return simple_shuffle(
llm_router_instance=self,
healthy_deployments=pass_through_deployments,
model=model,
)
elif (
self.routing_strategy == "least-busy"
and self.leastbusy_logger is not None
):
deployment = (
await self.leastbusy_logger.async_get_available_deployments(
model_group=model,
healthy_deployments=pass_through_deployments, # type: ignore
)
)
else:
deployment = None
if deployment is None:
exception = await async_raise_no_deployment_exception(
litellm_router_instance=self,
model=model,
parent_otel_span=parent_otel_span,
)
raise exception
verbose_router_logger.info(
f"async_get_available_deployment_for_pass_through model: {model}, selected deployment: {self.print_deployment(deployment)}"
)
end_time = time.perf_counter()
_duration = end_time - start_time
asyncio.create_task(
self.service_logger_obj.async_service_success_hook(
service=ServiceTypes.ROUTER,
duration=_duration,
call_type="<routing_strategy>.async_get_available_deployments",
parent_otel_span=parent_otel_span,
start_time=start_time,
end_time=end_time,
)
)
return deployment
except Exception as e:
traceback_exception = traceback.format_exc()
if request_kwargs is not None:
logging_obj = request_kwargs.get("litellm_logging_obj", None)
if logging_obj is not None:
threading.Thread(
target=logging_obj.failure_handler,
args=(e, traceback_exception),
).start()
asyncio.create_task(
logging_obj.async_failure_handler(e, traceback_exception) # type: ignore
)
raise e
async def async_pre_routing_hook(
self,
model: str,
@ -8184,6 +8332,169 @@ class Router:
)
return deployment
def get_available_deployment_for_pass_through(
self,
model: str,
messages: Optional[List[Dict[str, str]]] = None,
input: Optional[Union[str, List]] = None,
specific_deployment: Optional[bool] = False,
request_kwargs: Optional[Dict] = None,
):
"""
Returns deployments available for pass-through endpoints (based on load balancing strategy)
Similar to get_available_deployment, but only returns deployments with use_in_pass_through=True
Args:
model: Model name
messages: Optional list of messages
input: Optional input data
specific_deployment: Whether to find a specific deployment
request_kwargs: Optional request parameters
Returns:
Dict: Selected deployment configuration
Raises:
BadRequestError: If no deployment is configured with use_in_pass_through=True
RouterRateLimitError: If no pass-through deployments are available
"""
# 1. Perform common checks to get healthy deployments list
model, healthy_deployments = self._common_checks_available_deployment(
model=model,
messages=messages,
input=input,
specific_deployment=specific_deployment,
)
# 2. If the returned is a specific deployment (Dict), verify and return directly
if isinstance(healthy_deployments, dict):
litellm_params = healthy_deployments.get("litellm_params", {})
if litellm_params.get("use_in_pass_through"):
return healthy_deployments
else:
# Specific deployment does not support pass-through
raise litellm.BadRequestError(
message=f"Deployment {healthy_deployments.get('model_info', {}).get('id')} does not support pass-through endpoint (use_in_pass_through=False)",
model=model,
llm_provider="",
)
# 3. Filter deployments that support pass-through
pass_through_deployments = self._filter_pass_through_deployments(
healthy_deployments=healthy_deployments
)
if len(pass_through_deployments) == 0:
# No deployments support pass-through
raise litellm.BadRequestError(
message=f"Model {model} has no deployment configured with use_in_pass_through=True. Please add use_in_pass_through: true in the deployment configuration",
model=model,
llm_provider="",
)
# 4. Apply cooldown filtering
parent_otel_span: Optional[Span] = _get_parent_otel_span_from_kwargs(
request_kwargs
)
cooldown_deployments = _get_cooldown_deployments(
litellm_router_instance=self, parent_otel_span=parent_otel_span
)
pass_through_deployments = self._filter_cooldown_deployments(
healthy_deployments=pass_through_deployments,
cooldown_deployments=cooldown_deployments,
)
# 5. Apply pre-call checks (if enabled)
if self.enable_pre_call_checks and messages is not None:
pass_through_deployments = self._pre_call_checks(
model=model,
healthy_deployments=pass_through_deployments,
messages=messages,
request_kwargs=request_kwargs,
)
if len(pass_through_deployments) == 0:
model_ids = self.get_model_ids(model_name=model)
_cooldown_time = self.cooldown_cache.get_min_cooldown(
model_ids=model_ids, parent_otel_span=parent_otel_span
)
_cooldown_list = _get_cooldown_deployments(
litellm_router_instance=self, parent_otel_span=parent_otel_span
)
raise RouterRateLimitError(
model=model,
cooldown_time=_cooldown_time,
enable_pre_call_checks=self.enable_pre_call_checks,
cooldown_list=_cooldown_list,
)
# 6. Apply load balancing strategy
if self.routing_strategy == "least-busy" and self.leastbusy_logger is not None:
deployment = self.leastbusy_logger.get_available_deployments(
model_group=model, healthy_deployments=pass_through_deployments # type: ignore
)
elif self.routing_strategy == "simple-shuffle":
return simple_shuffle(
llm_router_instance=self,
healthy_deployments=pass_through_deployments,
model=model,
)
elif (
self.routing_strategy == "latency-based-routing"
and self.lowestlatency_logger is not None
):
deployment = self.lowestlatency_logger.get_available_deployments(
model_group=model,
healthy_deployments=pass_through_deployments, # type: ignore
request_kwargs=request_kwargs,
)
elif (
self.routing_strategy == "usage-based-routing"
and self.lowesttpm_logger is not None
):
deployment = self.lowesttpm_logger.get_available_deployments(
model_group=model,
healthy_deployments=pass_through_deployments, # type: ignore
messages=messages,
input=input,
)
elif (
self.routing_strategy == "usage-based-routing-v2"
and self.lowesttpm_logger_v2 is not None
):
deployment = self.lowesttpm_logger_v2.get_available_deployments(
model_group=model,
healthy_deployments=pass_through_deployments, # type: ignore
messages=messages,
input=input,
)
else:
deployment = None
if deployment is None:
verbose_router_logger.info(
f"get_available_deployment_for_pass_through model: {model}, no available deployments"
)
model_ids = self.get_model_ids(model_name=model)
_cooldown_time = self.cooldown_cache.get_min_cooldown(
model_ids=model_ids, parent_otel_span=parent_otel_span
)
_cooldown_list = _get_cooldown_deployments(
litellm_router_instance=self, parent_otel_span=parent_otel_span
)
raise RouterRateLimitError(
model=model,
cooldown_time=_cooldown_time,
enable_pre_call_checks=self.enable_pre_call_checks,
cooldown_list=_cooldown_list,
)
verbose_router_logger.info(
f"get_available_deployment_for_pass_through model: {model}, selected deployment: {self.print_deployment(deployment)}"
)
return deployment
def _filter_cooldown_deployments(
self, healthy_deployments: List[Dict], cooldown_deployments: List[str]
) -> List[Dict]:
@ -8206,6 +8517,34 @@ class Router:
if deployment["model_info"]["id"] not in cooldown_set
]
def _filter_pass_through_deployments(
self, healthy_deployments: List[Dict]
) -> List[Dict]:
"""
Filter out deployments configured with use_in_pass_through=True
Args:
healthy_deployments: List of healthy deployments
Returns:
List[Dict]: Only includes a list of deployments that support pass-through
"""
verbose_router_logger.debug(
f"Filter pass-through deployments from {len(healthy_deployments)} healthy deployments"
)
pass_through_deployments = [
deployment
for deployment in healthy_deployments
if deployment.get("litellm_params", {}).get("use_in_pass_through", False)
]
verbose_router_logger.debug(
f"Found {len(pass_through_deployments)} deployments with pass-through enabled"
)
return pass_through_deployments
def _track_deployment_metrics(
self, deployment, parent_otel_span: Optional[Span], response=None
):

View file

@ -175,6 +175,9 @@ DEFINED_PROMETHEUS_METRICS = Literal[
"litellm_remaining_api_key_budget_metric",
"litellm_api_key_max_budget_metric",
"litellm_api_key_budget_remaining_hours_metric",
"litellm_remaining_user_budget_metric",
"litellm_user_max_budget_metric",
"litellm_user_budget_remaining_hours_metric",
"litellm_deployment_state",
"litellm_deployment_failure_responses",
"litellm_deployment_total_requests",
@ -421,6 +424,18 @@ class PrometheusMetricLabels:
litellm_remaining_api_key_budget_metric
)
litellm_remaining_user_budget_metric = [
UserAPIKeyLabelNames.USER.value,
]
litellm_user_max_budget_metric = [
UserAPIKeyLabelNames.USER.value,
]
litellm_user_budget_remaining_hours_metric = [
UserAPIKeyLabelNames.USER.value,
]
# Add deployment metrics
litellm_deployment_failure_responses = [
UserAPIKeyLabelNames.REQUESTED_MODEL.value,

View file

@ -636,8 +636,10 @@ class ANTHROPIC_BETA_HEADER_VALUES(str, Enum):
ADVANCED_TOOL_USE_2025_11_20 = "advanced-tool-use-2025-11-20"
# Tool search beta header constant
# Tool search beta header constant (for Anthropic direct API and Microsoft Foundry)
ANTHROPIC_TOOL_SEARCH_BETA_HEADER = "advanced-tool-use-2025-11-20"
# Effort beta header constant
ANTHROPIC_EFFORT_BETA_HEADER = "effort-2025-11-24"

View file

@ -0,0 +1,36 @@
"""
Tool Search Beta Header Configuration
Reference: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool
"""
from typing import Dict
from litellm.types.utils import LlmProviders
# Tool search beta header values
TOOL_SEARCH_BETA_HEADER_ANTHROPIC = "advanced-tool-use-2025-11-20"
TOOL_SEARCH_BETA_HEADER_VERTEX = "tool-search-tool-2025-10-19"
TOOL_SEARCH_BETA_HEADER_BEDROCK = "tool-search-tool-2025-10-19"
# Mapping of custom_llm_provider -> tool search beta header
TOOL_SEARCH_BETA_HEADER_BY_PROVIDER: Dict[str, str] = {
LlmProviders.ANTHROPIC.value: TOOL_SEARCH_BETA_HEADER_ANTHROPIC,
LlmProviders.AZURE.value: TOOL_SEARCH_BETA_HEADER_ANTHROPIC,
LlmProviders.AZURE_AI.value: TOOL_SEARCH_BETA_HEADER_ANTHROPIC,
LlmProviders.VERTEX_AI.value: TOOL_SEARCH_BETA_HEADER_VERTEX,
LlmProviders.VERTEX_AI_BETA.value: TOOL_SEARCH_BETA_HEADER_VERTEX,
LlmProviders.BEDROCK.value: TOOL_SEARCH_BETA_HEADER_BEDROCK,
}
def get_tool_search_beta_header(custom_llm_provider: str) -> str:
"""
Get the tool search beta header for a given provider.
"""
return TOOL_SEARCH_BETA_HEADER_BY_PROVIDER.get(
custom_llm_provider,
TOOL_SEARCH_BETA_HEADER_ANTHROPIC
)

View file

@ -364,6 +364,11 @@ class CallTypes(str, Enum):
asend_message = "asend_message"
send_message = "send_message"
#########################################################
# Claude Code Call Types
#########################################################
acreate_skill = "acreate_skill"
CallTypesLiteral = Literal[
"embedding",
@ -420,6 +425,7 @@ CallTypesLiteral = Literal[
"send_message",
"aresponses",
"responses",
"acreate_skill",
]
# Mapping of API routes to their corresponding call types

View file

@ -8354,6 +8354,12 @@ class ProviderConfigManager:
)
return get_vertex_ai_image_generation_config(model)
elif LlmProviders.OPENROUTER == provider:
from litellm.llms.openrouter.image_generation import (
get_openrouter_image_generation_config,
)
return get_openrouter_image_generation_config(model)
return None
@staticmethod

View file

@ -28782,13 +28782,13 @@
"supports_web_search": true
},
"vertex_ai/zai-org/glm-4.7-maas": {
"input_cost_per_token": 3e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "vertex_ai-zai_models",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"output_cost_per_token": 2.2e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
"supports_function_calling": true,
"supports_reasoning": true,

View file

@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm"
version = "1.80.16"
version = "1.80.17"
description = "Library to easily interface with LLM API providers"
authors = ["BerriAI"]
license = "MIT"
@ -167,7 +167,7 @@ requires = ["poetry-core", "wheel"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "1.80.16"
version = "1.80.17"
version_files = [
"pyproject.toml:^version"
]

Some files were not shown because too many files have changed in this diff Show more