mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-06 08:16:43 +00:00
Merge pull request #19210 from BerriAI/main
merge main in bedrock passthrough
This commit is contained in:
commit
b6aa05df16
158 changed files with 7202 additions and 2640 deletions
|
|
@ -1,4 +0,0 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Hello, how are you?"}]}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "What is the weather today?"}]}}
|
||||
{"custom_id": "request-3", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Tell me a short joke"}]}}
|
||||
|
||||
|
|
@ -1,3 +1,3 @@
|
|||
ignore:
|
||||
- vulnerability: CVE-2019-1010022
|
||||
reason: no fixed glibc package is available yet in the Wolfi repositories, so this is ignored temporarily until an upstream release exists
|
||||
- vulnerability: CVE-2026-22184
|
||||
reason: no fixed zlib package is available yet in the Wolfi repositories, so this is ignored temporarily until an upstream release exists
|
||||
|
|
|
|||
|
|
@ -129,11 +129,14 @@ run_grype_scans() {
|
|||
"CVE-2025-13836" # Python 3.13 HTTP response reading OOM/DoS - no fix available in base image
|
||||
"CVE-2025-12084" # Python 3.13 xml.dom.minidom quadratic algorithm - no fix available in base image
|
||||
"CVE-2025-60876" # BusyBox wget HTTP request splitting - no fix available in Chainguard Wolfi base image
|
||||
"CVE-2026-0861" # Wolfi glibc still flagged even on 2.42-r5; upstream patched build unavailable yet
|
||||
"CVE-2010-4756" # glibc glob DoS - awaiting patched Wolfi glibc build
|
||||
"CVE-2019-1010022" # glibc stack guard bypass - awaiting patched Wolfi glibc build
|
||||
"CVE-2019-1010023" # glibc ldd remap issue - awaiting patched Wolfi glibc build
|
||||
"CVE-2019-1010024" # glibc ASLR mitigation bypass - awaiting patched Wolfi glibc build
|
||||
"CVE-2019-1010025" # glibc pthread heap address leak - awaiting patched Wolfi glibc build
|
||||
"CVE-2026-22184" # zlib untgz buffer overflow - untgz unused + no fixed Wolfi build yet
|
||||
"GHSA-58pv-8j8x-9vj2" # jaraco.context path traversal - setuptools vendored only (v5.3.0), not used in application code (using v6.1.0+)
|
||||
)
|
||||
|
||||
# Build JSON array of allowlisted CVE IDs for jq
|
||||
|
|
|
|||
195
cookbook/ai_coding_tool_guides/claude_code_quickstart/guide.md
Normal file
195
cookbook/ai_coding_tool_guides/claude_code_quickstart/guide.md
Normal file
|
|
@ -0,0 +1,195 @@
|
|||
# Claude Code with LiteLLM Quickstart
|
||||
|
||||
This guide shows how to call Claude models (and any LiteLLM-supported model) through LiteLLM proxy from Claude Code.
|
||||
|
||||
> **Note:** This integration is based on [Anthropic's official LiteLLM configuration documentation](https://docs.anthropic.com/en/docs/claude-code/llm-gateway#litellm-configuration). It allows you to use any LiteLLM supported model through Claude Code with centralized authentication, usage tracking, and cost controls.
|
||||
|
||||
## Video Walkthrough
|
||||
|
||||
Watch the full tutorial: https://www.loom.com/embed/3c17d683cdb74d36a3698763cc558f56
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code/overview) installed
|
||||
- API keys for your chosen providers
|
||||
|
||||
## Installation
|
||||
|
||||
First, install LiteLLM with proxy support:
|
||||
|
||||
```bash
|
||||
pip install 'litellm[proxy]'
|
||||
```
|
||||
|
||||
## Step 1: Setup config.yaml
|
||||
|
||||
Create a secure configuration using environment variables:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Claude models
|
||||
- model_name: claude-3-5-sonnet-20241022
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-20241022
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
- model_name: claude-3-5-haiku-20241022
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-haiku-20241022
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
|
||||
litellm_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
```
|
||||
|
||||
Set your environment variables:
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_API_KEY="your-anthropic-api-key"
|
||||
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
|
||||
```
|
||||
|
||||
## Step 2: Start Proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
## Step 3: Verify Setup
|
||||
|
||||
Test that your proxy is working correctly:
|
||||
|
||||
```bash
|
||||
curl -X POST http://0.0.0.0:4000/v1/messages \
|
||||
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet-20241022",
|
||||
"max_tokens": 1000,
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
## Step 4: Configure Claude Code
|
||||
|
||||
### Method 1: Unified Endpoint (Recommended)
|
||||
|
||||
Configure Claude Code to use LiteLLM's unified endpoint. Either a virtual key or master key can be used here:
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000"
|
||||
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
|
||||
```
|
||||
|
||||
> **Tip:** LITELLM_MASTER_KEY gives Claude access to all proxy models, whereas a virtual key would be limited to the models set in the UI.
|
||||
|
||||
### Method 2: Provider-specific Pass-through Endpoint
|
||||
|
||||
Alternatively, use the Anthropic pass-through endpoint:
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000/anthropic"
|
||||
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
|
||||
```
|
||||
|
||||
## Step 5: Use Claude Code
|
||||
|
||||
Start Claude Code and it will automatically use your configured models:
|
||||
|
||||
```bash
|
||||
# Claude Code will use the models configured in your LiteLLM proxy
|
||||
claude
|
||||
|
||||
# Or specify a model if you have multiple configured
|
||||
claude --model claude-3-5-sonnet-20241022
|
||||
claude --model claude-3-5-haiku-20241022
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
Common issues and solutions:
|
||||
|
||||
**Claude Code not connecting:**
|
||||
- Verify your proxy is running: `curl http://0.0.0.0:4000/health`
|
||||
- Check that `ANTHROPIC_BASE_URL` is set correctly
|
||||
- Ensure your `ANTHROPIC_AUTH_TOKEN` matches your LiteLLM master key
|
||||
|
||||
**Authentication errors:**
|
||||
- Verify your environment variables are set: `echo $LITELLM_MASTER_KEY`
|
||||
- Check that your API keys are valid and have sufficient credits
|
||||
- Ensure the `ANTHROPIC_AUTH_TOKEN` matches your LiteLLM master key
|
||||
|
||||
**Model not found:**
|
||||
- Ensure the model name in Claude Code matches exactly with your `config.yaml`
|
||||
- Check LiteLLM logs for detailed error messages
|
||||
|
||||
## Using Multiple Models and Providers
|
||||
|
||||
Expand your configuration to support multiple providers and models:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# OpenAI models
|
||||
- model_name: codex-mini
|
||||
litellm_params:
|
||||
model: openai/codex-mini
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: https://api.openai.com/v1
|
||||
|
||||
- model_name: o3-pro
|
||||
litellm_params:
|
||||
model: openai/o3-pro
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: https://api.openai.com/v1
|
||||
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: https://api.openai.com/v1
|
||||
|
||||
# Anthropic models
|
||||
- model_name: claude-3-5-sonnet-20241022
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-20241022
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
- model_name: claude-3-5-haiku-20241022
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-haiku-20241022
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
# AWS Bedrock
|
||||
- model_name: claude-bedrock
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-east-1
|
||||
|
||||
litellm_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
```
|
||||
|
||||
Switch between models seamlessly:
|
||||
|
||||
```bash
|
||||
# Use Claude for complex reasoning
|
||||
claude --model claude-3-5-sonnet-20241022
|
||||
|
||||
# Use Haiku for fast responses
|
||||
claude --model claude-3-5-haiku-20241022
|
||||
|
||||
# Use Bedrock deployment
|
||||
claude --model claude-bedrock
|
||||
```
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [LiteLLM Documentation](https://docs.litellm.ai/)
|
||||
- [Claude Code Documentation](https://docs.anthropic.com/en/docs/claude-code/overview)
|
||||
- [Anthropic's LiteLLM Configuration Guide](https://docs.anthropic.com/en/docs/claude-code/llm-gateway#litellm-configuration)
|
||||
|
||||
98
cookbook/ai_coding_tool_guides/index.json
Normal file
98
cookbook/ai_coding_tool_guides/index.json
Normal file
|
|
@ -0,0 +1,98 @@
|
|||
[{
|
||||
"title": "Claude Code Quickstart",
|
||||
"description": "This is a quickstart guide to using Claude Code with LiteLLM.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/claude_responses_api",
|
||||
"date": "2026-01-15",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"Claude Code",
|
||||
"LiteLLM"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Claude Code with MCPs",
|
||||
"description": "This is a guide to using Claude Code with MCPs via LiteLLM Proxy.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/claude_mcp",
|
||||
"date": "2026-01-15",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"Claude Code",
|
||||
"LiteLLM",
|
||||
"MCP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Claude Code with Non-Anthropic Models",
|
||||
"description": "This is a guide to using Claude Code with non-Anthropic models via LiteLLM Proxy.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/claude_non_anthropic_models",
|
||||
"date": "2026-01-16",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"Claude Code",
|
||||
"LiteLLM",
|
||||
"OpenAI",
|
||||
"Gemini"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Cursor Quickstart",
|
||||
"description": "This is a quickstart guide to using Cursor with LiteLLM.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/cursor_integration",
|
||||
"date": "2026-01-16",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"Cursor",
|
||||
"LiteLLM",
|
||||
"Quickstart"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Github Copilot Quickstart",
|
||||
"description": "This is a quickstart guide to using Github Copilot with LiteLLM.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/github_copilot_integration",
|
||||
"date": "2026-01-16",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"Github Copilot",
|
||||
"LiteLLM",
|
||||
"Quickstart"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "LiteLLM Gemini CLI Quickstart",
|
||||
"description": "This is a quickstart guide to using LiteLLM Gemini CLI.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/litellm_gemini_cli",
|
||||
"date": "2026-01-16",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"Gemini CLI",
|
||||
"Gemini",
|
||||
"LiteLLM",
|
||||
"Quickstart"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "OpenAI Codex CLI Quickstart",
|
||||
"description": "This is a quickstart guide to using OpenAI Codex CLI.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/openai_codex",
|
||||
"date": "2026-01-16",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"OpenAI Codex CLI",
|
||||
"OpenAI",
|
||||
"LiteLLM",
|
||||
"Quickstart"
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "OpenWebUI Quickstart",
|
||||
"description": "This is a quickstart guide to using OpenWebUI with LiteLLM.",
|
||||
"url": "https://docs.litellm.ai/docs/tutorials/openweb_ui",
|
||||
"date": "2026-01-16",
|
||||
"version": "1.0.0",
|
||||
"tags": [
|
||||
"OpenWebUI",
|
||||
"LiteLLM",
|
||||
"Quickstart"
|
||||
]
|
||||
}]
|
||||
|
|
@ -170,7 +170,8 @@ spec:
|
|||
{{- toYaml .Values.resources | nindent 12 }}
|
||||
volumeMounts:
|
||||
- name: litellm-config
|
||||
mountPath: /etc/litellm/
|
||||
mountPath: /etc/litellm/config.yaml
|
||||
subPath: config.yaml
|
||||
{{ if .Values.securityContext.readOnlyRootFilesystem }}
|
||||
- name: tmp
|
||||
mountPath: /tmp
|
||||
|
|
|
|||
|
|
@ -136,7 +136,8 @@ tests:
|
|||
path: spec.template.spec.containers[0].volumeMounts
|
||||
content:
|
||||
name: litellm-config
|
||||
mountPath: /etc/litellm/
|
||||
mountPath: /etc/litellm/config.yaml
|
||||
subPath: config.yaml
|
||||
- it: should work with lifecycle hooks
|
||||
template: deployment.yaml
|
||||
set:
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ import TabItem from '@theme/TabItem';
|
|||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Guardrails | ✅ | Applies to input prompts (non-streaming only) |
|
||||
| Supported Providers | OpenAI, Azure, Google AI Studio, Vertex AI, AWS Bedrock, Recraft, Xinference, Nscale | |
|
||||
| Supported Providers | OpenAI, Azure, Google AI Studio, Vertex AI, AWS Bedrock, Recraft, OpenRouter, Xinference, Nscale | |
|
||||
|
||||
## Quick Start
|
||||
|
||||
|
|
@ -238,6 +238,27 @@ print(response)
|
|||
|
||||
See Recraft usage with LiteLLM [here](./providers/recraft.md#image-generation)
|
||||
|
||||
## OpenRouter Image Generation Models
|
||||
|
||||
Use this for image generation models available through OpenRouter (e.g., Google Gemini image generation models)
|
||||
|
||||
#### Usage
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ['OPENROUTER_API_KEY'] = "your-api-key"
|
||||
|
||||
response = image_generation(
|
||||
model="openrouter/google/gemini-2.5-flash-image",
|
||||
prompt="A beautiful sunset over a calm ocean",
|
||||
size="1024x1024",
|
||||
quality="high",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## OpenAI Compatible Image Generation Models
|
||||
Use this for calling `/image_generation` endpoints on OpenAI Compatible Servers, example https://github.com/xorbitsai/inference
|
||||
|
||||
|
|
@ -301,5 +322,6 @@ print(f"response: {response}")
|
|||
| Vertex AI | [Vertex AI Image Generation →](./providers/vertex_image) |
|
||||
| AWS Bedrock | [Bedrock Image Generation →](./providers/bedrock) |
|
||||
| Recraft | [Recraft Image Generation →](./providers/recraft#image-generation) |
|
||||
| OpenRouter | [OpenRouter Image Generation →](./providers/openrouter#image-generation) |
|
||||
| Xinference | [Xinference Image Generation →](./providers/xinference#image-generation) |
|
||||
| Nscale | [Nscale Image Generation →](./providers/nscale#image-generation) |
|
||||
|
|
@ -40,6 +40,10 @@ import os
|
|||
# from https://logfire.pydantic.dev/
|
||||
os.environ["LOGFIRE_TOKEN"] = ""
|
||||
|
||||
# Optionally customize the base url
|
||||
# from https://logfire.pydantic.dev/
|
||||
os.environ["LOGFIRE_BASE_URL"] = ""
|
||||
|
||||
# LLM API Keys
|
||||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
|
|
|
|||
|
|
@ -93,3 +93,120 @@ response = embedding(
|
|||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Image Generation
|
||||
|
||||
OpenRouter supports image generation through select models like Google Gemini image generation models. LiteLLM transforms standard image generation requests to OpenRouter's chat completion format.
|
||||
|
||||
### Supported Parameters
|
||||
|
||||
- `size`: Maps to OpenRouter's `aspect_ratio` format
|
||||
- `1024x1024` → `1:1` (square)
|
||||
- `1536x1024` → `3:2` (landscape)
|
||||
- `1024x1536` → `2:3` (portrait)
|
||||
- `1792x1024` → `16:9` (wide landscape)
|
||||
- `1024x1792` → `9:16` (tall portrait)
|
||||
|
||||
- `quality`: Maps to OpenRouter's `image_size` format (Gemini models)
|
||||
- `low` or `standard` → `1K`
|
||||
- `medium` → `2K`
|
||||
- `high` or `hd` → `4K`
|
||||
|
||||
- `n`: Number of images to generate
|
||||
|
||||
### Usage
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ["OPENROUTER_API_KEY"] = "your-api-key"
|
||||
|
||||
# Basic image generation
|
||||
response = image_generation(
|
||||
model="openrouter/google/gemini-2.5-flash-image",
|
||||
prompt="A beautiful sunset over a calm ocean",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Advanced Usage with Parameters
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ["OPENROUTER_API_KEY"] = "your-api-key"
|
||||
|
||||
# Generate high-quality landscape image
|
||||
response = image_generation(
|
||||
model="openrouter/google/gemini-2.5-flash-image",
|
||||
prompt="A serene mountain landscape with a lake",
|
||||
size="1536x1024", # Landscape format
|
||||
quality="high", # High quality (4K)
|
||||
)
|
||||
|
||||
# Access the generated image
|
||||
image_data = response.data[0]
|
||||
if image_data.b64_json:
|
||||
# Base64 encoded image
|
||||
print(f"Generated base64 image: {image_data.b64_json[:50]}...")
|
||||
elif image_data.url:
|
||||
# Image URL
|
||||
print(f"Generated image URL: {image_data.url}")
|
||||
```
|
||||
|
||||
### Using OpenRouter-Specific Parameters
|
||||
|
||||
You can also pass OpenRouter-specific parameters directly using `image_config`:
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ["OPENROUTER_API_KEY"] = "your-api-key"
|
||||
|
||||
response = image_generation(
|
||||
model="openrouter/google/gemini-2.5-flash-image",
|
||||
prompt="A futuristic cityscape at night",
|
||||
image_config={
|
||||
"aspect_ratio": "16:9", # OpenRouter native format
|
||||
"image_size": "4K" # OpenRouter native format
|
||||
}
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Response Format
|
||||
|
||||
The response follows the standard LiteLLM ImageResponse format:
|
||||
|
||||
```python
|
||||
{
|
||||
"created": 1703658209,
|
||||
"data": [{
|
||||
"b64_json": "iVBORw0KGgoAAAANSUhEUgAA...", # Base64 encoded image
|
||||
"url": None,
|
||||
"revised_prompt": None
|
||||
}],
|
||||
"usage": {
|
||||
"input_tokens": 10,
|
||||
"output_tokens": 1290,
|
||||
"total_tokens": 1300
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Cost Tracking
|
||||
|
||||
OpenRouter provides cost information in the response, which LiteLLM automatically tracks:
|
||||
|
||||
```python
|
||||
response = image_generation(
|
||||
model="openrouter/google/gemini-2.5-flash-image",
|
||||
prompt="A cute baby sea otter",
|
||||
)
|
||||
|
||||
# Cost is available in the response metadata
|
||||
print(f"Request cost: ${response._hidden_params['additional_headers']['llm_provider-x-litellm-response-cost']}")
|
||||
```
|
||||
|
|
|
|||
|
|
@ -12,100 +12,340 @@ LiteLLM supports SAP Generative AI Hub's Orchestration Service.
|
|||
| Supported Endpoints | `/chat/completions`, `/embeddings` |
|
||||
| API Reference | [SAP AI Core Documentation](https://help.sap.com/docs/sap-ai-core) |
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Before you begin, ensure you have:
|
||||
|
||||
1. **SAP BTP Account** with access to SAP AI Core
|
||||
2. **AI Core Service Instance** provisioned in your subaccount
|
||||
3. **Service Key** created for your AI Core instance (this contains your credentials)
|
||||
4. **Resource Group** with deployed AI models (check with your SAP administrator)
|
||||
|
||||
:::tip Where to Find Your Credentials
|
||||
Your credentials come from the **Service Key** you create in SAP BTP Cockpit:
|
||||
|
||||
1. Navigate to your **Subaccount** → **Instances and Subscriptions**
|
||||
2. Find your **AI Core** instance and click on it
|
||||
3. Go to **Service Keys** and create one (or use existing)
|
||||
4. The JSON contains all values needed below
|
||||
|
||||
The service key JSON looks like this:
|
||||
|
||||
```json
|
||||
{
|
||||
"clientid": "sb-abc123...",
|
||||
"clientsecret": "xyz789...",
|
||||
"url": "https://myinstance.authentication.eu10.hana.ondemand.com",
|
||||
"serviceurls": {
|
||||
"AI_API_URL": "https://api.ai.prod.eu-central-1.aws.ml.hana.ondemand.com"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
:::info Resource Group
|
||||
The resource group is typically configured separately in your AI Core deployment, not in the service key itself. You can set it via the `AICORE_RESOURCE_GROUP` environment variable (defaults to "default").
|
||||
:::
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Step 1: Install LiteLLM
|
||||
|
||||
```bash
|
||||
pip install litellm
|
||||
```
|
||||
|
||||
### Step 2: Set Your Credentials
|
||||
|
||||
Choose **one** of these authentication methods:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="service-key" label="Service Key JSON (Recommended)">
|
||||
|
||||
The simplest approach - paste your entire service key as a single environment variable. The service key must be wrapped in a `credentials` object:
|
||||
|
||||
```bash
|
||||
export AICORE_SERVICE_KEY='{
|
||||
"credentials": {
|
||||
"clientid": "your-client-id",
|
||||
"clientsecret": "your-client-secret",
|
||||
"url": "https://<your-instance>.authentication.sap.hana.ondemand.com",
|
||||
"serviceurls": {
|
||||
"AI_API_URL": "https://api.ai.<your-region>.aws.ml.hana.ondemand.com"
|
||||
}
|
||||
}
|
||||
}'
|
||||
export AICORE_RESOURCE_GROUP="default"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="individual" label="Individual Variables">
|
||||
|
||||
Alternatively, instead of using the service key above, you could set each credential separately:
|
||||
|
||||
```bash
|
||||
export AICORE_AUTH_URL="https://<your-instance>.authentication.sap.hana.ondemand.com/oauth/token"
|
||||
export AICORE_CLIENT_ID="your-client-id"
|
||||
export AICORE_CLIENT_SECRET="your-client-secret"
|
||||
export AICORE_RESOURCE_GROUP="default"
|
||||
export AICORE_BASE_URL="https://api.ai.<your-region>.aws.ml.hana.ondemand.com/v2"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Step 3: Make Your First Request
|
||||
|
||||
```python title="test_sap.py"
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello from LiteLLM!"}]
|
||||
)
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
Run it:
|
||||
|
||||
```bash
|
||||
python test_sap.py
|
||||
```
|
||||
|
||||
**Expected output:**
|
||||
|
||||
```text
|
||||
Hello! How can I assist you today?
|
||||
```
|
||||
|
||||
### Step 4: Verify Your Setup (Optional)
|
||||
|
||||
Test that everything is working with this diagnostic script:
|
||||
|
||||
```python title="verify_sap_setup.py"
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Enable debug logging to see what's happening
|
||||
import os
|
||||
os.environ["LITELLM_LOG"] = "DEBUG"
|
||||
|
||||
# Either use AICORE_SERVICE_KEY (contains all credentials including resourcegroup)
|
||||
# OR use individual variables (all required together)
|
||||
individual_vars = ["AICORE_AUTH_URL", "AICORE_CLIENT_ID", "AICORE_CLIENT_SECRET", "AICORE_BASE_URL", "AICORE_RESOURCE_GROUP"]
|
||||
|
||||
print("=== SAP Gen AI Hub Setup Verification ===\n")
|
||||
|
||||
# Check for service key method
|
||||
if os.environ.get("AICORE_SERVICE_KEY"):
|
||||
print("✓ Using AICORE_SERVICE_KEY authentication (includes resource group)")
|
||||
else:
|
||||
# Check individual variables
|
||||
missing = [v for v in individual_vars if not os.environ.get(v)]
|
||||
if missing:
|
||||
print(f"✗ Missing environment variables: {missing}")
|
||||
else:
|
||||
print("✓ Using individual variable authentication")
|
||||
print(f"✓ Resource group: {os.environ.get('AICORE_RESOURCE_GROUP')}")
|
||||
|
||||
# Test API connection
|
||||
print("\n=== Testing API Connection ===\n")
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[{"role": "user", "content": "Say 'Connection successful!' and nothing else."}],
|
||||
max_tokens=20
|
||||
)
|
||||
print(f"✓ API Response: {response.choices[0].message.content}")
|
||||
print("\n🎉 Setup complete! You're ready to use SAP Gen AI Hub with LiteLLM.")
|
||||
except Exception as e:
|
||||
print(f"✗ API Error: {e}")
|
||||
print("\nTroubleshooting tips:")
|
||||
print(" 1. Verify your service key credentials are correct")
|
||||
print(" 2. Check that 'gpt-4o' is deployed in your resource group")
|
||||
print(" 3. Ensure your SAP AI Core instance is running")
|
||||
```
|
||||
|
||||
Run the verification:
|
||||
|
||||
```bash
|
||||
python verify_sap_setup.py
|
||||
```
|
||||
|
||||
**Expected output on success:**
|
||||
|
||||
```text
|
||||
=== SAP Gen AI Hub Setup Verification ===
|
||||
|
||||
✓ Using AICORE_SERVICE_KEY authentication
|
||||
✓ Resource group: default
|
||||
|
||||
=== Testing API Connection ===
|
||||
|
||||
✓ API Response: Connection successful!
|
||||
|
||||
🎉 Setup complete! You're ready to use SAP Gen AI Hub with LiteLLM.
|
||||
```
|
||||
|
||||
## Authentication
|
||||
|
||||
SAP Generative AI Hub uses service key authentication. You can provide credentials via:
|
||||
SAP Generative AI Hub uses OAuth2 service keys for authentication. See [Quick Start](#quick-start) for setup instructions.
|
||||
|
||||
1. **Environment variable** - Set `AICORE_SERVICE_KEY` with your service key JSON
|
||||
2. **Direct parameter** - Pass `api_key` with the service key JSON string
|
||||
### Environment Variables Reference
|
||||
|
||||
```python showLineNumbers title="Environment Variable"
|
||||
import os
|
||||
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
|
||||
| Variable | Required | Description |
|
||||
|----------|----------|-------------|
|
||||
| `AICORE_SERVICE_KEY` | Yes* | Complete service key JSON (recommended method) |
|
||||
| `AICORE_RESOURCE_GROUP` | Yes | Your AI Core resource group name |
|
||||
| `AICORE_AUTH_URL` | Yes* | OAuth token URL (alternative to service key) |
|
||||
| `AICORE_CLIENT_ID` | Yes* | OAuth client ID (alternative to service key) |
|
||||
| `AICORE_CLIENT_SECRET` | Yes* | OAuth client secret (alternative to service key) |
|
||||
| `AICORE_BASE_URL` | Yes* | AI Core API base URL (alternative to service key) |
|
||||
|
||||
*Choose either `AICORE_SERVICE_KEY` OR the individual variables (`AICORE_AUTH_URL`, `AICORE_CLIENT_ID`, `AICORE_CLIENT_SECRET`, `AICORE_BASE_URL`).
|
||||
|
||||
## Model Naming Conventions
|
||||
|
||||
Understanding model naming is crucial for using SAP Gen AI Hub correctly. The naming pattern differs depending on whether you're using the SDK directly or through the proxy.
|
||||
|
||||
### Direct SDK Usage
|
||||
|
||||
When calling LiteLLM's SDK directly, you **must** include the `sap/` prefix in the model name:
|
||||
|
||||
```python
|
||||
# Correct - includes sap/ prefix
|
||||
model="sap/gpt-4o"
|
||||
model="sap/anthropic--claude-4.5-sonnet"
|
||||
model="sap/gemini-2.5-pro"
|
||||
|
||||
# Incorrect - missing prefix
|
||||
model="gpt-4o" # ❌ Won't work
|
||||
```
|
||||
3. **Environment variables** - Set the following list of credentials in .env file
|
||||
<pre>
|
||||
AICORE_AUTH_URL = "https://* * * .authentication.sap.hana.ondemand.com/oauth/token",
|
||||
AICORE_CLIENT_ID = " *** ",
|
||||
AICORE_CLIENT_SECRET = " *** ",
|
||||
AICORE_RESOURCE_GROUP = " *** ",
|
||||
AICORE_BASE_URL = "https://api.ai.***.cfapps.sap.hana.ondemand.com/v2"
|
||||
</pre>
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers title="SAP Chat Completion"
|
||||
from litellm import completion
|
||||
import os
|
||||
### Proxy Usage
|
||||
|
||||
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
|
||||
When using the LiteLLM Proxy, you use the **friendly `model_name`** defined in your configuration. The proxy automatically handles the `sap/` prefix routing.
|
||||
|
||||
response = completion(
|
||||
model="sap/gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello from LiteLLM"}]
|
||||
```yaml
|
||||
# In config.yaml, define the mapping
|
||||
model_list:
|
||||
- model_name: gpt-4o # ← Use this name in client requests
|
||||
litellm_params:
|
||||
model: sap/gpt-4o # ← Proxy handles the sap/ prefix
|
||||
```
|
||||
|
||||
```python
|
||||
# Client request - no sap/ prefix needed
|
||||
client.chat.completions.create(
|
||||
model="gpt-4o", # ✓ Correct for proxy usage
|
||||
messages=[...]
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="SAP Chat Completion - Streaming"
|
||||
### Anthropic Models Special Syntax
|
||||
|
||||
Anthropic models use a double-dash (`--`) prefix convention:
|
||||
|
||||
| Provider | Model Example | LiteLLM Format |
|
||||
|----------|---------------|----------------|
|
||||
| OpenAI | GPT-4o | `sap/gpt-4o` |
|
||||
| Anthropic | Claude 4.5 Sonnet | `sap/anthropic--claude-4.5-sonnet` |
|
||||
| Google | Gemini 2.5 Pro | `sap/gemini-2.5-pro` |
|
||||
| Mistral | Mistral Large | `sap/mistral-large` |
|
||||
|
||||
### Quick Reference Table
|
||||
|
||||
| Usage Type | Model Format | Example |
|
||||
|------------|--------------|---------|
|
||||
| Direct SDK | `sap/<model-name>` | `sap/gpt-4o` |
|
||||
| Direct SDK (Anthropic) | `sap/anthropic--<model>` | `sap/anthropic--claude-4.5-sonnet` |
|
||||
| Proxy Client | `<friendly-name>` | `gpt-4o` or `claude-sonnet` |
|
||||
|
||||
## Using the Python SDK
|
||||
|
||||
The LiteLLM Python SDK automatically detects your authentication method. Simply set your environment variables and make requests.
|
||||
|
||||
```python showLineNumbers title="Basic Completion"
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
|
||||
|
||||
# Assumes AICORE_AUTH_URL, AICORE_CLIENT_ID, etc. are set
|
||||
response = completion(
|
||||
model="sap/gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello from LiteLLM"}],
|
||||
stream=True
|
||||
model="sap/anthropic--claude-4.5-sonnet",
|
||||
messages=[{"role": "user", "content": "Explain quantum computing"}]
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk.choices[0].delta.content or "", end="")
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="SAP Embedding"
|
||||
from litellm import embedding
|
||||
import os
|
||||
Both authentication methods (individual variables or service key JSON) work automatically - no code changes required.
|
||||
|
||||
os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}'
|
||||
## Using the Proxy Server
|
||||
|
||||
result = embedding(
|
||||
model="sap/text-embedding-3-small",
|
||||
input="Answer to the ultimate question of life, the universe, and everything is 42")
|
||||
print(result.data[0])
|
||||
```
|
||||
The LiteLLM Proxy provides a unified OpenAI-compatible API for your SAP models.
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
### Configuration
|
||||
|
||||
Add to your LiteLLM Proxy config:
|
||||
Create a `config.yaml` file in your project directory with your model mappings and credentials:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: "sap/*"
|
||||
# OpenAI models
|
||||
- model_name: gpt-5
|
||||
litellm_params:
|
||||
model: "sap/*"
|
||||
model: sap/gpt-5
|
||||
|
||||
general_settings:
|
||||
master_key: your-proxy-api-key
|
||||
# Anthropic models (note the double-dash)
|
||||
- model_name: claude-sonnet
|
||||
litellm_params:
|
||||
model: sap/anthropic--claude-4.5-sonnet
|
||||
|
||||
- model_name: claude-opus
|
||||
litellm_params:
|
||||
model: sap/anthropic--claude-4.5-opus
|
||||
|
||||
# Embeddings
|
||||
- model_name: text-embedding-3-small
|
||||
litellm_params:
|
||||
model: sap/text-embedding-3-small
|
||||
|
||||
litellm_settings:
|
||||
drop_params: true
|
||||
set_verbose: false
|
||||
request_timeout: 600
|
||||
num_retries: 2
|
||||
forward_client_headers_to_llm_api: ["anthropic-version"]
|
||||
|
||||
general_settings:
|
||||
master_key: "sk-1234" # Enter here your desired master key starting with 'sk-'.
|
||||
|
||||
# UI Admin is not required but helpful including the management of keys for your team(s). If you are using a database, these parameters are required:
|
||||
database_url: "Enter you database URL."
|
||||
UI_USERNAME: "Your desired UI admin account name"
|
||||
UI_PASSWORD: "Your desired and strong pwd"
|
||||
|
||||
# Authentication
|
||||
environment_variables:
|
||||
AICORE_SERVICE_KEY: '{"clientid": "...", "clientsecret": "...", ...}'
|
||||
AICORE_SERVICE_KEY: '{"credentials": {"clientid": "...", "clientsecret": "...", "url": "...", "serviceurls": {"AI_API_URL": "..."}}}'
|
||||
AICORE_RESOURCE_GROUP: "default"
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
### Starting the Proxy
|
||||
|
||||
```bash showLineNumbers title="Start Proxy"
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
The proxy will start on `http://localhost:4000` by default.
|
||||
|
||||
### Making Requests
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Test Request"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "sap/gpt-4",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello"}]
|
||||
}'
|
||||
```
|
||||
|
|
@ -118,11 +358,11 @@ from openai import OpenAI
|
|||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-proxy-api-key"
|
||||
api_key="sk-1234"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="sap/gpt-4",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
print(response.choices[0].message.content)
|
||||
|
|
@ -134,12 +374,14 @@ print(response.choices[0].message.content)
|
|||
```python showLineNumbers title="LiteLLM SDK"
|
||||
import os
|
||||
import litellm
|
||||
os.environ["LITELLM_PROXY_API_KEY"] = "your-proxy-api-key"
|
||||
litellm.use_litellm_proxy = True # it is important to set this parameter
|
||||
|
||||
os.environ["LITELLM_PROXY_API_KEY"] = "sk-1234"
|
||||
litellm.use_litellm_proxy = True
|
||||
|
||||
response = litellm.completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}],
|
||||
api_base="http://your-proxy-api-base"
|
||||
model="claude-sonnet",
|
||||
messages=[{"content": "Hello, how are you?", "role": "user"}],
|
||||
api_base="http://localhost:4000"
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
|
@ -148,15 +390,170 @@ print(response)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported Parameters
|
||||
## Features
|
||||
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `temperature` | Controls randomness |
|
||||
| `max_tokens` | Maximum tokens in response |
|
||||
| `top_p` | Nucleus sampling |
|
||||
| `tools` | Function calling tools |
|
||||
| `tool_choice` | Tool selection behavior |
|
||||
| `response_format` | Output format (json_object, json_schema) |
|
||||
| `stream` | Enable streaming |
|
||||
### Streaming Responses
|
||||
|
||||
Stream responses in real-time for better user experience:
|
||||
|
||||
```python showLineNumbers title="Streaming Chat Completion"
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[{"role": "user", "content": "Count from 1 to 10"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content:
|
||||
print(chunk.choices[0].delta.content, end="", flush=True)
|
||||
```
|
||||
|
||||
### Structured Output
|
||||
|
||||
#### JSON Schema (Recommended)
|
||||
|
||||
Use JSON Schema for structured output with strict validation:
|
||||
|
||||
```python showLineNumbers title="JSON Schema Response"
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": "Generate info about Tokyo"
|
||||
}],
|
||||
response_format={
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "city_info",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {"type": "string"},
|
||||
"population": {"type": "number"},
|
||||
"country": {"type": "string"}
|
||||
},
|
||||
"required": ["name", "population", "country"],
|
||||
"additionalProperties": False
|
||||
},
|
||||
"strict": True
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
# Output: {"name":"Tokyo","population":37000000,"country":"Japan"}
|
||||
```
|
||||
|
||||
#### JSON Object Format
|
||||
|
||||
For flexible JSON output without schema validation:
|
||||
|
||||
```python showLineNumbers title="JSON Object Response"
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": "Generate a person object in JSON format with name and age"
|
||||
}],
|
||||
response_format={"type": "json_object"}
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
:::note SAP Platform Requirement
|
||||
When using `json_object` type, SAP's orchestration service requires the word "json" to appear in your prompt. This ensures explicit intent for JSON formatting. For schema-validated output without this requirement, use `json_schema` instead (recommended).
|
||||
:::
|
||||
|
||||
### Multi-turn Conversations
|
||||
|
||||
Maintain conversation context across multiple turns:
|
||||
|
||||
```python showLineNumbers title="Multi-turn Conversation"
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="sap/gpt-4o",
|
||||
messages=[
|
||||
{"role": "user", "content": "My name is Alice"},
|
||||
{"role": "assistant", "content": "Hello Alice! Nice to meet you."},
|
||||
{"role": "user", "content": "What is my name?"}
|
||||
]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
# Output: Your name is Alice.
|
||||
```
|
||||
|
||||
### Embeddings
|
||||
|
||||
Generate vector embeddings for semantic search and retrieval:
|
||||
|
||||
```python showLineNumbers title="Create Embeddings"
|
||||
from litellm import embedding
|
||||
|
||||
response = embedding(
|
||||
model="sap/text-embedding-3-small",
|
||||
input=["Hello world", "Machine learning is fascinating"]
|
||||
)
|
||||
|
||||
print(response.data[0]["embedding"]) # Vector representation
|
||||
```
|
||||
|
||||
## Reference
|
||||
|
||||
### Supported Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `model` | string | Model identifier (with `sap/` prefix for SDK) |
|
||||
| `messages` | array | Conversation messages |
|
||||
| `temperature` | float | Controls randomness (0-2) |
|
||||
| `max_tokens` | integer | Maximum tokens in response |
|
||||
| `top_p` | float | Nucleus sampling threshold |
|
||||
| `stream` | boolean | Enable streaming responses |
|
||||
| `response_format` | object | Output format (`json_object`, `json_schema`) |
|
||||
| `tools` | array | Function calling tool definitions |
|
||||
| `tool_choice` | string/object | Tool selection behavior |
|
||||
|
||||
### Supported Models
|
||||
|
||||
For the complete and up-to-date list of available models provided by SAP Gen AI Hub, please refer to the [SAP AI Core Generative AI Hub documentation](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/models-and-scenarios-in-generative-ai-hub).
|
||||
|
||||
:::info Model Availability
|
||||
Model availability varies by SAP deployment region and your subscription. Contact your SAP administrator to confirm which models are available in your environment.
|
||||
:::
|
||||
|
||||
### Troubleshooting
|
||||
|
||||
**Authentication Errors**
|
||||
|
||||
If you receive authentication errors:
|
||||
|
||||
1. Verify all required environment variables are set correctly
|
||||
2. Check that your service key hasn't expired
|
||||
3. Confirm your resource group has access to the desired models
|
||||
4. Ensure the `AICORE_AUTH_URL` and `AICORE_BASE_URL` match your SAP region
|
||||
|
||||
**Model Not Found**
|
||||
|
||||
If a model returns "not found":
|
||||
|
||||
1. Verify the model is available in your SAP deployment
|
||||
2. Check you're using the correct model name format (`sap/` prefix for SDK)
|
||||
3. Confirm your resource group has access to that specific model
|
||||
4. For Anthropic models, ensure you're using the `anthropic--` double-dash prefix
|
||||
|
||||
**Rate Limiting**
|
||||
|
||||
SAP Gen AI Hub enforces rate limits based on your subscription. If you hit limits:
|
||||
|
||||
1. Implement exponential backoff retry logic
|
||||
2. Consider using the proxy's built-in rate limiting features
|
||||
3. Contact your SAP administrator to review quota allocations
|
||||
|
|
|
|||
|
|
@ -744,6 +744,7 @@ router_settings:
|
|||
| LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging
|
||||
| LITELM_ENVIRONMENT | Environment for LiteLLM Instance. This is currently only logged to DeepEval to determine the environment for DeepEval integration.
|
||||
| LOGFIRE_TOKEN | Token for Logfire logging service
|
||||
| LOGFIRE_BASE_URL | Base URL for Logfire logging service (useful for self hosted deployments)
|
||||
| LOGGING_WORKER_CONCURRENCY | Maximum number of concurrent coroutine slots for the logging worker on the asyncio event loop. Default is 100. Setting too high will flood the event loop with logging tasks which will lower the overall latency of the requests.
|
||||
| LOGGING_WORKER_MAX_QUEUE_SIZE | Maximum size of the logging worker queue. When the queue is full, the worker aggressively clears tasks to make room instead of dropping logs. Default is 50,000
|
||||
| LOGGING_WORKER_MAX_TIME_PER_COROUTINE | Maximum time in seconds allowed for each coroutine in the logging worker before timing out. Default is 20.0
|
||||
|
|
|
|||
|
|
@ -9,7 +9,6 @@ LiteLLM provides flexible cost tracking and pricing customization for all LLM pr
|
|||
- **Custom Pricing** - Override default model costs or set pricing for custom models
|
||||
- **Cost Per Token** - Track costs based on input/output tokens (most common)
|
||||
- **Cost Per Second** - Track costs based on runtime (e.g., Sagemaker)
|
||||
- **Zero-Cost Models** - Bypass budget checks for free/on-premises models by setting costs to 0
|
||||
- **[Provider Discounts](./provider_discounts.md)** - Apply percentage-based discounts to specific providers
|
||||
- **[Provider Margins](./provider_margins.md)** - Add fees/margins to LLM costs for internal billing
|
||||
- **Base Model Mapping** - Ensure accurate cost tracking for Azure deployments
|
||||
|
|
@ -107,51 +106,6 @@ There are other keys you can use to specify costs for different scenarios and mo
|
|||
|
||||
These keys evolve based on how new models handle multimodality. The latest version can be found at [https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json).
|
||||
|
||||
## Zero-Cost Models (Bypass Budget Checks)
|
||||
|
||||
**Use Case**: You have on-premises or free models that should be accessible even when users exceed their budget limits.
|
||||
|
||||
**Solution** ✅: Set both `input_cost_per_token` and `output_cost_per_token` to `0` (explicitly) to bypass all budget checks for that model.
|
||||
|
||||
:::info
|
||||
|
||||
When a model is configured with zero cost, LiteLLM will automatically skip ALL budget checks (user, team, team member, end-user, organization, and global proxy budget) for requests to that model.
|
||||
|
||||
**Important**: Both costs must be **explicitly set to 0**. If costs are `null` or undefined, the model will be treated as having cost and budget checks will apply.
|
||||
|
||||
:::
|
||||
|
||||
### Configuration Example
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# On-premises model - free to use
|
||||
- model_name: on-prem-llama
|
||||
litellm_params:
|
||||
model: ollama/llama3
|
||||
api_base: http://localhost:11434
|
||||
model_info:
|
||||
input_cost_per_token: 0 # 👈 Explicitly set to 0
|
||||
output_cost_per_token: 0 # 👈 Explicitly set to 0
|
||||
|
||||
# Paid cloud model - budget checks apply
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
# No model_info - uses default pricing from cost map
|
||||
```
|
||||
|
||||
### Behavior
|
||||
|
||||
With the above configuration:
|
||||
|
||||
- **User over budget** → Can still use `on-prem-llama` ✅, but blocked from `gpt-4` ❌
|
||||
- **Team over budget** → Can still use `on-prem-llama` ✅, but blocked from `gpt-4` ❌
|
||||
- **End-user over budget** → Can still use `on-prem-llama` ✅, but blocked from `gpt-4` ❌
|
||||
|
||||
This ensures your free/on-premises models remain accessible regardless of budget constraints, while paid models are still properly governed.
|
||||
|
||||
## Set 'base_model' for Cost Tracking (e.g. Azure deployments)
|
||||
|
||||
**Problem**: Azure returns `gpt-4` in the response when `azure/gpt-4-1106-preview` is used. This leads to inaccurate cost tracking
|
||||
|
|
|
|||
|
|
@ -22,19 +22,22 @@ Customer Usage enables you to track spend and usage for individual customers (en
|
|||
|
||||
## How to Track Spend
|
||||
|
||||
Track customer spend by including a `user` field in your API requests. The customer ID will be automatically tracked and associated with all spend from that request.
|
||||
Track customer spend by including a `user` field in your API requests or by passing a customer ID header. The customer ID will be automatically tracked and associated with all spend from that request.
|
||||
|
||||
### Example using cURL
|
||||
<Tabs>
|
||||
<TabItem value="body" label="Request Body" default>
|
||||
|
||||
### Using Request Body
|
||||
|
||||
Make a `/chat/completions` call with the `user` field containing your customer ID:
|
||||
|
||||
```bash showLineNumbers title="Track spend with customer ID"
|
||||
```bash showLineNumbers title="Track spend with customer ID in body"
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \ # 👈 YOUR PROXY KEY
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"user": "customer-123", # 👈 CUSTOMER ID
|
||||
"user": "customer-123",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -44,7 +47,49 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
}'
|
||||
```
|
||||
|
||||
The customer ID (`customer-123`) will be automatically upserted into the database with the new spend. If the customer ID already exists, spend will be incremented.
|
||||
</TabItem>
|
||||
<TabItem value="header" label="Request Header">
|
||||
|
||||
### Using Request Headers
|
||||
|
||||
You can also pass the customer ID via HTTP headers. This is useful for tools that support custom headers but don't allow modifying the request body (like Claude Code with `ANTHROPIC_CUSTOM_HEADERS`).
|
||||
|
||||
LiteLLM automatically recognizes these standard headers (no configuration required):
|
||||
- `x-litellm-customer-id`
|
||||
- `x-litellm-end-user-id`
|
||||
|
||||
```bash showLineNumbers title="Track spend with customer ID in header"
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'x-litellm-customer-id: customer-123' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What is the capital of France?"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
#### Using with Claude Code
|
||||
|
||||
Claude Code supports custom headers via the `ANTHROPIC_CUSTOM_HEADERS` environment variable. Set it to pass your customer ID:
|
||||
|
||||
```bash title="Configure Claude Code with customer tracking"
|
||||
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000/v1/messages"
|
||||
export ANTHROPIC_API_KEY="sk-1234"
|
||||
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-customer-id: my-customer-id"
|
||||
```
|
||||
|
||||
Now all requests from Claude Code will automatically track spend under `my-customer-id`.
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
The customer ID will be automatically upserted into the database with the new spend. If the customer ID already exists, spend will be incremented.
|
||||
|
||||
### Example using OpenWebUI
|
||||
|
||||
|
|
|
|||
|
|
@ -1827,6 +1827,64 @@ This approach allows you to:
|
|||
- Share callbacks across different environments
|
||||
- Version control callback files in cloud storage
|
||||
|
||||
#### Step 2c - Mounting Custom Callbacks in Helm/Kubernetes (Alternative)
|
||||
|
||||
When deploying with Helm or Kubernetes, you can mount custom callback Python files alongside your `config.yaml` using `subPath` to avoid overwriting the config directory.
|
||||
|
||||
**The Problem:**
|
||||
Mounting a volume to a directory (e.g., `/app/`) would normally hide all existing files in that directory, including your `config.yaml`.
|
||||
|
||||
**The Solution:**
|
||||
Use `subPath` in your `volumeMounts` to mount individual files without overwriting the entire directory.
|
||||
|
||||
**Example - Helm values.yaml:**
|
||||
|
||||
```yaml
|
||||
# values.yaml
|
||||
volumes:
|
||||
- name: callback-files
|
||||
configMap:
|
||||
name: litellm-callback-files
|
||||
|
||||
volumeMounts:
|
||||
- name: callback-files
|
||||
mountPath: /app/custom_callbacks.py # Mount to specific FILE path
|
||||
subPath: custom_callbacks.py # Required to avoid overwriting directory
|
||||
```
|
||||
|
||||
**Create the ConfigMap with your callback file:**
|
||||
|
||||
```yaml
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: litellm-callback-files
|
||||
data:
|
||||
custom_callbacks.py: |
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
|
||||
class MyCustomHandler(CustomLogger):
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
print(f"Success! Model: {kwargs.get('model')}")
|
||||
|
||||
proxy_handler_instance = MyCustomHandler()
|
||||
```
|
||||
|
||||
**Reference in your config.yaml:**
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
callbacks: custom_callbacks.proxy_handler_instance
|
||||
```
|
||||
|
||||
**How it works:**
|
||||
1. The `subPath` parameter tells Kubernetes to mount only the specific file
|
||||
2. This places `custom_callbacks.py` in `/app/` alongside your existing `config.yaml`
|
||||
3. LiteLLM automatically finds the callback file in the same directory as the config
|
||||
4. No files are overwritten or hidden
|
||||
|
||||
**Note:** You can mount multiple callback files by adding more `volumeMounts` entries, each with its own `subPath`.
|
||||
|
||||
#### Step 3 - Start proxy + test request
|
||||
|
||||
```shell
|
||||
|
|
|
|||
|
|
@ -30,6 +30,9 @@ general_settings:
|
|||
# Optional: set how frequently cleanup should run - default is daily
|
||||
maximum_spend_logs_retention_interval: "1d" # Run cleanup daily
|
||||
|
||||
# Optional: set exact time for cleanup (Cron syntax)
|
||||
maximum_spend_logs_cleanup_cron: "0 4 * * *" # Run at 04:00 AM daily
|
||||
|
||||
litellm_settings:
|
||||
cache: true
|
||||
cache_params:
|
||||
|
|
@ -51,6 +54,15 @@ How long logs should be kept before deletion. Supported formats:
|
|||
|
||||
How often the cleanup job should run. Uses the same format as above. If not set, cleanup will run every 24 hours if and only if `maximum_spend_logs_retention_period` is set.
|
||||
|
||||
#### `maximum_spend_logs_cleanup_cron` (optional)
|
||||
|
||||
Schedule the cleanup using standard cron syntax. This takes precedence over `maximum_spend_logs_retention_interval`.
|
||||
|
||||
Examples:
|
||||
- `"0 4 * * *"` – Run at 04:00 AM daily
|
||||
- `"0 0 * * 0"` – Run at midnight every Sunday
|
||||
- `"*/30 * * * *"` – Run every 30 minutes
|
||||
|
||||
## How it works
|
||||
|
||||
### Step 1. Lock Acquisition (Optional with Redis)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,99 @@
|
|||
# Claude Code - Granular Cost Tracking
|
||||
|
||||
Track Claude Code usage by customer or tags using LiteLLM proxy. This enables granular cost attribution for billing, budgeting, and analytics.
|
||||
|
||||
## How It Works
|
||||
|
||||
Claude Code supports custom headers via `ANTHROPIC_CUSTOM_HEADERS`. LiteLLM automatically tracks requests with specific headers for cost attribution.
|
||||
|
||||
## Tracking Options
|
||||
|
||||
Choose how you want to attribute costs:
|
||||
|
||||
| Track By | Header | Use Case |
|
||||
|----------|--------|----------|
|
||||
| Customer | `x-litellm-customer-id` | Bill customers, per-user budgets |
|
||||
| Tags | `x-litellm-tags` | Project tracking, cost centers, environments |
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description | Example |
|
||||
|----------|-------------|---------|
|
||||
| `ANTHROPIC_BASE_URL` | LiteLLM proxy URL | `http://localhost:4000` |
|
||||
| `ANTHROPIC_API_KEY` | LiteLLM API key | `sk-1234` |
|
||||
| `ANTHROPIC_CUSTOM_HEADERS` | Custom headers (`header-name: value` format) | See examples below |
|
||||
|
||||
## Option 1: Track by Customer
|
||||
|
||||
Use this to attribute costs to specific customers or end-users.
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_BASE_URL=http://localhost:4000
|
||||
export ANTHROPIC_API_KEY=sk-1234
|
||||
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-customer-id: claude-ishaan-local"
|
||||
```
|
||||
|
||||
## Option 2: Track by Tags
|
||||
|
||||
Use this to attribute costs to projects, cost centers, or environments. Pass comma-separated tags.
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_BASE_URL=http://localhost:4000
|
||||
export ANTHROPIC_API_KEY=sk-1234
|
||||
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-tags: project:acme,env:prod,team:backend"
|
||||
```
|
||||
|
||||
|
||||
## Quick Start
|
||||
|
||||
### 1. Set Environment Variables
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_BASE_URL=http://localhost:4000
|
||||
export ANTHROPIC_API_KEY=sk-1234
|
||||
export ANTHROPIC_CUSTOM_HEADERS="x-litellm-customer-id: claude-ishaan-local"
|
||||
```
|
||||
|
||||
### 2. Use Claude Code
|
||||
|
||||
```bash
|
||||
claude
|
||||
```
|
||||
|
||||
All requests will now be tracked under the customer ID `claude-ishaan-local`.
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 3. View Usage in LiteLLM UI
|
||||
|
||||
Navigate to the **Logs** tab in the LiteLLM UI.
|
||||
|
||||

|
||||
|
||||
Click on a request to see details.
|
||||
|
||||

|
||||
|
||||
Filter by customer ID to see all requests for that customer.
|
||||
|
||||

|
||||
|
||||
## Supported Headers
|
||||
|
||||
| Header | Description |
|
||||
|--------|-------------|
|
||||
| `x-litellm-customer-id` | Track by customer/end-user ID |
|
||||
| `x-litellm-end-user-id` | Alternative customer ID header |
|
||||
| `x-litellm-tags` | Comma-separated tags for cost attribution |
|
||||
|
||||
## Related
|
||||
|
||||
- [Claude Code Quickstart](./claude_responses_api.md)
|
||||
- [Customer Budgets](../proxy/customers.md)
|
||||
- [Tag Budgets](../proxy/tag_budgets.md)
|
||||
- [Track Usage for Coding Tools](./cost_tracking_coding.md)
|
||||
|
||||
93
docs/my-website/docs/tutorials/claude_mcp.md
Normal file
93
docs/my-website/docs/tutorials/claude_mcp.md
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Use Claude Code with MCPs
|
||||
|
||||
This tutorial shows how to connect MCP servers to Claude Code via LiteLLM Proxy.
|
||||
|
||||
Note: LiteLLM supports OAuth for MCP servers as well. [Learn more](https://docs.litellm.ai/docs/mcp#mcp-oauth)
|
||||
|
||||
## Connecting MCP Servers
|
||||
|
||||
You can also connect MCP servers to Claude Code via LiteLLM Proxy.
|
||||
|
||||
|
||||
1. Add the MCP server to your `config.yaml`
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="github" label="GitHub MCP">
|
||||
|
||||
In this example, we'll add the Github MCP server to our `config.yaml`
|
||||
|
||||
```yaml title="config.yaml" showLineNumbers
|
||||
mcp_servers:
|
||||
github_mcp:
|
||||
url: "https://api.githubcopilot.com/mcp"
|
||||
auth_type: oauth2
|
||||
client_id: os.environ/GITHUB_OAUTH_CLIENT_ID
|
||||
client_secret: os.environ/GITHUB_OAUTH_CLIENT_SECRET
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="atlassian" label="Atlassian MCP">
|
||||
|
||||
In this example, we'll add the Atlassian MCP server to our `config.yaml`
|
||||
|
||||
```yaml title="config.yaml" showLineNumbers
|
||||
atlassian_mcp:
|
||||
server_id: atlassian_mcp_id
|
||||
url: "https://mcp.atlassian.com/v1/sse"
|
||||
transport: "sse"
|
||||
auth_type: oauth2
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Use the MCP server in Claude Code
|
||||
|
||||
```bash
|
||||
claude mcp add --transport http litellm_proxy http://0.0.0.0:4000/github_mcp/mcp --header "Authorization: Bearer sk-LITELLM_VIRTUAL_KEY"
|
||||
```
|
||||
|
||||
For MCP servers that require dynamic client registration (such as Atlassian), please set `x-litellm-api-key: Bearer sk-LITELLM_VIRTUAL_KEY` instead of using `Authorization: Bearer LITELLM_VIRTUAL_KEY`.
|
||||
|
||||
4. Authenticate via Claude Code
|
||||
|
||||
a. Start Claude Code
|
||||
|
||||
```bash
|
||||
claude
|
||||
```
|
||||
|
||||
b. Authenticate via Claude Code
|
||||
|
||||
```bash
|
||||
/mcp
|
||||
```
|
||||
|
||||
c. Select the MCP server
|
||||
|
||||
```bash
|
||||
> litellm_proxy
|
||||
```
|
||||
|
||||
d. Start Oauth flow via Claude Code
|
||||
|
||||
```bash
|
||||
> 1. Authenticate
|
||||
2. Reconnect
|
||||
3. Disable
|
||||
```
|
||||
|
||||
e. Once completed, you should see this success message:
|
||||
|
||||
<img src={require('../../img/oauth_2_success.png').default} alt="OAuth 2.0 Success" style={{ width: '500px', height: 'auto' }} />
|
||||
316
docs/my-website/docs/tutorials/claude_non_anthropic_models.md
Normal file
316
docs/my-website/docs/tutorials/claude_non_anthropic_models.md
Normal file
|
|
@ -0,0 +1,316 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Use Claude Code with Non-Anthropic Models
|
||||
|
||||
This tutorial shows how to use Claude Code with non-Anthropic models like OpenAI, Gemini, and other LLM providers through LiteLLM proxy.
|
||||
|
||||
:::info
|
||||
|
||||
LiteLLM automatically translates between different provider formats, allowing you to use any supported LLM provider with Claude Code while maintaining the Anthropic Messages API format.
|
||||
|
||||
:::
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code/overview) installed
|
||||
- API keys for your chosen providers (OpenAI, Vertex AI, etc.)
|
||||
|
||||
## Installation
|
||||
|
||||
First, install LiteLLM with proxy support:
|
||||
|
||||
```bash
|
||||
pip install 'litellm[proxy]'
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
Create a configuration file with your preferred non-Anthropic models:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# OpenAI GPT-4o
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
# OpenAI GPT-4o-mini
|
||||
- model_name: gpt-4o-mini
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-mini
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
Set your environment variables:
|
||||
|
||||
```bash
|
||||
export OPENAI_API_KEY="your-openai-api-key"
|
||||
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="gemini" label="Google AI Studio">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Google Gemini
|
||||
- model_name: gemini-3.0-flash-exp
|
||||
litellm_params:
|
||||
model: gemini/gemini-3.0-flash-exp
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
Set your environment variables:
|
||||
|
||||
```bash
|
||||
export GEMINI_API_KEY="your-gemini-api-key"
|
||||
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="vertex_ai" label="Vertex AI">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Google Gemini
|
||||
- model_name: vertex-gemini-3-flash-preview
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-3-flash-preview
|
||||
vertex_credentials: os.environ/VERTEX_FILE_PATH_ENV_VAR # os.environ["VERTEX_FILE_PATH_ENV_VAR"] = "/path/to/service_account.json"
|
||||
vertex_project: "my-test-project"
|
||||
vertex_location: "us-east-1"
|
||||
|
||||
# Anthropic Claude
|
||||
- model_name: anthropic-vertex
|
||||
litellm_params:
|
||||
model: vertex_ai/claude-3-sonnet@20240229
|
||||
vertex_ai_project: "my-test-project"
|
||||
vertex_ai_location: "us-east-1"
|
||||
vertex_credentials: os.environ/VERTEX_FILE_PATH_ENV_VAR # os.environ["VERTEX_FILE_PATH_ENV_VAR"] = "/path/to/service_account.json"
|
||||
```
|
||||
|
||||
Set your environment variables:
|
||||
|
||||
```bash
|
||||
export VERTEX_FILE_PATH_ENV_VAR="/path/to/service_account.json"
|
||||
export LITELLM_MASTER_KEY="sk-1234567890"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="multi" label="Azure OpenAI">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Azure OpenAI
|
||||
- model_name: azure-gpt-4
|
||||
litellm_params:
|
||||
model: azure/gpt-4
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_version: "2024-02-01"
|
||||
```
|
||||
|
||||
Set your environment variables:
|
||||
|
||||
```bash
|
||||
export AZURE_API_KEY="your-azure-api-key"
|
||||
export AZURE_API_BASE="https://your-resource.openai.azure.com"
|
||||
export LITELLM_MASTER_KEY="sk-1234567890"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### 2. Start LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Verify Setup
|
||||
|
||||
Test that your proxy is working correctly:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-test" label="OpenAI">
|
||||
|
||||
```bash
|
||||
curl -X POST http://0.0.0.0:4000/v1/messages \
|
||||
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-4o",
|
||||
"max_tokens": 1000,
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="gemini-test" label="Google AI Studio">
|
||||
|
||||
```bash
|
||||
curl -X POST http://0.0.0.0:4000/v1/messages \
|
||||
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gemini-3.0-flash-exp",
|
||||
"max_tokens": 1000,
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="vertex-test" label="Vertex AI">
|
||||
|
||||
```bash
|
||||
curl -X POST http://0.0.0.0:4000/v1/messages \
|
||||
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gemini-3.0-flash-exp",
|
||||
"max_tokens": 1000,
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="azure-test" label="Azure OpenAI">
|
||||
|
||||
```bash
|
||||
curl -X POST http://0.0.0.0:4000/v1/messages \
|
||||
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "azure-gpt-4",
|
||||
"max_tokens": 1000,
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### 4. Configure Claude Code
|
||||
|
||||
Configure Claude Code to use your LiteLLM proxy:
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000"
|
||||
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
|
||||
```
|
||||
|
||||
:::tip
|
||||
The `LITELLM_MASTER_KEY` gives Claude Code access to all proxy models. You can also create virtual keys in the LiteLLM UI to limit access to specific models.
|
||||
:::
|
||||
|
||||
### 5. Use Claude Code with Non-Anthropic Models
|
||||
|
||||
Start Claude Code and specify which model to use:
|
||||
|
||||
```bash
|
||||
# Use OpenAI GPT-4o
|
||||
claude --model gpt-4o
|
||||
|
||||
# Use OpenAI GPT-4o-mini for faster responses
|
||||
claude --model gpt-4o-mini
|
||||
|
||||
# Use Google Gemini
|
||||
claude --model gemini-3.0-flash-exp
|
||||
|
||||
# Use Vertex AI Gemini
|
||||
claude --model vertex-gemini-3-flash-preview
|
||||
|
||||
# Use Vertex AI Anthropic Claude
|
||||
claude --model anthropic-vertex
|
||||
|
||||
# Use Azure OpenAI
|
||||
claude --model azure-gpt-4
|
||||
```
|
||||
|
||||
## How It Works
|
||||
|
||||
LiteLLM acts as a unified interface that:
|
||||
|
||||
1. **Receives requests** from Claude Code in Anthropic Messages API format
|
||||
2. **Translates** the request to the target provider's format (OpenAI, Gemini, etc.)
|
||||
3. **Forwards** the request to the actual provider
|
||||
4. **Translates** the response back to Anthropic Messages API format
|
||||
5. **Returns** the response to Claude Code
|
||||
|
||||
This allows you to use Claude Code's interface with any LLM provider supported by LiteLLM.
|
||||
|
||||
## Advanced Features
|
||||
|
||||
### Load Balancing and Fallbacks
|
||||
|
||||
Configure multiple deployments with automatic fallback:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-4o # virtual model name
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: gpt-4o # same virtual name
|
||||
litellm_params:
|
||||
model: azure/gpt-4o
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
||||
router_settings:
|
||||
routing_strategy: simple-shuffle # Load balance between deployments
|
||||
num_retries: 2
|
||||
timeout: 30
|
||||
```
|
||||
|
||||
### Usage Tracking and Budgets
|
||||
|
||||
Track usage and set budgets through the LiteLLM UI:
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
database_url: "postgresql://..." # Enable database for tracking
|
||||
|
||||
general_settings:
|
||||
store_model_in_db: true
|
||||
```
|
||||
|
||||
Start the proxy with the UI:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
Access the UI at `http://0.0.0.0:4000/ui` to:
|
||||
- View usage analytics
|
||||
- Set budget limits per user/key
|
||||
- Monitor costs across different providers
|
||||
- Create virtual keys with specific permissions
|
||||
|
||||
|
||||
## Supported Providers
|
||||
|
||||
LiteLLM supports 100+ providers. Here are some popular ones for use with Claude Code:
|
||||
|
||||
- **OpenAI**: GPT-4o, GPT-4o-mini, o1, o3-mini
|
||||
- **Google**: Gemini 2.0 Flash, Gemini 1.5 Pro/Flash
|
||||
- **Azure OpenAI**: All OpenAI models via Azure
|
||||
- **AWS Bedrock**: Llama, Mistral, and other models
|
||||
- **Vertex AI**: Gemini, Claude, and other models on Google Cloud
|
||||
- **Groq**: Fast inference for Llama and Mixtral
|
||||
- **Together AI**: Llama, Mixtral, and other open source models
|
||||
- **Deepseek**: Deepseek-chat, Deepseek-coder
|
||||
|
||||
[View full list of supported providers →](https://docs.litellm.ai/docs/providers)
|
||||
|
|
@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Claude Code
|
||||
# Claude Code Quickstart
|
||||
|
||||
This tutorial shows how to call Claude models through LiteLLM proxy from Claude Code.
|
||||
|
||||
|
|
@ -142,7 +142,7 @@ Common issues and solutions:
|
|||
- Ensure the model name in Claude Code matches exactly with your `config.yaml`
|
||||
- Check LiteLLM logs for detailed error messages
|
||||
|
||||
## Using Multiple Models
|
||||
## Using Bedrock/Vertex AI/Azure Foundry Models
|
||||
|
||||
Expand your configuration to support multiple providers and models:
|
||||
|
||||
|
|
@ -151,25 +151,6 @@ Expand your configuration to support multiple providers and models:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
# OpenAI models
|
||||
- model_name: codex-mini
|
||||
litellm_params:
|
||||
model: openai/codex-mini
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: https://api.openai.com/v1
|
||||
|
||||
- model_name: o3-pro
|
||||
litellm_params:
|
||||
model: openai/o3-pro
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: https://api.openai.com/v1
|
||||
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: https://api.openai.com/v1
|
||||
|
||||
# Anthropic models
|
||||
- model_name: claude-3-5-sonnet-20241022
|
||||
litellm_params:
|
||||
|
|
@ -189,6 +170,24 @@ model_list:
|
|||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-east-1
|
||||
|
||||
# Azure Foundry
|
||||
- model_name: claude-4-azure
|
||||
litellm_params:
|
||||
model: azure_ai/claude-opus-4-1
|
||||
api_key: os.environ/AZURE_AI_API_KEY
|
||||
api_base: os.environ/AZURE_AI_API_BASE # https://my-resource.services.ai.azure.com/anthropic
|
||||
|
||||
# Google Vertex AI
|
||||
- model_name: anthropic-vertex
|
||||
litellm_params:
|
||||
model: vertex_ai/claude-haiku-4-5@20251001
|
||||
vertex_ai_project: "my-test-project"
|
||||
vertex_ai_location: "us-east-1"
|
||||
vertex_credentials: os.environ/VERTEX_FILE_PATH_ENV_VAR # os.environ["VERTEX_FILE_PATH_ENV_VAR"] = "/path/to/service_account.json"
|
||||
|
||||
|
||||
|
||||
|
||||
litellm_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
```
|
||||
|
|
@ -204,6 +203,12 @@ claude --model claude-3-5-haiku-20241022
|
|||
|
||||
# Use Bedrock deployment
|
||||
claude --model claude-bedrock
|
||||
|
||||
# Use Azure Foundry deployment
|
||||
claude --model claude-4-azure
|
||||
|
||||
# Use Vertex AI deployment
|
||||
claude --model anthropic-vertex
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -211,96 +216,3 @@ claude --model claude-bedrock
|
|||
|
||||
<Image img={require('../../img/release_notes/claude_code_demo.png')} style={{ width: '500px', height: 'auto' }} />
|
||||
|
||||
|
||||
## Connecting MCP Servers
|
||||
|
||||
You can also connect MCP servers to Claude Code via LiteLLM Proxy.
|
||||
|
||||
:::note
|
||||
|
||||
Limitations:
|
||||
|
||||
- Currently, only HTTP MCP servers are supported
|
||||
|
||||
:::
|
||||
|
||||
1. Add the MCP server to your `config.yaml`
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="github" label="GitHub MCP">
|
||||
|
||||
In this example, we'll add the Github MCP server to our `config.yaml`
|
||||
|
||||
```yaml title="config.yaml" showLineNumbers
|
||||
mcp_servers:
|
||||
github_mcp:
|
||||
url: "https://api.githubcopilot.com/mcp"
|
||||
auth_type: oauth2
|
||||
client_id: os.environ/GITHUB_OAUTH_CLIENT_ID
|
||||
client_secret: os.environ/GITHUB_OAUTH_CLIENT_SECRET
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="atlassian" label="Atlassian MCP">
|
||||
|
||||
In this example, we'll add the Atlassian MCP server to our `config.yaml`
|
||||
|
||||
```yaml title="config.yaml" showLineNumbers
|
||||
atlassian_mcp:
|
||||
server_id: atlassian_mcp_id
|
||||
url: "https://mcp.atlassian.com/v1/sse"
|
||||
transport: "sse"
|
||||
auth_type: oauth2
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Use the MCP server in Claude Code
|
||||
|
||||
```bash
|
||||
claude mcp add --transport http litellm_proxy http://0.0.0.0:4000/github_mcp/mcp --header "Authorization: Bearer sk-LITELLM_VIRTUAL_KEY"
|
||||
```
|
||||
|
||||
For MCP servers that require dynamic client registration (such as Atlassian), please set `x-litellm-api-key: Bearer sk-LITELLM_VIRTUAL_KEY` instead of using `Authorization: Bearer LITELLM_VIRTUAL_KEY`.
|
||||
|
||||
4. Authenticate via Claude Code
|
||||
|
||||
a. Start Claude Code
|
||||
|
||||
```bash
|
||||
claude
|
||||
```
|
||||
|
||||
b. Authenticate via Claude Code
|
||||
|
||||
```bash
|
||||
/mcp
|
||||
```
|
||||
|
||||
c. Select the MCP server
|
||||
|
||||
```bash
|
||||
> litellm_proxy
|
||||
```
|
||||
|
||||
d. Start Oauth flow via Claude Code
|
||||
|
||||
```bash
|
||||
> 1. Authenticate
|
||||
2. Reconnect
|
||||
3. Disable
|
||||
```
|
||||
|
||||
e. Once completed, you should see this success message:
|
||||
|
||||
<Image img={require('../../img/oauth_2_success.png')} style={{ width: '500px', height: 'auto' }} />
|
||||
|
||||
|
|
|
|||
|
|
@ -108,15 +108,30 @@ const sidebars = {
|
|||
{
|
||||
type: "category",
|
||||
label: "AI Tools (OpenWebUI, Claude Code, etc.)",
|
||||
link: {
|
||||
type: "generated-index",
|
||||
title: "AI Tools",
|
||||
description: "Integrate LiteLLM with AI tools like OpenWebUI, Claude Code, and more",
|
||||
slug: "/ai_tools"
|
||||
},
|
||||
items: [
|
||||
"tutorials/claude_responses_api",
|
||||
"tutorials/openweb_ui",
|
||||
{
|
||||
type: "category",
|
||||
label: "Claude Code",
|
||||
items: [
|
||||
"tutorials/claude_responses_api",
|
||||
"tutorials/claude_code_customer_tracking",
|
||||
"tutorials/claude_mcp",
|
||||
"tutorials/claude_non_anthropic_models",
|
||||
]
|
||||
},
|
||||
"tutorials/cost_tracking_coding",
|
||||
"tutorials/cursor_integration",
|
||||
"tutorials/github_copilot_integration",
|
||||
"tutorials/litellm_gemini_cli",
|
||||
"tutorials/litellm_qwen_code_cli",
|
||||
"tutorials/openai_codex",
|
||||
"tutorials/openweb_ui"
|
||||
"tutorials/openai_codex"
|
||||
]
|
||||
},
|
||||
|
||||
|
|
@ -862,10 +877,11 @@ const sidebars = {
|
|||
type: "category",
|
||||
label: "Tutorials",
|
||||
items: [
|
||||
"tutorials/openweb_ui",
|
||||
"tutorials/openai_codex",
|
||||
"tutorials/litellm_gemini_cli",
|
||||
"tutorials/litellm_qwen_code_cli",
|
||||
{
|
||||
type: "link",
|
||||
label: "AI Coding Tools (OpenWebUI, Claude Code, Gemini CLI, OpenAI Codex, etc.)",
|
||||
href: "/docs/ai_tools",
|
||||
},
|
||||
"tutorials/anthropic_file_usage",
|
||||
"tutorials/default_team_self_serve",
|
||||
"tutorials/msft_sso",
|
||||
|
|
@ -875,7 +891,6 @@ const sidebars = {
|
|||
"tutorials/presidio_pii_masking",
|
||||
"tutorials/elasticsearch_logging",
|
||||
"tutorials/gemini_realtime_with_audio",
|
||||
"tutorials/claude_responses_api",
|
||||
{
|
||||
type: "category",
|
||||
label: "LiteLLM Python SDK Tutorials",
|
||||
|
|
|
|||
19
document.txt
19
document.txt
|
|
@ -1,19 +0,0 @@
|
|||
LiteLLM provides a unified interface for calling 100+ different LLM providers.
|
||||
|
||||
Key capabilities:
|
||||
- Translate requests to provider-specific formats
|
||||
- Consistent OpenAI-compatible responses
|
||||
- Retry and fallback logic across deployments
|
||||
- Proxy server with authentication and rate limiting
|
||||
- Support for streaming, function calling, and embeddings
|
||||
|
||||
Popular providers supported:
|
||||
- OpenAI (GPT-4, GPT-3.5)
|
||||
- Anthropic (Claude)
|
||||
- AWS Bedrock
|
||||
- Azure OpenAI
|
||||
- Google Vertex AI
|
||||
- Cohere
|
||||
- And 95+ more
|
||||
|
||||
This allows developers to easily switch between providers without code changes.
|
||||
|
|
@ -1,6 +1,6 @@
|
|||
[tool.poetry]
|
||||
name = "litellm-enterprise"
|
||||
version = "0.1.27"
|
||||
version = "0.1.28"
|
||||
description = "Package for LiteLLM Enterprise features"
|
||||
authors = ["BerriAI"]
|
||||
readme = "README.md"
|
||||
|
|
@ -22,7 +22,7 @@ requires = ["poetry-core"]
|
|||
build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "0.1.27"
|
||||
version = "0.1.28"
|
||||
version_files = [
|
||||
"pyproject.toml:version",
|
||||
"../requirements.txt:litellm-enterprise==",
|
||||
|
|
|
|||
Binary file not shown.
|
Before Width: | Height: | Size: 172 KiB |
|
|
@ -1073,6 +1073,13 @@ LITELLM_TRUNCATED_PAYLOAD_FIELD = "litellm_truncated"
|
|||
|
||||
########################### LiteLLM Proxy Specific Constants ###########################
|
||||
########################################################################################
|
||||
|
||||
# Standard headers that are always checked for customer/end-user ID (no configuration required)
|
||||
# These headers work out-of-the-box for tools like Claude Code that support custom headers
|
||||
STANDARD_CUSTOMER_ID_HEADERS = [
|
||||
"x-litellm-customer-id",
|
||||
"x-litellm-end-user-id",
|
||||
]
|
||||
MAX_SPENDLOG_ROWS_TO_QUERY = int(
|
||||
os.getenv("MAX_SPENDLOG_ROWS_TO_QUERY", 1_000_000)
|
||||
) # if spendLogs has more than 1M rows, do not query the DB
|
||||
|
|
|
|||
|
|
@ -952,7 +952,8 @@ def completion_cost( # noqa: PLR0915
|
|||
)
|
||||
|
||||
potential_model_names = [selected_model, _get_response_model(completion_response)]
|
||||
|
||||
if model is not None:
|
||||
potential_model_names.append(model)
|
||||
|
||||
for idx, model in enumerate(potential_model_names):
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -404,6 +404,7 @@ def image_generation( # noqa: PLR0915
|
|||
litellm.LlmProviders.STABILITY,
|
||||
litellm.LlmProviders.RUNWAYML,
|
||||
litellm.LlmProviders.VERTEX_AI,
|
||||
litellm.LlmProviders.OPENROUTER
|
||||
):
|
||||
if image_generation_config is None:
|
||||
raise ValueError(
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ from typing import (
|
|||
import litellm
|
||||
from litellm._logging import print_verbose, verbose_logger
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import LiteLLM_TeamTable, UserAPIKeyAuth
|
||||
from litellm.proxy._types import LiteLLM_TeamTable, LiteLLM_UserTable, UserAPIKeyAuth
|
||||
from litellm.types.integrations.prometheus import *
|
||||
from litellm.types.integrations.prometheus import _sanitize_prometheus_label_name
|
||||
from litellm.types.utils import StandardLoggingPayload
|
||||
|
|
@ -52,7 +52,7 @@ def _get_cached_end_user_id_for_cost_tracking():
|
|||
|
||||
class PrometheusLogger(CustomLogger):
|
||||
# Class variables or attributes
|
||||
def __init__(
|
||||
def __init__( # noqa: PLR0915
|
||||
self,
|
||||
**kwargs,
|
||||
):
|
||||
|
|
@ -193,6 +193,30 @@ class PrometheusLogger(CustomLogger):
|
|||
),
|
||||
)
|
||||
|
||||
# Remaining Budget for User
|
||||
self.litellm_remaining_user_budget_metric = self._gauge_factory(
|
||||
"litellm_remaining_user_budget_metric",
|
||||
"Remaining budget for user",
|
||||
labelnames=self.get_labels_for_metric(
|
||||
"litellm_remaining_user_budget_metric"
|
||||
),
|
||||
)
|
||||
|
||||
# Max Budget for User
|
||||
self.litellm_user_max_budget_metric = self._gauge_factory(
|
||||
"litellm_user_max_budget_metric",
|
||||
"Maximum budget set for user",
|
||||
labelnames=self.get_labels_for_metric("litellm_user_max_budget_metric"),
|
||||
)
|
||||
|
||||
self.litellm_user_budget_remaining_hours_metric = self._gauge_factory(
|
||||
"litellm_user_budget_remaining_hours_metric",
|
||||
"Remaining hours for user budget to be reset",
|
||||
labelnames=self.get_labels_for_metric(
|
||||
"litellm_user_budget_remaining_hours_metric"
|
||||
),
|
||||
)
|
||||
|
||||
########################################
|
||||
# LiteLLM Virtual API KEY metrics
|
||||
########################################
|
||||
|
|
@ -960,6 +984,7 @@ class PrometheusLogger(CustomLogger):
|
|||
user_api_key_alias=user_api_key_alias,
|
||||
litellm_params=litellm_params,
|
||||
response_cost=response_cost,
|
||||
user_id=user_id,
|
||||
)
|
||||
|
||||
# set proxy virtual key rpm/tpm metrics
|
||||
|
|
@ -1120,6 +1145,7 @@ class PrometheusLogger(CustomLogger):
|
|||
user_api_key_alias: Optional[str],
|
||||
litellm_params: dict,
|
||||
response_cost: float,
|
||||
user_id: Optional[str] = None,
|
||||
):
|
||||
_team_spend = litellm_params.get("metadata", {}).get(
|
||||
"user_api_key_team_spend", None
|
||||
|
|
@ -1134,6 +1160,14 @@ class PrometheusLogger(CustomLogger):
|
|||
_api_key_max_budget = litellm_params.get("metadata", {}).get(
|
||||
"user_api_key_max_budget", None
|
||||
)
|
||||
|
||||
_user_spend = litellm_params.get("metadata", {}).get(
|
||||
"user_api_key_user_spend", None
|
||||
)
|
||||
_user_max_budget = litellm_params.get("metadata", {}).get(
|
||||
"user_api_key_user_max_budget", None
|
||||
)
|
||||
|
||||
await self._set_api_key_budget_metrics_after_api_request(
|
||||
user_api_key=user_api_key,
|
||||
user_api_key_alias=user_api_key_alias,
|
||||
|
|
@ -1150,6 +1184,13 @@ class PrometheusLogger(CustomLogger):
|
|||
response_cost=response_cost,
|
||||
)
|
||||
|
||||
await self._set_user_budget_metrics_after_api_request(
|
||||
user_id=user_id,
|
||||
user_spend=_user_spend,
|
||||
user_max_budget=_user_max_budget,
|
||||
response_cost=response_cost,
|
||||
)
|
||||
|
||||
def _increment_top_level_request_and_spend_metrics(
|
||||
self,
|
||||
end_user_id: Optional[str],
|
||||
|
|
@ -2229,6 +2270,37 @@ class PrometheusLogger(CustomLogger):
|
|||
data_type="keys",
|
||||
)
|
||||
|
||||
async def _initialize_user_budget_metrics(self):
|
||||
"""
|
||||
Initialize user budget metrics by reusing the generic pagination logic.
|
||||
"""
|
||||
from litellm.proxy._types import LiteLLM_UserTable
|
||||
from litellm.proxy.proxy_server import prisma_client
|
||||
|
||||
if prisma_client is None:
|
||||
verbose_logger.debug(
|
||||
"Prometheus: skipping user metrics initialization, DB not initialized"
|
||||
)
|
||||
return
|
||||
|
||||
async def fetch_users(
|
||||
page_size: int, page: int
|
||||
) -> Tuple[List[LiteLLM_UserTable], Optional[int]]:
|
||||
skip = (page - 1) * page_size
|
||||
users = await prisma_client.db.litellm_usertable.find_many(
|
||||
skip=skip,
|
||||
take=page_size,
|
||||
order={"created_at": "desc"},
|
||||
)
|
||||
total_count = await prisma_client.db.litellm_usertable.count()
|
||||
return users, total_count
|
||||
|
||||
await self._initialize_budget_metrics(
|
||||
data_fetch_function=fetch_users,
|
||||
set_metrics_function=self._set_user_list_budget_metrics,
|
||||
data_type="users",
|
||||
)
|
||||
|
||||
async def initialize_remaining_budget_metrics(self):
|
||||
"""
|
||||
Handler for initializing remaining budget metrics for all teams to avoid metric discrepancies.
|
||||
|
|
@ -2261,11 +2333,12 @@ class PrometheusLogger(CustomLogger):
|
|||
|
||||
async def _initialize_remaining_budget_metrics(self):
|
||||
"""
|
||||
Helper to initialize remaining budget metrics for all teams and API keys.
|
||||
Helper to initialize remaining budget metrics for all teams, API keys, and users.
|
||||
"""
|
||||
verbose_logger.debug("Emitting key, team budget metrics....")
|
||||
verbose_logger.debug("Emitting key, team, user budget metrics....")
|
||||
await self._initialize_team_budget_metrics()
|
||||
await self._initialize_api_key_budget_metrics()
|
||||
await self._initialize_user_budget_metrics()
|
||||
|
||||
async def _set_key_list_budget_metrics(
|
||||
self, keys: List[Union[str, UserAPIKeyAuth]]
|
||||
|
|
@ -2280,6 +2353,11 @@ class PrometheusLogger(CustomLogger):
|
|||
for team in teams:
|
||||
self._set_team_budget_metrics(team)
|
||||
|
||||
async def _set_user_list_budget_metrics(self, users: List[LiteLLM_UserTable]):
|
||||
"""Helper function to set budget metrics for a list of users"""
|
||||
for user in users:
|
||||
self._set_user_budget_metrics(user)
|
||||
|
||||
async def _set_team_budget_metrics_after_api_request(
|
||||
self,
|
||||
user_api_team: Optional[str],
|
||||
|
|
@ -2497,6 +2575,122 @@ class PrometheusLogger(CustomLogger):
|
|||
|
||||
return user_api_key_dict
|
||||
|
||||
async def _set_user_budget_metrics_after_api_request(
|
||||
self,
|
||||
user_id: Optional[str],
|
||||
user_spend: Optional[float],
|
||||
user_max_budget: Optional[float],
|
||||
response_cost: float,
|
||||
):
|
||||
"""
|
||||
Set user budget metrics after an LLM API request
|
||||
|
||||
- Assemble a LiteLLM_UserTable object
|
||||
- looks up user info from db if not available in metadata
|
||||
- Set user budget metrics
|
||||
"""
|
||||
if user_id:
|
||||
user_object = await self._assemble_user_object(
|
||||
user_id=user_id,
|
||||
spend=user_spend,
|
||||
max_budget=user_max_budget,
|
||||
response_cost=response_cost,
|
||||
)
|
||||
|
||||
self._set_user_budget_metrics(user_object)
|
||||
|
||||
async def _assemble_user_object(
|
||||
self,
|
||||
user_id: str,
|
||||
spend: Optional[float],
|
||||
max_budget: Optional[float],
|
||||
response_cost: float,
|
||||
) -> LiteLLM_UserTable:
|
||||
"""
|
||||
Assemble a LiteLLM_UserTable object
|
||||
|
||||
for fields not available in metadata, we fetch from db
|
||||
Fields not available in metadata:
|
||||
- `budget_reset_at`
|
||||
"""
|
||||
from litellm.proxy.auth.auth_checks import get_user_object
|
||||
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
||||
|
||||
_total_user_spend = (spend or 0) + response_cost
|
||||
user_object = LiteLLM_UserTable(
|
||||
user_id=user_id,
|
||||
spend=_total_user_spend,
|
||||
max_budget=max_budget,
|
||||
)
|
||||
try:
|
||||
user_info = await get_user_object(
|
||||
user_id=user_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
user_id_upsert=False,
|
||||
check_db_only=True,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(
|
||||
f"[Non-Blocking] Prometheus: Error getting user info: {str(e)}"
|
||||
)
|
||||
return user_object
|
||||
|
||||
if user_info:
|
||||
user_object.budget_reset_at = user_info.budget_reset_at
|
||||
|
||||
return user_object
|
||||
|
||||
def _set_user_budget_metrics(
|
||||
self,
|
||||
user: LiteLLM_UserTable,
|
||||
):
|
||||
"""
|
||||
Set user budget metrics for a single user
|
||||
|
||||
- Remaining Budget
|
||||
- Max Budget
|
||||
- Budget Reset At
|
||||
"""
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
user=user.user_id,
|
||||
)
|
||||
|
||||
_labels = prometheus_label_factory(
|
||||
supported_enum_labels=self.get_labels_for_metric(
|
||||
metric_name="litellm_remaining_user_budget_metric"
|
||||
),
|
||||
enum_values=enum_values,
|
||||
)
|
||||
self.litellm_remaining_user_budget_metric.labels(**_labels).set(
|
||||
self._safe_get_remaining_budget(
|
||||
max_budget=user.max_budget,
|
||||
spend=user.spend,
|
||||
)
|
||||
)
|
||||
|
||||
if user.max_budget is not None:
|
||||
_labels = prometheus_label_factory(
|
||||
supported_enum_labels=self.get_labels_for_metric(
|
||||
metric_name="litellm_user_max_budget_metric"
|
||||
),
|
||||
enum_values=enum_values,
|
||||
)
|
||||
self.litellm_user_max_budget_metric.labels(**_labels).set(user.max_budget)
|
||||
|
||||
if user.budget_reset_at is not None:
|
||||
_labels = prometheus_label_factory(
|
||||
supported_enum_labels=self.get_labels_for_metric(
|
||||
metric_name="litellm_user_budget_remaining_hours_metric"
|
||||
),
|
||||
enum_values=enum_values,
|
||||
)
|
||||
self.litellm_user_budget_remaining_hours_metric.labels(**_labels).set(
|
||||
self._get_remaining_hours_for_budget_reset(
|
||||
budget_reset_at=user.budget_reset_at
|
||||
)
|
||||
)
|
||||
|
||||
def _get_remaining_hours_for_budget_reset(self, budget_reset_at: datetime) -> float:
|
||||
"""
|
||||
Get remaining hours for budget reset
|
||||
|
|
|
|||
|
|
@ -3743,10 +3743,10 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
|
|||
OpenTelemetry,
|
||||
OpenTelemetryConfig,
|
||||
)
|
||||
|
||||
logfire_base_url = os.getenv("LOGFIRE_BASE_URL", "https://logfire-api.pydantic.dev")
|
||||
otel_config = OpenTelemetryConfig(
|
||||
exporter="otlp_http",
|
||||
endpoint="https://logfire-api.pydantic.dev/v1/traces",
|
||||
endpoint = f"{logfire_base_url.rstrip('/')}/v1/traces",
|
||||
headers=f"Authorization={os.getenv('LOGFIRE_TOKEN')}",
|
||||
)
|
||||
for callback in _in_memory_loggers:
|
||||
|
|
@ -4488,7 +4488,7 @@ class StandardLoggingPayloadSetup:
|
|||
|
||||
@staticmethod
|
||||
def get_usage_from_response_obj(
|
||||
response_obj: Optional[Union[dict, BaseModel]], combined_usage_object: Optional[Usage] = None
|
||||
response_obj: Optional[dict], combined_usage_object: Optional[Usage] = None
|
||||
) -> Usage:
|
||||
## BASE CASE ##
|
||||
if combined_usage_object is not None:
|
||||
|
|
@ -4500,32 +4500,27 @@ class StandardLoggingPayloadSetup:
|
|||
total_tokens=0,
|
||||
)
|
||||
|
||||
usage = _safe_extract_usage_from_obj(response_obj)
|
||||
|
||||
if usage is None:
|
||||
usage = response_obj.get("usage", None) or {}
|
||||
if usage is None or (
|
||||
not isinstance(usage, dict) and not isinstance(usage, Usage)
|
||||
):
|
||||
return Usage(
|
||||
prompt_tokens=0,
|
||||
completion_tokens=0,
|
||||
total_tokens=0,
|
||||
)
|
||||
|
||||
if isinstance(usage, Usage):
|
||||
elif isinstance(usage, Usage):
|
||||
return usage
|
||||
|
||||
transformed_usage = _try_transform_response_api_usage(usage)
|
||||
if transformed_usage is not None:
|
||||
return transformed_usage
|
||||
|
||||
if isinstance(usage, dict):
|
||||
created_usage = _try_create_usage_from_dict(usage)
|
||||
if created_usage is not None:
|
||||
return created_usage
|
||||
|
||||
return Usage(
|
||||
prompt_tokens=0,
|
||||
completion_tokens=0,
|
||||
total_tokens=0,
|
||||
)
|
||||
elif isinstance(usage, dict):
|
||||
if ResponseAPILoggingUtils._is_response_api_usage(usage):
|
||||
return (
|
||||
ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
|
||||
usage
|
||||
)
|
||||
)
|
||||
return Usage(**usage)
|
||||
|
||||
raise ValueError(f"usage is required, got={usage} of type {type(usage)}")
|
||||
|
||||
@staticmethod
|
||||
def get_model_cost_information(
|
||||
|
|
@ -4566,18 +4561,13 @@ class StandardLoggingPayloadSetup:
|
|||
|
||||
@staticmethod
|
||||
def get_final_response_obj(
|
||||
response_obj: Union[dict, BaseModel], init_response_obj: Union[Any, BaseModel, dict], kwargs: dict
|
||||
response_obj: dict, init_response_obj: Union[Any, BaseModel, dict], kwargs: dict
|
||||
) -> Optional[Union[dict, str, list]]:
|
||||
"""
|
||||
Get final response object after redacting the message input/output from logging
|
||||
"""
|
||||
if response_obj:
|
||||
if isinstance(response_obj, BaseModel):
|
||||
final_response_obj: Optional[Union[dict, str, list]] = _safe_model_dump(
|
||||
response_obj, default={}
|
||||
)
|
||||
else:
|
||||
final_response_obj = response_obj
|
||||
final_response_obj: Optional[Union[dict, str, list]] = response_obj
|
||||
elif isinstance(init_response_obj, list) or isinstance(init_response_obj, str):
|
||||
final_response_obj = init_response_obj
|
||||
else:
|
||||
|
|
@ -4591,7 +4581,7 @@ class StandardLoggingPayloadSetup:
|
|||
if modified_final_response_obj is not None and isinstance(
|
||||
modified_final_response_obj, BaseModel
|
||||
):
|
||||
final_response_obj = _safe_model_dump(modified_final_response_obj, default={})
|
||||
final_response_obj = modified_final_response_obj.model_dump()
|
||||
else:
|
||||
final_response_obj = modified_final_response_obj
|
||||
|
||||
|
|
@ -4862,125 +4852,6 @@ class StandardLoggingPayloadSetup:
|
|||
return request_tags
|
||||
|
||||
|
||||
def _safe_model_dump(
|
||||
obj: BaseModel, default: Optional[Union[dict, str, list]] = None
|
||||
) -> Union[dict, str, list]:
|
||||
"""
|
||||
Safely call model_dump() on a BaseModel with fallback strategies.
|
||||
|
||||
Args:
|
||||
obj: BaseModel instance to dump
|
||||
default: Default value to return if all strategies fail
|
||||
|
||||
Returns:
|
||||
Dict representation of the BaseModel, or fallback value
|
||||
"""
|
||||
if default is None:
|
||||
default = {}
|
||||
|
||||
try:
|
||||
return obj.model_dump()
|
||||
except (AttributeError, TypeError) as e:
|
||||
verbose_logger.debug(
|
||||
f"Error calling model_dump() on BaseModel: {e}, type: {type(obj)}"
|
||||
)
|
||||
try:
|
||||
if hasattr(obj, "__dict__"):
|
||||
return obj.__dict__
|
||||
else:
|
||||
return str(obj)
|
||||
except Exception:
|
||||
return default
|
||||
|
||||
|
||||
def _safe_get_attribute(
|
||||
obj: Union[dict, BaseModel, Any], attr_name: str, default: Any = None
|
||||
) -> Any:
|
||||
"""
|
||||
Safely get an attribute from a dict or BaseModel object.
|
||||
|
||||
Args:
|
||||
obj: Object to get attribute from (dict, BaseModel, or any object)
|
||||
attr_name: Name of the attribute to get
|
||||
default: Default value to return if attribute doesn't exist
|
||||
|
||||
Returns:
|
||||
Attribute value or default
|
||||
"""
|
||||
try:
|
||||
if isinstance(obj, dict):
|
||||
return obj.get(attr_name, default)
|
||||
else:
|
||||
return getattr(obj, attr_name, default)
|
||||
except (AttributeError, TypeError) as e:
|
||||
verbose_logger.debug(
|
||||
f"Error getting attribute '{attr_name}' from object: {e}, type: {type(obj)}"
|
||||
)
|
||||
return default
|
||||
|
||||
|
||||
def _safe_extract_usage_from_obj(
|
||||
response_obj: Union[dict, BaseModel, Any]
|
||||
) -> Optional[Union[dict, Usage, Any]]:
|
||||
"""
|
||||
Safely extract usage from response_obj (dict or BaseModel).
|
||||
|
||||
Args:
|
||||
response_obj: Response object (dict, BaseModel, or any object)
|
||||
|
||||
Returns:
|
||||
Usage object, dict, or None
|
||||
"""
|
||||
return _safe_get_attribute(response_obj, "usage", None)
|
||||
|
||||
|
||||
def _try_transform_response_api_usage(usage: Any) -> Optional[Usage]:
|
||||
"""
|
||||
Try to transform ResponseAPIUsage to Usage object.
|
||||
|
||||
Args:
|
||||
usage: Usage object (dict, ResponseAPIUsage, or other)
|
||||
|
||||
Returns:
|
||||
Transformed Usage object, or None if transformation fails
|
||||
"""
|
||||
try:
|
||||
if ResponseAPILoggingUtils._is_response_api_usage(usage):
|
||||
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage)
|
||||
except (AttributeError, TypeError, KeyError) as e:
|
||||
verbose_logger.debug(
|
||||
f"Error checking/transforming ResponseAPIUsage: {e}, type: {type(usage)}"
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _try_create_usage_from_dict(usage: dict) -> Optional[Usage]:
|
||||
"""
|
||||
Try to create Usage object from dict.
|
||||
|
||||
Args:
|
||||
usage: Dict containing usage information
|
||||
|
||||
Returns:
|
||||
Usage object, or None if creation fails
|
||||
"""
|
||||
try:
|
||||
return Usage(**usage)
|
||||
except (TypeError, ValueError) as e:
|
||||
# Avoid logging full dict contents, which may include sensitive data
|
||||
try:
|
||||
usage_keys = list(usage.keys())
|
||||
except Exception:
|
||||
usage_keys = None
|
||||
verbose_logger.debug(
|
||||
"Error creating Usage from dict: %s, usage keys: %s, usage type: %s",
|
||||
e,
|
||||
usage_keys,
|
||||
type(usage),
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _get_status_fields(
|
||||
status: StandardLoggingPayloadStatus,
|
||||
guardrail_information: Optional[List[dict]],
|
||||
|
|
@ -5030,21 +4901,17 @@ def _get_status_fields(
|
|||
def _extract_response_obj_and_hidden_params(
|
||||
init_response_obj: Union[Any, BaseModel, dict],
|
||||
original_exception: Optional[Exception],
|
||||
) -> Tuple[Union[dict, BaseModel], Optional[dict]]:
|
||||
|
||||
) -> Tuple[dict, Optional[dict]]:
|
||||
"""Extract response_obj and hidden_params from init_response_obj."""
|
||||
hidden_params: Optional[dict] = None
|
||||
if init_response_obj is None:
|
||||
response_obj: Union[dict, BaseModel] = {}
|
||||
response_obj = {}
|
||||
elif isinstance(init_response_obj, BaseModel):
|
||||
response_obj = init_response_obj
|
||||
hidden_params = _safe_get_attribute(init_response_obj, "_hidden_params", None)
|
||||
response_obj = init_response_obj.model_dump()
|
||||
hidden_params = getattr(init_response_obj, "_hidden_params", None)
|
||||
elif isinstance(init_response_obj, dict):
|
||||
response_obj = init_response_obj
|
||||
else:
|
||||
verbose_logger.debug(
|
||||
f"Unknown init_response_obj type: {type(init_response_obj)}, defaulting to empty dict"
|
||||
)
|
||||
response_obj = {}
|
||||
|
||||
if original_exception is not None and hidden_params is None:
|
||||
|
|
@ -5104,10 +4971,7 @@ def get_standard_logging_object_payload(
|
|||
),
|
||||
)
|
||||
|
||||
# Preserve falsy values (0, "", False) if they exist in response_obj
|
||||
id = _safe_get_attribute(response_obj, "id", None)
|
||||
if id is None:
|
||||
id = kwargs.get("litellm_call_id")
|
||||
id = response_obj.get("id", kwargs.get("litellm_call_id"))
|
||||
|
||||
_model_id = metadata.get("model_info", {}).get("id", "")
|
||||
_model_group = metadata.get("model_group", "")
|
||||
|
|
|
|||
|
|
@ -45,7 +45,6 @@ from .common_utils import (
|
|||
infer_content_type_from_url_and_content,
|
||||
is_non_content_values_set,
|
||||
parse_tool_call_arguments,
|
||||
unpack_defs,
|
||||
)
|
||||
from .image_handling import convert_url_to_base64
|
||||
|
||||
|
|
@ -1463,56 +1462,6 @@ def convert_to_gemini_tool_call_invoke(
|
|||
)
|
||||
|
||||
|
||||
def _clean_refs_for_gemini(obj: Any) -> None:
|
||||
"""
|
||||
Recursively clean $defs, $ref, and definitions from a dict for Gemini compatibility.
|
||||
|
||||
Gemini rejects:
|
||||
- $defs sections (even after $ref has been inlined)
|
||||
- Any remaining $ref (circular refs, external URLs)
|
||||
|
||||
This function:
|
||||
1. Removes all $defs/definitions keys
|
||||
2. Replaces any remaining $ref with a placeholder object
|
||||
"""
|
||||
if isinstance(obj, dict):
|
||||
# Remove $defs and definitions at this level
|
||||
obj.pop("$defs", None)
|
||||
obj.pop("definitions", None)
|
||||
|
||||
# Check for and handle remaining $ref (circular or external)
|
||||
if "$ref" in obj:
|
||||
ref_value = obj.pop("$ref")
|
||||
# Replace with a generic object type as placeholder
|
||||
obj["type"] = "object"
|
||||
obj["description"] = f"(schema reference: {ref_value})"
|
||||
|
||||
# Recurse into values
|
||||
for value in obj.values():
|
||||
_clean_refs_for_gemini(value)
|
||||
elif isinstance(obj, list):
|
||||
for item in obj:
|
||||
_clean_refs_for_gemini(item)
|
||||
|
||||
|
||||
def _prepare_response_for_gemini(response_data: dict) -> dict:
|
||||
"""
|
||||
Prepare a tool response dict for Gemini by inlining $ref and removing $defs.
|
||||
|
||||
Gemini rejects JSON schemas with $defs/$ref in function_response content.
|
||||
This function applies unpack_defs to inline references, then cleans up
|
||||
any remaining $defs sections and unresolved $refs (circular or external).
|
||||
|
||||
Returns a new dict (does not mutate the input).
|
||||
"""
|
||||
import copy
|
||||
|
||||
result = copy.deepcopy(response_data)
|
||||
unpack_defs(result, {})
|
||||
_clean_refs_for_gemini(result)
|
||||
return result
|
||||
|
||||
|
||||
def convert_to_gemini_tool_call_result(
|
||||
message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage],
|
||||
last_message_with_tool_calls: Optional[dict],
|
||||
|
|
@ -1621,11 +1570,6 @@ def convert_to_gemini_tool_call_result(
|
|||
# Not valid JSON, wrap in content field
|
||||
response_data = {"content": content_str}
|
||||
|
||||
# Gemini rejects JSON schemas with $defs/$ref in function_response content.
|
||||
# Inline $refs and clean up for Gemini compatibility.
|
||||
if isinstance(response_data, dict):
|
||||
response_data = _prepare_response_for_gemini(response_data)
|
||||
|
||||
# We can't determine from openai message format whether it's a successful or
|
||||
# error call result so default to the successful result template
|
||||
_function_response = VertexFunctionResponse(
|
||||
|
|
|
|||
|
|
@ -132,7 +132,7 @@ class ChunkProcessor:
|
|||
)
|
||||
return response
|
||||
|
||||
def get_combined_tool_content(
|
||||
def get_combined_tool_content( # noqa: PLR0915
|
||||
self, tool_call_chunks: List[Dict[str, Any]]
|
||||
) -> List[ChatCompletionMessageToolCall]:
|
||||
tool_calls_list: List[ChatCompletionMessageToolCall] = []
|
||||
|
|
@ -147,10 +147,26 @@ class ChunkProcessor:
|
|||
tool_calls = delta.get("tool_calls", [])
|
||||
|
||||
for tool_call in tool_calls:
|
||||
if not tool_call or not hasattr(tool_call, "function"):
|
||||
# Handle both dict and object formats
|
||||
if not tool_call:
|
||||
continue
|
||||
|
||||
# Check if tool_call has function (either as attribute or dict key)
|
||||
has_function = False
|
||||
if isinstance(tool_call, dict):
|
||||
has_function = "function" in tool_call and tool_call["function"] is not None
|
||||
else:
|
||||
has_function = hasattr(tool_call, "function") and tool_call.function is not None
|
||||
|
||||
if not has_function:
|
||||
continue
|
||||
|
||||
index = getattr(tool_call, "index", 0)
|
||||
# Get index (handle both dict and object)
|
||||
if isinstance(tool_call, dict):
|
||||
index = tool_call.get("index", 0)
|
||||
else:
|
||||
index = getattr(tool_call, "index", 0)
|
||||
|
||||
if index not in tool_call_map:
|
||||
tool_call_map[index] = {
|
||||
"id": None,
|
||||
|
|
@ -160,30 +176,56 @@ class ChunkProcessor:
|
|||
"provider_specific_fields": None,
|
||||
}
|
||||
|
||||
if hasattr(tool_call, "id") and tool_call.id:
|
||||
tool_call_map[index]["id"] = tool_call.id
|
||||
if hasattr(tool_call, "type") and tool_call.type:
|
||||
tool_call_map[index]["type"] = tool_call.type
|
||||
if hasattr(tool_call, "function"):
|
||||
if (
|
||||
hasattr(tool_call.function, "name")
|
||||
and tool_call.function.name
|
||||
):
|
||||
tool_call_map[index]["name"] = tool_call.function.name
|
||||
if (
|
||||
hasattr(tool_call.function, "arguments")
|
||||
and tool_call.function.arguments
|
||||
):
|
||||
tool_call_map[index]["arguments"].append(
|
||||
tool_call.function.arguments
|
||||
)
|
||||
# Extract id, type, and function data (handle both dict and object)
|
||||
if isinstance(tool_call, dict):
|
||||
if tool_call.get("id"):
|
||||
tool_call_map[index]["id"] = tool_call["id"]
|
||||
if tool_call.get("type"):
|
||||
tool_call_map[index]["type"] = tool_call["type"]
|
||||
|
||||
function = tool_call.get("function", {})
|
||||
if isinstance(function, dict):
|
||||
if function.get("name"):
|
||||
tool_call_map[index]["name"] = function["name"]
|
||||
if function.get("arguments"):
|
||||
tool_call_map[index]["arguments"].append(function["arguments"])
|
||||
else:
|
||||
# function is an object
|
||||
if hasattr(function, "name") and function.name:
|
||||
tool_call_map[index]["name"] = function.name
|
||||
if hasattr(function, "arguments") and function.arguments:
|
||||
tool_call_map[index]["arguments"].append(function.arguments)
|
||||
else:
|
||||
# tool_call is an object
|
||||
if hasattr(tool_call, "id") and tool_call.id:
|
||||
tool_call_map[index]["id"] = tool_call.id
|
||||
if hasattr(tool_call, "type") and tool_call.type:
|
||||
tool_call_map[index]["type"] = tool_call.type
|
||||
if hasattr(tool_call, "function"):
|
||||
if (
|
||||
hasattr(tool_call.function, "name")
|
||||
and tool_call.function.name
|
||||
):
|
||||
tool_call_map[index]["name"] = tool_call.function.name
|
||||
if (
|
||||
hasattr(tool_call.function, "arguments")
|
||||
and tool_call.function.arguments
|
||||
):
|
||||
tool_call_map[index]["arguments"].append(
|
||||
tool_call.function.arguments
|
||||
)
|
||||
|
||||
# Preserve provider_specific_fields from streaming chunks
|
||||
provider_fields = None
|
||||
if hasattr(tool_call, "provider_specific_fields") and tool_call.provider_specific_fields:
|
||||
provider_fields = tool_call.provider_specific_fields
|
||||
elif hasattr(tool_call, "function") and hasattr(tool_call.function, "provider_specific_fields") and tool_call.function.provider_specific_fields:
|
||||
provider_fields = tool_call.function.provider_specific_fields
|
||||
if isinstance(tool_call, dict):
|
||||
provider_fields = tool_call.get("provider_specific_fields")
|
||||
if not provider_fields and isinstance(tool_call.get("function"), dict):
|
||||
provider_fields = tool_call["function"].get("provider_specific_fields")
|
||||
else:
|
||||
if hasattr(tool_call, "provider_specific_fields") and tool_call.provider_specific_fields:
|
||||
provider_fields = tool_call.provider_specific_fields
|
||||
elif hasattr(tool_call, "function") and hasattr(tool_call.function, "provider_specific_fields") and tool_call.function.provider_specific_fields:
|
||||
provider_fields = tool_call.function.provider_specific_fields
|
||||
|
||||
if provider_fields:
|
||||
# Merge provider_specific_fields if multiple chunks have them
|
||||
|
|
@ -222,6 +264,7 @@ class ChunkProcessor:
|
|||
|
||||
return tool_calls_list
|
||||
|
||||
|
||||
def get_combined_function_call_content(
|
||||
self, function_call_chunks: List[Dict[str, Any]]
|
||||
) -> FunctionCall:
|
||||
|
|
|
|||
|
|
@ -2,7 +2,8 @@ from typing import Any, AsyncIterator, Dict, List, Optional, Tuple
|
|||
|
||||
import httpx
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj, verbose_logger
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.litellm_core_utils.litellm_logging import verbose_logger
|
||||
from litellm.llms.base_llm.anthropic_messages.transformation import (
|
||||
BaseAnthropicMessagesConfig,
|
||||
)
|
||||
|
|
@ -13,9 +14,10 @@ from litellm.types.llms.anthropic import (
|
|||
from litellm.types.llms.anthropic_messages.anthropic_response import (
|
||||
AnthropicMessagesResponse,
|
||||
)
|
||||
from litellm.types.llms.anthropic_tool_search import get_tool_search_beta_header
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
from ...common_utils import AnthropicError
|
||||
from ...common_utils import AnthropicError, AnthropicModelInfo
|
||||
|
||||
DEFAULT_ANTHROPIC_API_BASE = "https://api.anthropic.com"
|
||||
DEFAULT_ANTHROPIC_API_VERSION = "2023-06-01"
|
||||
|
|
@ -75,9 +77,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
if "content-type" not in headers:
|
||||
headers["content-type"] = "application/json"
|
||||
|
||||
headers = self._update_headers_with_optional_anthropic_beta(
|
||||
headers = self._update_headers_with_anthropic_beta(
|
||||
headers=headers,
|
||||
context_management=optional_params.get("context_management"),
|
||||
optional_params=optional_params,
|
||||
)
|
||||
|
||||
return headers, api_base
|
||||
|
|
@ -153,16 +155,44 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
)
|
||||
|
||||
@staticmethod
|
||||
def _update_headers_with_optional_anthropic_beta(
|
||||
headers: dict, context_management: Optional[Dict]
|
||||
def _update_headers_with_anthropic_beta(
|
||||
headers: dict,
|
||||
optional_params: dict,
|
||||
custom_llm_provider: str = "anthropic",
|
||||
) -> dict:
|
||||
if context_management is None:
|
||||
return headers
|
||||
|
||||
"""
|
||||
Auto-inject anthropic-beta headers based on features used.
|
||||
|
||||
Handles:
|
||||
- context_management: adds 'context-management-2025-06-27'
|
||||
- tool_search: adds provider-specific tool search header
|
||||
|
||||
Args:
|
||||
headers: Request headers dict
|
||||
optional_params: Optional parameters including tools, context_management
|
||||
custom_llm_provider: Provider name for looking up correct tool search header
|
||||
"""
|
||||
beta_values: set = set()
|
||||
|
||||
# Get existing beta headers if any
|
||||
existing_beta = headers.get("anthropic-beta")
|
||||
beta_value = ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value
|
||||
if existing_beta is None:
|
||||
headers["anthropic-beta"] = beta_value
|
||||
elif beta_value not in [beta.strip() for beta in existing_beta.split(",")]:
|
||||
headers["anthropic-beta"] = f"{existing_beta}, {beta_value}"
|
||||
if existing_beta:
|
||||
beta_values.update(b.strip() for b in existing_beta.split(","))
|
||||
|
||||
# Check for context management
|
||||
if optional_params.get("context_management") is not None:
|
||||
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value)
|
||||
|
||||
# Check for tool search tools
|
||||
tools = optional_params.get("tools")
|
||||
if tools:
|
||||
anthropic_model_info = AnthropicModelInfo()
|
||||
if anthropic_model_info.is_tool_search_used(tools):
|
||||
# Use provider-specific tool search header
|
||||
tool_search_header = get_tool_search_beta_header(custom_llm_provider)
|
||||
beta_values.add(tool_search_header)
|
||||
|
||||
if beta_values:
|
||||
headers["anthropic-beta"] = ",".join(sorted(beta_values))
|
||||
|
||||
return headers
|
||||
|
|
|
|||
|
|
@ -664,8 +664,29 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM):
|
|||
**data, timeout=timeout
|
||||
)
|
||||
headers = dict(raw_response.headers)
|
||||
response = raw_response.parse()
|
||||
|
||||
# Convert json.JSONDecodeError to AzureOpenAIError for two critical reasons:
|
||||
#
|
||||
# 1. ROUTER BEHAVIOR: The router relies on exception.status_code to determine cooldown logic:
|
||||
# - JSONDecodeError has no status_code → router skips cooldown evaluation
|
||||
# - AzureOpenAIError has status_code → router properly evaluates for cooldown
|
||||
#
|
||||
# 2. CONNECTION CLEANUP: When response.parse() throws JSONDecodeError, the response
|
||||
# body may not be fully consumed, preventing httpx from properly returning the
|
||||
# connection to the pool. By catching the exception and accessing raw_response.status_code,
|
||||
# we trigger httpx's internal cleanup logic. Without this:
|
||||
# - parse() fails → JSONDecodeError bubbles up → httpx never knows response was acknowledged → connection leak
|
||||
# This completely eliminates "Unclosed connection" warnings during high load.
|
||||
try:
|
||||
response = raw_response.parse()
|
||||
except json.JSONDecodeError as json_error:
|
||||
raise AzureOpenAIError(
|
||||
status_code=raw_response.status_code or 500,
|
||||
message=f"Failed to parse raw Azure embedding response: {str(json_error)}"
|
||||
) from json_error
|
||||
|
||||
stringified_response = response.model_dump()
|
||||
|
||||
## LOGGING
|
||||
logging_obj.post_call(
|
||||
input=input,
|
||||
|
|
|
|||
|
|
@ -62,10 +62,10 @@ class AzureAnthropicMessagesConfig(AnthropicMessagesConfig):
|
|||
if "content-type" not in headers:
|
||||
headers["content-type"] = "application/json"
|
||||
|
||||
# Update headers with optional anthropic beta features
|
||||
headers = self._update_headers_with_optional_anthropic_beta(
|
||||
# Update headers with anthropic beta features (context management, tool search, etc.)
|
||||
headers = self._update_headers_with_anthropic_beta(
|
||||
headers=headers,
|
||||
context_management=optional_params.get("context_management"),
|
||||
optional_params=optional_params,
|
||||
)
|
||||
|
||||
return headers, api_base
|
||||
|
|
|
|||
|
|
@ -425,6 +425,15 @@ def strip_bedrock_routing_prefix(model: str) -> str:
|
|||
return model
|
||||
|
||||
|
||||
def strip_bedrock_throughput_suffix(model: str) -> str:
|
||||
""" Strip throughput tier suffixes from Bedrock model names. """
|
||||
import re
|
||||
|
||||
# Pattern matches model:version:throughput where throughput is like 51k, 18k, etc.
|
||||
# Keep the model:version part, strip the :throughput suffix
|
||||
return re.sub(r"(:\d+):\d+k$", r"\1", model)
|
||||
|
||||
|
||||
def get_bedrock_base_model(model: str) -> str:
|
||||
"""
|
||||
Get the base model from the given model name.
|
||||
|
|
@ -432,9 +441,11 @@ def get_bedrock_base_model(model: str) -> str:
|
|||
Handle model names like:
|
||||
- "us.meta.llama3-2-11b-instruct-v1:0" -> "meta.llama3-2-11b-instruct-v1"
|
||||
- "bedrock/converse/model" -> "model"
|
||||
- "anthropic.claude-3-5-sonnet-20241022-v2:0:51k" -> "anthropic.claude-3-5-sonnet-20241022-v2:0"
|
||||
"""
|
||||
model = strip_bedrock_routing_prefix(model)
|
||||
model = extract_model_name_from_bedrock_arn(model)
|
||||
model = strip_bedrock_throughput_suffix(model)
|
||||
|
||||
potential_region = model.split(".", 1)[0]
|
||||
alt_potential_region = model.split("/", 1)[0]
|
||||
|
|
|
|||
|
|
@ -129,6 +129,37 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
if isinstance(cache_control, dict) and "ttl" in cache_control:
|
||||
cache_control.pop("ttl", None)
|
||||
|
||||
def _get_tool_search_beta_header_for_bedrock(
|
||||
self,
|
||||
model: str,
|
||||
tool_search_used: bool,
|
||||
programmatic_tool_calling_used: bool,
|
||||
input_examples_used: bool,
|
||||
beta_set: set,
|
||||
) -> None:
|
||||
"""
|
||||
Adjust tool search beta header for Bedrock.
|
||||
|
||||
Bedrock requires a different beta header for tool search on Opus 4 models
|
||||
when tool search is used without programmatic tool calling or input examples.
|
||||
|
||||
Note: On Amazon Bedrock, server-side tool search is only supported on Claude Opus 4
|
||||
with the `tool-search-tool-2025-10-19` beta header.
|
||||
|
||||
Ref: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool
|
||||
|
||||
Args:
|
||||
model: The model name
|
||||
tool_search_used: Whether tool search is used
|
||||
programmatic_tool_calling_used: Whether programmatic tool calling is used
|
||||
input_examples_used: Whether input examples are used
|
||||
beta_set: The set of beta headers to modify in-place
|
||||
"""
|
||||
if tool_search_used and not (programmatic_tool_calling_used or input_examples_used):
|
||||
beta_set.discard(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
|
||||
if "opus-4" in model.lower() or "opus_4" in model.lower():
|
||||
beta_set.add("tool-search-tool-2025-10-19")
|
||||
|
||||
def transform_anthropic_messages_request(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -189,13 +220,13 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
)
|
||||
beta_set.update(auto_betas)
|
||||
|
||||
if (
|
||||
tool_search_used
|
||||
and not (programmatic_tool_calling_used or input_examples_used)
|
||||
):
|
||||
beta_set.discard(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
|
||||
if "opus-4" in model.lower() or "opus_4" in model.lower():
|
||||
beta_set.add("tool-search-tool-2025-10-19")
|
||||
self._get_tool_search_beta_header_for_bedrock(
|
||||
model=model,
|
||||
tool_search_used=tool_search_used,
|
||||
programmatic_tool_calling_used=programmatic_tool_calling_used,
|
||||
input_examples_used=input_examples_used,
|
||||
beta_set=beta_set,
|
||||
)
|
||||
|
||||
if beta_set:
|
||||
anthropic_messages_request["anthropic_beta"] = list(beta_set)
|
||||
|
|
|
|||
|
|
@ -57,6 +57,19 @@ class OpenAIRealtime(OpenAIChatCompletion):
|
|||
|
||||
try:
|
||||
ssl_context = get_shared_realtime_ssl_context()
|
||||
# Log a masked request preview consistent with other endpoints.
|
||||
logging_obj.pre_call(
|
||||
input=None,
|
||||
api_key=api_key,
|
||||
additional_args={
|
||||
"api_base": url,
|
||||
"headers": {
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"OpenAI-Beta": "realtime=v1",
|
||||
},
|
||||
"complete_input_dict": {"query_params": query_params},
|
||||
},
|
||||
)
|
||||
async with websockets.connect( # type: ignore
|
||||
url,
|
||||
additional_headers={
|
||||
|
|
|
|||
13
litellm/llms/openrouter/image_generation/__init__.py
Normal file
13
litellm/llms/openrouter/image_generation/__init__.py
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
from litellm.llms.base_llm.image_generation.transformation import (
|
||||
BaseImageGenerationConfig,
|
||||
)
|
||||
|
||||
from .transformation import OpenRouterImageGenerationConfig
|
||||
|
||||
__all__ = [
|
||||
"OpenRouterImageGenerationConfig",
|
||||
]
|
||||
|
||||
|
||||
def get_openrouter_image_generation_config(model: str) -> BaseImageGenerationConfig:
|
||||
return OpenRouterImageGenerationConfig()
|
||||
414
litellm/llms/openrouter/image_generation/transformation.py
Normal file
414
litellm/llms/openrouter/image_generation/transformation.py
Normal file
|
|
@ -0,0 +1,414 @@
|
|||
"""
|
||||
OpenRouter Image Generation Support
|
||||
|
||||
OpenRouter provides image generation through chat completion endpoints.
|
||||
Models like google/gemini-2.5-flash-image return images in the message content.
|
||||
|
||||
Response format:
|
||||
{
|
||||
"choices": [{
|
||||
"message": {
|
||||
"content": "Here is a beautiful sunset for you! ",
|
||||
"role": "assistant",
|
||||
"images": [{
|
||||
"image_url": {"url": "data:image/png;base64,..."},
|
||||
"index": 0,
|
||||
"type": "image_url"
|
||||
}]
|
||||
}
|
||||
}],
|
||||
"usage": {
|
||||
"completion_tokens": 1299,
|
||||
"prompt_tokens": 6,
|
||||
"total_tokens": 1305,
|
||||
"completion_tokens_details": {"image_tokens": 1290},
|
||||
"cost": 0.0387243
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
from typing import TYPE_CHECKING, Any, List, Optional, Union
|
||||
|
||||
import httpx
|
||||
|
||||
import litellm
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.image_generation.transformation import (
|
||||
BaseImageGenerationConfig,
|
||||
)
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams, AllMessageValues
|
||||
from litellm.types.utils import ImageObject, ImageResponse, ImageUsage, ImageUsageInputTokensDetails
|
||||
from litellm.llms.openrouter.common_utils import OpenRouterException
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
else:
|
||||
LiteLLMLoggingObj = Any
|
||||
|
||||
|
||||
class OpenRouterImageGenerationConfig(BaseImageGenerationConfig):
|
||||
"""
|
||||
Configuration for OpenRouter image generation via chat completions.
|
||||
|
||||
OpenRouter uses chat completion endpoints for image generation,
|
||||
so we need to transform image generation requests to chat format
|
||||
and extract images from chat responses.
|
||||
"""
|
||||
|
||||
def get_supported_openai_params(
|
||||
self, model: str
|
||||
) -> List[OpenAIImageGenerationOptionalParams]:
|
||||
"""
|
||||
Get supported OpenAI parameters for OpenRouter image generation.
|
||||
|
||||
Since OpenRouter uses chat completions for image generation,
|
||||
we support standard image generation params.
|
||||
"""
|
||||
return [
|
||||
"size",
|
||||
"quality",
|
||||
"n",
|
||||
]
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
optional_params: dict,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict:
|
||||
"""
|
||||
Map image generation params to OpenRouter chat completion format.
|
||||
|
||||
Maps OpenAI parameters to OpenRouter's image_config format:
|
||||
- size -> image_config.aspect_ratio
|
||||
- quality -> image_config.image_size
|
||||
"""
|
||||
supported_params = self.get_supported_openai_params(model)
|
||||
|
||||
for key, value in non_default_params.items():
|
||||
if key in supported_params:
|
||||
if key == "size":
|
||||
# Map OpenAI size to OpenRouter aspect_ratio
|
||||
aspect_ratio = self._map_size_to_aspect_ratio(value)
|
||||
if "image_config" not in optional_params:
|
||||
optional_params["image_config"] = {}
|
||||
optional_params["image_config"]["aspect_ratio"] = aspect_ratio
|
||||
elif key == "quality":
|
||||
# Map OpenAI quality to OpenRouter image_size
|
||||
image_size = self._map_quality_to_image_size(value)
|
||||
if image_size:
|
||||
if "image_config" not in optional_params:
|
||||
optional_params["image_config"] = {}
|
||||
optional_params["image_config"]["image_size"] = image_size
|
||||
else:
|
||||
# Pass through other supported params (like n)
|
||||
optional_params[key] = value
|
||||
elif not drop_params:
|
||||
# If not supported and drop_params is False, pass through
|
||||
optional_params[key] = value
|
||||
|
||||
return optional_params
|
||||
|
||||
def _map_size_to_aspect_ratio(self, size: str) -> str:
|
||||
"""
|
||||
Map OpenAI size format to OpenRouter aspect_ratio format.
|
||||
|
||||
OpenAI sizes:
|
||||
- 1024x1024 (square)
|
||||
- 1536x1024 (landscape)
|
||||
- 1024x1536 (portrait)
|
||||
- 1792x1024 (wide landscape, dall-e-3)
|
||||
- 1024x1792 (tall portrait, dall-e-3)
|
||||
- 256x256, 512x512 (dall-e-2)
|
||||
- auto (default)
|
||||
|
||||
OpenRouter aspect_ratios:
|
||||
- 1:1 → 1024×1024 (default)
|
||||
- 2:3 → 832×1248
|
||||
- 3:2 → 1248×832
|
||||
- 3:4 → 864×1184
|
||||
- 4:3 → 1184×864
|
||||
- 4:5 → 896×1152
|
||||
- 5:4 → 1152×896
|
||||
- 9:16 → 768×1344
|
||||
- 16:9 → 1344×768
|
||||
- 21:9 → 1536×672
|
||||
"""
|
||||
size_to_aspect_ratio = {
|
||||
# Square formats
|
||||
"256x256": "1:1",
|
||||
"512x512": "1:1",
|
||||
"1024x1024": "1:1",
|
||||
# Landscape formats
|
||||
"1536x1024": "3:2", # 1.5:1 ratio, closest to 3:2
|
||||
"1792x1024": "16:9", # 1.75:1 ratio, closest to 16:9
|
||||
# Portrait formats
|
||||
"1024x1536": "2:3", # 0.67:1 ratio, closest to 2:3
|
||||
"1024x1792": "9:16", # 0.57:1 ratio, closest to 9:16
|
||||
# Default
|
||||
"auto": "1:1",
|
||||
}
|
||||
return size_to_aspect_ratio.get(size, "1:1")
|
||||
|
||||
def _map_quality_to_image_size(self, quality: str) -> Optional[str]:
|
||||
"""
|
||||
Map OpenAI quality to OpenRouter image_size format.
|
||||
|
||||
OpenAI quality values:
|
||||
- auto (default) - automatically select best quality
|
||||
- high, medium, low - for GPT image models
|
||||
- hd, standard - for dall-e-3
|
||||
|
||||
OpenRouter image_size values (Gemini only):
|
||||
- 1K → Standard resolution (default)
|
||||
- 2K → Higher resolution
|
||||
- 4K → Highest resolution
|
||||
"""
|
||||
quality_to_image_size = {
|
||||
# OpenAI quality mappings
|
||||
"low": "1K",
|
||||
"standard": "1K",
|
||||
"medium": "2K",
|
||||
"high": "4K",
|
||||
"hd": "4K",
|
||||
# Auto defaults to standard
|
||||
"auto": "1K",
|
||||
}
|
||||
return quality_to_image_size.get(quality)
|
||||
|
||||
def _set_usage_and_cost(
|
||||
self,
|
||||
model_response: ImageResponse,
|
||||
response_json: dict,
|
||||
model: str,
|
||||
) -> None:
|
||||
"""
|
||||
Extract and set usage and cost information from OpenRouter response.
|
||||
|
||||
Args:
|
||||
model_response: ImageResponse object to populate
|
||||
response_json: Parsed JSON response from OpenRouter
|
||||
model: The model name
|
||||
"""
|
||||
usage_data = response_json.get("usage", {})
|
||||
if usage_data:
|
||||
prompt_tokens = usage_data.get("prompt_tokens", 0)
|
||||
total_tokens = usage_data.get("total_tokens", 0)
|
||||
|
||||
completion_tokens_details = usage_data.get("completion_tokens_details", {})
|
||||
image_tokens = completion_tokens_details.get("image_tokens", 0)
|
||||
|
||||
model_response.usage = ImageUsage(
|
||||
input_tokens=prompt_tokens,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
image_tokens=0, # Input doesn't contain images for generation
|
||||
text_tokens=prompt_tokens,
|
||||
),
|
||||
output_tokens=image_tokens,
|
||||
total_tokens=total_tokens,
|
||||
)
|
||||
|
||||
cost = usage_data.get("cost")
|
||||
if cost is not None:
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response._hidden_params["additional_headers"][
|
||||
"llm_provider-x-litellm-response-cost"
|
||||
] = float(cost)
|
||||
|
||||
cost_details = usage_data.get("cost_details", {})
|
||||
if cost_details:
|
||||
if "response_cost_details" not in model_response._hidden_params:
|
||||
model_response._hidden_params["response_cost_details"] = {}
|
||||
model_response._hidden_params["response_cost_details"].update(cost_details)
|
||||
|
||||
model_response._hidden_params["model"] = response_json.get("model", model)
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: Optional[str],
|
||||
api_key: Optional[str],
|
||||
model: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
stream: Optional[bool] = None,
|
||||
) -> str:
|
||||
"""
|
||||
Get the complete URL for OpenRouter image generation.
|
||||
|
||||
OpenRouter uses chat completions endpoint for image generation.
|
||||
Default: https://openrouter.ai/api/v1/chat/completions
|
||||
"""
|
||||
if api_base:
|
||||
if not api_base.endswith("/chat/completions"):
|
||||
api_base = api_base.rstrip("/")
|
||||
return f"{api_base}/chat/completions"
|
||||
return api_base
|
||||
|
||||
return "https://openrouter.ai/api/v1/chat/completions"
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
headers: dict,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
api_key: Optional[str] = None,
|
||||
api_base: Optional[str] = None,
|
||||
) -> dict:
|
||||
api_key = (
|
||||
api_key
|
||||
or litellm.api_key
|
||||
or get_secret_str("OPENROUTER_API_KEY")
|
||||
)
|
||||
headers.update(
|
||||
{
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
}
|
||||
)
|
||||
return headers
|
||||
|
||||
def transform_image_generation_request(
|
||||
self,
|
||||
model: str,
|
||||
prompt: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
headers: dict,
|
||||
) -> dict:
|
||||
"""
|
||||
Transform image generation request to OpenRouter chat completion format.
|
||||
|
||||
Args:
|
||||
model: The model name
|
||||
prompt: The image generation prompt
|
||||
optional_params: Optional parameters (including image_config)
|
||||
litellm_params: LiteLLM parameters
|
||||
headers: Request headers
|
||||
|
||||
Returns:
|
||||
dict: Request body in chat completion format with image_config
|
||||
"""
|
||||
request_body = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": prompt
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
# These will be passed through to OpenRouter
|
||||
for key, value in optional_params.items():
|
||||
if key not in ["model", "messages", "modalities"]:
|
||||
request_body[key] = value
|
||||
|
||||
return request_body
|
||||
|
||||
def transform_image_generation_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
model_response: ImageResponse,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
request_data: dict,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
encoding: Any,
|
||||
api_key: Optional[str] = None,
|
||||
json_mode: Optional[bool] = None,
|
||||
) -> ImageResponse:
|
||||
"""
|
||||
Transform OpenRouter chat completion response to ImageResponse format.
|
||||
|
||||
Extracts images from the message content and maps usage/cost information.
|
||||
|
||||
Args:
|
||||
model: The model name
|
||||
raw_response: Raw HTTP response from OpenRouter
|
||||
model_response: ImageResponse object to populate
|
||||
logging_obj: Logging object
|
||||
request_data: Original request data
|
||||
optional_params: Optional parameters
|
||||
litellm_params: LiteLLM parameters
|
||||
encoding: Encoding
|
||||
api_key: API key
|
||||
json_mode: JSON mode flag
|
||||
|
||||
Returns:
|
||||
ImageResponse: Populated image response
|
||||
"""
|
||||
try:
|
||||
response_json = raw_response.json()
|
||||
except Exception as e:
|
||||
raise OpenRouterException(
|
||||
message=f"Error parsing OpenRouter response: {str(e)}",
|
||||
status_code=raw_response.status_code,
|
||||
headers=raw_response.headers,
|
||||
)
|
||||
|
||||
if not model_response.data:
|
||||
model_response.data = []
|
||||
|
||||
try:
|
||||
choices = response_json.get("choices", [])
|
||||
|
||||
for choice in choices:
|
||||
message = choice.get("message", {})
|
||||
images = message.get("images", [])
|
||||
|
||||
for image_data in images:
|
||||
image_url_obj = image_data.get("image_url", {})
|
||||
image_url = image_url_obj.get("url")
|
||||
|
||||
if image_url:
|
||||
if image_url.startswith("data:"):
|
||||
# Extract base64 data
|
||||
# Format: data:image/png;base64,<base64_data>
|
||||
parts = image_url.split(",", 1)
|
||||
b64_data = parts[1] if len(parts) > 1 else None
|
||||
|
||||
model_response.data.append(
|
||||
ImageObject(
|
||||
b64_json=b64_data,
|
||||
url=None,
|
||||
revised_prompt=None,
|
||||
)
|
||||
)
|
||||
else:
|
||||
model_response.data.append(
|
||||
ImageObject(
|
||||
b64_json=None,
|
||||
url=image_url,
|
||||
revised_prompt=None,
|
||||
)
|
||||
)
|
||||
|
||||
# Extract and set usage and cost information
|
||||
self._set_usage_and_cost(model_response, response_json, model)
|
||||
|
||||
return model_response
|
||||
|
||||
except Exception as e:
|
||||
raise OpenRouterException(
|
||||
message=f"Error transforming OpenRouter image generation response: {str(e)}",
|
||||
status_code=500,
|
||||
headers={},
|
||||
)
|
||||
|
||||
def get_error_class(
|
||||
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
|
||||
) -> BaseLLMException:
|
||||
"""Get the appropriate error class for OpenRouter errors."""
|
||||
return OpenRouterException(
|
||||
message=error_message,
|
||||
status_code=status_code,
|
||||
headers=headers,
|
||||
)
|
||||
|
|
@ -665,11 +665,11 @@ def add_object_type(schema):
|
|||
if "required" in schema and schema["required"] is None:
|
||||
schema.pop("required", None)
|
||||
# Gemini doesn't accept empty properties for object types
|
||||
# If properties is empty, remove it and the type field
|
||||
# If properties is empty, remove it but keep type as object
|
||||
if not properties:
|
||||
schema.pop("properties", None)
|
||||
schema.pop("type", None)
|
||||
schema.pop("required", None)
|
||||
schema["type"] = "object"
|
||||
else:
|
||||
schema["type"] = "object"
|
||||
for name, value in properties.items():
|
||||
|
|
@ -776,6 +776,16 @@ def get_vertex_location_from_url(url: str) -> Optional[str]:
|
|||
return match.group(1) if match else None
|
||||
|
||||
|
||||
def get_vertex_model_id_from_url(url: str) -> Optional[str]:
|
||||
"""
|
||||
Get the vertex model id from the url
|
||||
|
||||
`https://${LOCATION}-aiplatform.googleapis.com/v1/projects/${PROJECT_ID}/locations/${LOCATION}/publishers/google/models/${MODEL_ID}:streamGenerateContent`
|
||||
"""
|
||||
match = re.search(r"/models/([^/:]+)", url)
|
||||
return match.group(1) if match else None
|
||||
|
||||
|
||||
def replace_project_and_location_in_route(
|
||||
requested_route: str, vertex_project: str, vertex_location: str
|
||||
) -> str:
|
||||
|
|
@ -825,6 +835,15 @@ def construct_target_url(
|
|||
if "cachedContent" in requested_route:
|
||||
vertex_version = "v1beta1"
|
||||
|
||||
# Check if the requested route starts with a version
|
||||
# e.g. /v1beta1/publishers/google/models/gemini-3-pro-preview:streamGenerateContent
|
||||
if requested_route.startswith("/v1/"):
|
||||
vertex_version = "v1"
|
||||
requested_route = requested_route.replace("/v1/", "/", 1)
|
||||
elif requested_route.startswith("/v1beta1/"):
|
||||
vertex_version = "v1beta1"
|
||||
requested_route = requested_route.replace("/v1beta1/", "/", 1)
|
||||
|
||||
base_requested_route = "{}/projects/{}/locations/{}".format(
|
||||
vertex_version, vertex_project, vertex_location
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,11 +1,16 @@
|
|||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
|
||||
AnthropicMessagesConfig,
|
||||
)
|
||||
from litellm.types.llms.anthropic import (
|
||||
ANTHROPIC_BETA_HEADER_VALUES,
|
||||
ANTHROPIC_HOSTED_TOOLS,
|
||||
)
|
||||
from litellm.types.llms.anthropic_tool_search import get_tool_search_beta_header
|
||||
from litellm.types.llms.vertex_ai import VertexPartnerProvider
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES, ANTHROPIC_HOSTED_TOOLS
|
||||
|
||||
from ....vertex_llm_base import VertexBase
|
||||
|
||||
|
|
@ -51,13 +56,28 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
|
|||
|
||||
headers["content-type"] = "application/json"
|
||||
|
||||
# Add web search beta header for Vertex AI only if not already set
|
||||
if "anthropic-beta" not in headers:
|
||||
tools = optional_params.get("tools", [])
|
||||
for tool in tools:
|
||||
if isinstance(tool, dict) and tool.get("type", "").startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value):
|
||||
headers["anthropic-beta"] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
|
||||
break
|
||||
# Add beta headers for Vertex AI
|
||||
tools = optional_params.get("tools", [])
|
||||
beta_values: set[str] = set()
|
||||
|
||||
# Get existing beta headers if any
|
||||
existing_beta = headers.get("anthropic-beta")
|
||||
if existing_beta:
|
||||
beta_values.update(b.strip() for b in existing_beta.split(","))
|
||||
|
||||
# Check for web search tool
|
||||
for tool in tools:
|
||||
if isinstance(tool, dict) and tool.get("type", "").startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value):
|
||||
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value)
|
||||
break
|
||||
|
||||
# Check for tool search tools - Vertex AI uses different beta header
|
||||
anthropic_model_info = AnthropicModelInfo()
|
||||
if anthropic_model_info.is_tool_search_used(tools):
|
||||
beta_values.add(get_tool_search_beta_header("vertex_ai"))
|
||||
|
||||
if beta_values:
|
||||
headers["anthropic-beta"] = ",".join(beta_values)
|
||||
|
||||
return headers, api_base
|
||||
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ from typing import (
|
|||
Callable,
|
||||
Coroutine,
|
||||
Dict,
|
||||
Iterable,
|
||||
List,
|
||||
Literal,
|
||||
Mapping,
|
||||
|
|
@ -1094,23 +1095,68 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
# validate tool_choice
|
||||
tool_choice = validate_chat_completion_tool_choice(tool_choice=tool_choice)
|
||||
|
||||
######### unpacking kwargs #####################
|
||||
args = locals()
|
||||
|
||||
skip_mcp_handler = kwargs.pop("_skip_mcp_handler", False)
|
||||
if not skip_mcp_handler and tools:
|
||||
from litellm.responses.mcp.chat_completions_handler import (
|
||||
handle_chat_completion_with_mcp,
|
||||
acompletion_with_mcp,
|
||||
)
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
from litellm.types.llms.openai import ToolParam
|
||||
|
||||
mcp_handler_context = locals().copy()
|
||||
completion_callable = globals().get("acompletion")
|
||||
mcp_result = run_async_function(
|
||||
handle_chat_completion_with_mcp,
|
||||
mcp_handler_context,
|
||||
completion_callable,
|
||||
)
|
||||
if mcp_result is not None:
|
||||
return mcp_result
|
||||
######### unpacking kwargs #####################
|
||||
args = locals()
|
||||
# Check if MCP tools are present (following responses pattern)
|
||||
# Cast tools to Optional[Iterable[ToolParam]] for type checking
|
||||
tools_for_mcp = cast(Optional[Iterable[ToolParam]], tools)
|
||||
if LiteLLM_Proxy_MCP_Handler._should_use_litellm_mcp_gateway(tools=tools_for_mcp):
|
||||
# Return coroutine - acompletion will await it
|
||||
# completion() can return a coroutine when MCP tools are present, which acompletion() awaits
|
||||
return acompletion_with_mcp( # type: ignore[return-value]
|
||||
model=model,
|
||||
messages=messages,
|
||||
functions=functions,
|
||||
function_call=function_call,
|
||||
timeout=timeout,
|
||||
temperature=temperature,
|
||||
top_p=top_p,
|
||||
n=n,
|
||||
stream=stream,
|
||||
stream_options=stream_options,
|
||||
stop=stop,
|
||||
max_tokens=max_tokens,
|
||||
max_completion_tokens=max_completion_tokens,
|
||||
modalities=modalities,
|
||||
prediction=prediction,
|
||||
audio=audio,
|
||||
presence_penalty=presence_penalty,
|
||||
frequency_penalty=frequency_penalty,
|
||||
logit_bias=logit_bias,
|
||||
user=user,
|
||||
response_format=response_format,
|
||||
seed=seed,
|
||||
tools=tools,
|
||||
tool_choice=tool_choice,
|
||||
parallel_tool_calls=parallel_tool_calls,
|
||||
logprobs=logprobs,
|
||||
top_logprobs=top_logprobs,
|
||||
deployment_id=deployment_id,
|
||||
reasoning_effort=reasoning_effort,
|
||||
verbosity=verbosity,
|
||||
safety_identifier=safety_identifier,
|
||||
service_tier=service_tier,
|
||||
base_url=base_url,
|
||||
api_version=api_version,
|
||||
api_key=api_key,
|
||||
model_list=model_list,
|
||||
extra_headers=extra_headers,
|
||||
thinking=thinking,
|
||||
web_search_options=web_search_options,
|
||||
shared_session=shared_session,
|
||||
**kwargs,
|
||||
)
|
||||
api_base = kwargs.get("api_base", None)
|
||||
mock_response: Optional[MOCK_RESPONSE_TYPE] = kwargs.get("mock_response", None)
|
||||
mock_tool_calls = kwargs.get("mock_tool_calls", None)
|
||||
|
|
|
|||
|
|
@ -28782,13 +28782,13 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"vertex_ai/zai-org/glm-4.7-maas": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "vertex_ai-zai_models",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_token": 2.2e-06,
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -14,76 +14,3 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/gpt-4.1-mini
|
||||
|
||||
|
||||
# guardrails:
|
||||
# - guardrail_name: generic-guardrail
|
||||
# litellm_params:
|
||||
# guardrail: generic_guardrail_api
|
||||
# mode: ["pre_call"]
|
||||
# headers:
|
||||
# Authorization: Bearer mock-bedrock-token-12345
|
||||
# api_base: http://localhost:8080
|
||||
# default_on: true
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "harmful-content-filter"
|
||||
litellm_params:
|
||||
guardrail: litellm_content_filter
|
||||
mode: "pre_call"
|
||||
default_on: true
|
||||
# Model configuration
|
||||
image_model: "claude-sonnet-4-5-20250929"
|
||||
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium" # Block medium+
|
||||
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit
|
||||
|
||||
- category: "harmful_illegal_weapons"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "low" # Strictest
|
||||
|
||||
- category: "bias_gender"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "bias_sexual_orientation"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "denied_medical_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "denied_legal_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "denied_financial_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
|
||||
prompts:
|
||||
- prompt_id: "simple_prompt"
|
||||
litellm_params:
|
||||
guardrail: generic_guardrail_api
|
||||
mode: ["post_call"]
|
||||
headers:
|
||||
Authorization: Bearer mock-bedrock-token-12345
|
||||
api_base: http://localhost:8080
|
||||
api_key: os.environ/BRAINTRUST_API_KEY
|
||||
ignore_prompt_manager_model: true
|
||||
ignore_prompt_manager_optional_params: true
|
||||
|
|
|
|||
|
|
@ -2189,6 +2189,8 @@ class UserAPIKeyAuth(
|
|||
user_tpm_limit: Optional[int] = None
|
||||
user_rpm_limit: Optional[int] = None
|
||||
user_email: Optional[str] = None
|
||||
user_spend: Optional[float] = None
|
||||
user_max_budget: Optional[float] = None
|
||||
request_route: Optional[str] = None
|
||||
user: Optional[Any] = None # Expanded user object when expand=user is used
|
||||
|
||||
|
|
|
|||
|
|
@ -74,75 +74,6 @@ db_cache_expiry = DEFAULT_IN_MEMORY_TTL # refresh every 5s
|
|||
all_routes = LiteLLMRoutes.openai_routes.value + LiteLLMRoutes.management_routes.value
|
||||
|
||||
|
||||
def _is_model_cost_zero(
|
||||
model: Optional[Union[str, List[str]]], llm_router: Optional[Router]
|
||||
) -> bool:
|
||||
"""
|
||||
Check if a model has zero cost (no configured pricing).
|
||||
|
||||
Uses the router's get_model_group_info method to get pricing information.
|
||||
|
||||
Args:
|
||||
model: The model name or list of model names
|
||||
llm_router: The LiteLLM router instance
|
||||
|
||||
Returns:
|
||||
bool: True if all costs for the model are zero, False otherwise
|
||||
"""
|
||||
if model is None or llm_router is None:
|
||||
return False
|
||||
|
||||
# Handle list of models
|
||||
model_list = [model] if isinstance(model, str) else model
|
||||
|
||||
for model_name in model_list:
|
||||
try:
|
||||
# Use router's get_model_group_info method directly for better reliability
|
||||
model_group_info = llm_router.get_model_group_info(model_group=model_name)
|
||||
|
||||
if model_group_info is None:
|
||||
# Model not found or no pricing info available
|
||||
# Conservative approach: assume it has cost
|
||||
verbose_proxy_logger.debug(
|
||||
f"No model group info found for {model_name}, assuming it has cost"
|
||||
)
|
||||
return False
|
||||
|
||||
# Check costs for this model
|
||||
# Only allow bypass if BOTH costs are explicitly set to 0 (not None)
|
||||
input_cost = model_group_info.input_cost_per_token
|
||||
output_cost = model_group_info.output_cost_per_token
|
||||
|
||||
# If costs are not explicitly configured (None), assume it has cost
|
||||
if input_cost is None or output_cost is None:
|
||||
verbose_proxy_logger.debug(
|
||||
f"Model {model_name} has undefined cost (input: {input_cost}, output: {output_cost}), assuming it has cost"
|
||||
)
|
||||
return False
|
||||
|
||||
# If either cost is non-zero, return False
|
||||
if input_cost > 0 or output_cost > 0:
|
||||
verbose_proxy_logger.debug(
|
||||
f"Model {model_name} has non-zero cost (input: {input_cost}, output: {output_cost})"
|
||||
)
|
||||
return False
|
||||
|
||||
# This model has zero cost explicitly configured
|
||||
verbose_proxy_logger.debug(
|
||||
f"Model {model_name} has zero cost explicitly configured (input: {input_cost}, output: {output_cost})"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
# If we can't determine the cost, assume it has cost (conservative approach)
|
||||
verbose_proxy_logger.debug(
|
||||
f"Error checking cost for model {model_name}: {str(e)}, assuming it has cost"
|
||||
)
|
||||
return False
|
||||
|
||||
# All models checked have zero cost
|
||||
return True
|
||||
|
||||
|
||||
async def common_checks(
|
||||
request_body: dict,
|
||||
team_object: Optional[LiteLLM_TeamTable],
|
||||
|
|
@ -155,7 +86,6 @@ async def common_checks(
|
|||
proxy_logging_obj: ProxyLogging,
|
||||
valid_token: Optional[UserAPIKeyAuth],
|
||||
request: Request,
|
||||
skip_budget_checks: bool = False,
|
||||
) -> bool:
|
||||
"""
|
||||
Common checks across jwt + key-based auth.
|
||||
|
|
@ -207,66 +137,64 @@ async def common_checks(
|
|||
user_object=user_object,
|
||||
)
|
||||
|
||||
# If this is a free model, skip all budget checks
|
||||
if not skip_budget_checks:
|
||||
# 3. If team is in budget
|
||||
await _team_max_budget_check(
|
||||
team_object=team_object,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
valid_token=valid_token,
|
||||
)
|
||||
# 3. If team is in budget
|
||||
await _team_max_budget_check(
|
||||
team_object=team_object,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
valid_token=valid_token,
|
||||
)
|
||||
|
||||
# 3.1. If organization is in budget
|
||||
await _organization_max_budget_check(
|
||||
valid_token=valid_token,
|
||||
team_object=team_object,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
# 3.1. If organization is in budget
|
||||
await _organization_max_budget_check(
|
||||
valid_token=valid_token,
|
||||
team_object=team_object,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
|
||||
await _tag_max_budget_check(
|
||||
request_body=request_body,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
valid_token=valid_token,
|
||||
)
|
||||
await _tag_max_budget_check(
|
||||
request_body=request_body,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
valid_token=valid_token,
|
||||
)
|
||||
|
||||
# 4. If user is in budget
|
||||
## 4.1 check personal budget, if personal key
|
||||
if (
|
||||
(team_object is None or team_object.team_id is None)
|
||||
and user_object is not None
|
||||
and user_object.max_budget is not None
|
||||
):
|
||||
user_budget = user_object.max_budget
|
||||
if user_budget < user_object.spend:
|
||||
raise litellm.BudgetExceededError(
|
||||
current_cost=user_object.spend,
|
||||
max_budget=user_budget,
|
||||
message=f"ExceededBudget: User={user_object.user_id} over budget. Spend={user_object.spend}, Budget={user_budget}",
|
||||
)
|
||||
# 4. If user is in budget
|
||||
## 4.1 check personal budget, if personal key
|
||||
if (
|
||||
(team_object is None or team_object.team_id is None)
|
||||
and user_object is not None
|
||||
and user_object.max_budget is not None
|
||||
):
|
||||
user_budget = user_object.max_budget
|
||||
if user_budget < user_object.spend:
|
||||
raise litellm.BudgetExceededError(
|
||||
current_cost=user_object.spend,
|
||||
max_budget=user_budget,
|
||||
message=f"ExceededBudget: User={user_object.user_id} over budget. Spend={user_object.spend}, Budget={user_budget}",
|
||||
)
|
||||
|
||||
## 4.2 check team member budget, if team key
|
||||
await _check_team_member_budget(
|
||||
team_object=team_object,
|
||||
user_object=user_object,
|
||||
valid_token=valid_token,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
## 4.2 check team member budget, if team key
|
||||
await _check_team_member_budget(
|
||||
team_object=team_object,
|
||||
user_object=user_object,
|
||||
valid_token=valid_token,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
|
||||
# 5. If end_user ('user' passed to /chat/completions, /embeddings endpoint) is in budget
|
||||
if end_user_object is not None and end_user_object.litellm_budget_table is not None:
|
||||
end_user_budget = end_user_object.litellm_budget_table.max_budget
|
||||
if end_user_budget is not None and end_user_object.spend > end_user_budget:
|
||||
raise litellm.BudgetExceededError(
|
||||
current_cost=end_user_object.spend,
|
||||
max_budget=end_user_budget,
|
||||
message=f"ExceededBudget: End User={end_user_object.user_id} over budget. Spend={end_user_object.spend}, Budget={end_user_budget}",
|
||||
)
|
||||
# 5. If end_user ('user' passed to /chat/completions, /embeddings endpoint) is in budget
|
||||
if end_user_object is not None and end_user_object.litellm_budget_table is not None:
|
||||
end_user_budget = end_user_object.litellm_budget_table.max_budget
|
||||
if end_user_budget is not None and end_user_object.spend > end_user_budget:
|
||||
raise litellm.BudgetExceededError(
|
||||
current_cost=end_user_object.spend,
|
||||
max_budget=end_user_budget,
|
||||
message=f"ExceededBudget: End User={end_user_object.user_id} over budget. Spend={end_user_object.spend}, Budget={end_user_budget}",
|
||||
)
|
||||
|
||||
# 6. [OPTIONAL] If 'enforce_user_param' enabled - did developer pass in 'user' param for openai endpoints
|
||||
if (
|
||||
|
|
@ -309,7 +237,6 @@ async def common_checks(
|
|||
# 7. [OPTIONAL] If 'litellm.max_budget' is set (>0), is proxy under budget
|
||||
if (
|
||||
litellm.max_budget > 0
|
||||
and not skip_budget_checks
|
||||
and global_proxy_spend is not None
|
||||
# only run global budget checks for OpenAI routes
|
||||
# Reason - the Admin UI should continue working if the proxy crosses it's global budget
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ from fastapi import HTTPException, Request, status
|
|||
|
||||
from litellm import Router, provider_list
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.constants import STANDARD_CUSTOMER_ID_HEADERS
|
||||
from litellm.proxy._types import *
|
||||
from litellm.types.router import CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS
|
||||
|
||||
|
|
@ -561,6 +562,32 @@ def get_customer_user_header_from_mapping(user_id_mapping) -> Optional[str]:
|
|||
return header_name
|
||||
return None
|
||||
|
||||
def _get_customer_id_from_standard_headers(
|
||||
request_headers: Optional[dict],
|
||||
) -> Optional[str]:
|
||||
"""
|
||||
Check standard customer ID headers for a customer/end-user ID.
|
||||
|
||||
This enables tools like Claude Code to pass customer IDs via ANTHROPIC_CUSTOM_HEADERS.
|
||||
No configuration required - these headers are always checked.
|
||||
|
||||
Args:
|
||||
request_headers: The request headers dict
|
||||
|
||||
Returns:
|
||||
The customer ID if found in standard headers, None otherwise
|
||||
"""
|
||||
if request_headers is None:
|
||||
return None
|
||||
|
||||
for standard_header in STANDARD_CUSTOMER_ID_HEADERS:
|
||||
for header_name, header_value in request_headers.items():
|
||||
if header_name.lower() == standard_header.lower():
|
||||
user_id_str = str(header_value) if header_value is not None else ""
|
||||
if user_id_str.strip():
|
||||
return user_id_str
|
||||
return None
|
||||
|
||||
|
||||
def get_end_user_id_from_request_body(
|
||||
request_body: dict, request_headers: Optional[dict] = None
|
||||
|
|
@ -569,7 +596,12 @@ def get_end_user_id_from_request_body(
|
|||
# and to ensure it's fetched at runtime.
|
||||
from litellm.proxy.proxy_server import general_settings
|
||||
|
||||
# Check 1 : Follow the user header mappings feature, if not found, then check for deprecated user_header_name (only if request_headers is provided)
|
||||
# Check 1: Standard customer ID headers (always checked, no configuration required)
|
||||
customer_id = _get_customer_id_from_standard_headers(request_headers=request_headers)
|
||||
if customer_id is not None:
|
||||
return customer_id
|
||||
|
||||
# Check 2: Follow the user header mappings feature, if not found, then check for deprecated user_header_name (only if request_headers is provided)
|
||||
# User query: "system not respecting user_header_name property"
|
||||
# This implies the key in general_settings is 'user_header_name'.
|
||||
if request_headers is not None:
|
||||
|
|
@ -602,19 +634,19 @@ def get_end_user_id_from_request_body(
|
|||
if user_id_str.strip():
|
||||
return user_id_str
|
||||
|
||||
# Check 2: 'user' field in request_body (commonly OpenAI)
|
||||
# Check 3: 'user' field in request_body (commonly OpenAI)
|
||||
if "user" in request_body and request_body["user"] is not None:
|
||||
user_from_body_user_field = request_body["user"]
|
||||
return str(user_from_body_user_field)
|
||||
|
||||
# Check 3: 'litellm_metadata.user' in request_body (commonly Anthropic)
|
||||
# Check 4: 'litellm_metadata.user' in request_body (commonly Anthropic)
|
||||
litellm_metadata = request_body.get("litellm_metadata")
|
||||
if isinstance(litellm_metadata, dict):
|
||||
user_from_litellm_metadata = litellm_metadata.get("user")
|
||||
if user_from_litellm_metadata is not None:
|
||||
return str(user_from_litellm_metadata)
|
||||
|
||||
# Check 4: 'metadata.user_id' in request_body (another common pattern)
|
||||
# Check 5: 'metadata.user_id' in request_body (another common pattern)
|
||||
metadata_dict = request_body.get("metadata")
|
||||
if isinstance(metadata_dict, dict):
|
||||
user_id_from_metadata_field = metadata_dict.get("user_id")
|
||||
|
|
|
|||
|
|
@ -586,21 +586,6 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
|
|||
if team_object is not None
|
||||
else None,
|
||||
)
|
||||
|
||||
# Check if model has zero cost - if so, skip all budget checks
|
||||
model = get_model_from_request(request_data, route)
|
||||
skip_budget_checks = False
|
||||
if model is not None and llm_router is not None:
|
||||
from litellm.proxy.auth.auth_checks import _is_model_cost_zero
|
||||
|
||||
skip_budget_checks = _is_model_cost_zero(
|
||||
model=model, llm_router=llm_router
|
||||
)
|
||||
if skip_budget_checks:
|
||||
verbose_proxy_logger.info(
|
||||
f"Skipping all budget checks for zero-cost model: {model}"
|
||||
)
|
||||
|
||||
# run through common checks
|
||||
_ = await common_checks(
|
||||
request=request,
|
||||
|
|
@ -614,7 +599,6 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
|
|||
llm_router=llm_router,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
valid_token=valid_token,
|
||||
skip_budget_checks=skip_budget_checks,
|
||||
)
|
||||
|
||||
# return UserAPIKeyAuth object
|
||||
|
|
@ -1006,22 +990,8 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
|
|||
)
|
||||
user_obj = None
|
||||
|
||||
# Check 2a. Check if model has zero cost - if so, skip all budget checks
|
||||
model = get_model_from_request(request_data, route)
|
||||
skip_budget_checks = False
|
||||
if model is not None and llm_router is not None:
|
||||
from litellm.proxy.auth.auth_checks import _is_model_cost_zero
|
||||
|
||||
skip_budget_checks = _is_model_cost_zero(
|
||||
model=model, llm_router=llm_router
|
||||
)
|
||||
if skip_budget_checks:
|
||||
verbose_proxy_logger.info(
|
||||
f"Skipping all budget checks for zero-cost model: {model}"
|
||||
)
|
||||
|
||||
# Check 3. Check if user is in their team budget
|
||||
if not skip_budget_checks and valid_token.team_member_spend is not None:
|
||||
if valid_token.team_member_spend is not None:
|
||||
if prisma_client is not None:
|
||||
_cache_key = f"{valid_token.team_id}_{valid_token.user_id}"
|
||||
|
||||
|
|
@ -1085,47 +1055,46 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
|
|||
param=abbreviate_api_key(api_key=api_key),
|
||||
)
|
||||
|
||||
if not skip_budget_checks:
|
||||
# Check 4. Token Spend is under budget
|
||||
if RouteChecks.is_llm_api_route(route=route):
|
||||
await _virtual_key_max_budget_check(
|
||||
valid_token=valid_token,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
user_obj=user_obj,
|
||||
)
|
||||
|
||||
# Check 5. Max Budget Alert Check
|
||||
await _virtual_key_max_budget_alert_check(
|
||||
# Check 4. Token Spend is under budget
|
||||
if RouteChecks.is_llm_api_route(route=route):
|
||||
await _virtual_key_max_budget_check(
|
||||
valid_token=valid_token,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
user_obj=user_obj,
|
||||
)
|
||||
|
||||
# Check 6. Soft Budget Check
|
||||
await _virtual_key_soft_budget_check(
|
||||
valid_token=valid_token,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
user_obj=user_obj,
|
||||
# Check 5. Max Budget Alert Check
|
||||
await _virtual_key_max_budget_alert_check(
|
||||
valid_token=valid_token,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
user_obj=user_obj,
|
||||
)
|
||||
|
||||
# Check 6. Soft Budget Check
|
||||
await _virtual_key_soft_budget_check(
|
||||
valid_token=valid_token,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
user_obj=user_obj,
|
||||
)
|
||||
|
||||
# Check 5. Token Model Spend is under Model budget
|
||||
max_budget_per_model = valid_token.model_max_budget
|
||||
current_model = request_data.get("model", None)
|
||||
|
||||
if (
|
||||
max_budget_per_model is not None
|
||||
and isinstance(max_budget_per_model, dict)
|
||||
and len(max_budget_per_model) > 0
|
||||
and prisma_client is not None
|
||||
and current_model is not None
|
||||
and valid_token.token is not None
|
||||
):
|
||||
## GET THE SPEND FOR THIS MODEL
|
||||
await model_max_budget_limiter.is_key_within_model_budget(
|
||||
user_api_key_dict=valid_token,
|
||||
model=current_model,
|
||||
)
|
||||
|
||||
# Check 5. Token Model Spend is under Model budget
|
||||
max_budget_per_model = valid_token.model_max_budget
|
||||
current_model = request_data.get("model", None)
|
||||
|
||||
if (
|
||||
max_budget_per_model is not None
|
||||
and isinstance(max_budget_per_model, dict)
|
||||
and len(max_budget_per_model) > 0
|
||||
and prisma_client is not None
|
||||
and current_model is not None
|
||||
and valid_token.token is not None
|
||||
):
|
||||
## GET THE SPEND FOR THIS MODEL
|
||||
await model_max_budget_limiter.is_key_within_model_budget(
|
||||
user_api_key_dict=valid_token,
|
||||
model=current_model,
|
||||
)
|
||||
|
||||
# Check 6: Additional Common Checks across jwt + key auth
|
||||
if valid_token.team_id is not None:
|
||||
_team_obj: Optional[LiteLLM_TeamTable] = LiteLLM_TeamTable(
|
||||
|
|
@ -1193,7 +1162,6 @@ async def _user_api_key_auth_builder( # noqa: PLR0915
|
|||
llm_router=llm_router,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
valid_token=valid_token,
|
||||
skip_budget_checks=skip_budget_checks,
|
||||
)
|
||||
# Token passed all checks
|
||||
if valid_token is None:
|
||||
|
|
@ -1335,6 +1303,8 @@ async def _return_user_api_key_auth_obj(
|
|||
user_tpm_limit=user_obj.tpm_limit,
|
||||
user_rpm_limit=user_obj.rpm_limit,
|
||||
user_email=user_obj.user_email,
|
||||
user_spend=getattr(user_obj, "spend", None),
|
||||
user_max_budget=getattr(user_obj, "max_budget", None),
|
||||
)
|
||||
if user_obj is not None and _is_user_proxy_admin(user_obj=user_obj):
|
||||
user_api_key_kwargs.update(
|
||||
|
|
|
|||
|
|
@ -50,8 +50,32 @@ from litellm.types.proxy.guardrails.guardrail_hooks.litellm_content_filter impor
|
|||
ContentFilterDetection,
|
||||
PatternDetection,
|
||||
)
|
||||
from .patterns import PATTERN_EXTRA_CONFIG, get_compiled_pattern
|
||||
|
||||
from .patterns import get_compiled_pattern
|
||||
MAX_KEYWORD_VALUE_GAP_WORDS = 1
|
||||
GAP_WORD_TOKENIZER = re.compile(r"\b\w+\b")
|
||||
|
||||
|
||||
WORD_NUMBER_MAP = {
|
||||
"zero": "0",
|
||||
"oh": "0",
|
||||
"one": "1",
|
||||
"two": "2",
|
||||
"three": "3",
|
||||
"four": "4",
|
||||
"five": "5",
|
||||
"six": "6",
|
||||
"seven": "7",
|
||||
"eight": "8",
|
||||
"nine": "9",
|
||||
}
|
||||
|
||||
WORD_NUMBER_TOKEN_REGEX = "|".join(WORD_NUMBER_MAP.keys())
|
||||
WORD_NUMBER_SEQUENCE_PATTERN = re.compile(
|
||||
rf"(?<![A-Za-z])(?:{WORD_NUMBER_TOKEN_REGEX})(?:[\s\-]+(?:{WORD_NUMBER_TOKEN_REGEX}))+(?![A-Za-z])",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
WORD_NUMBER_TOKEN_FINDER = re.compile(rf"(?:{WORD_NUMBER_TOKEN_REGEX})", re.IGNORECASE)
|
||||
|
||||
|
||||
# Helper data structure for category-based detection
|
||||
|
|
@ -144,9 +168,9 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
self.image_model = image_model
|
||||
# Store loaded categories
|
||||
self.loaded_categories: Dict[str, CategoryConfig] = {}
|
||||
self.category_keywords: Dict[str, Tuple[str, str, ContentFilterAction]] = (
|
||||
{}
|
||||
) # keyword -> (category, severity, action)
|
||||
self.category_keywords: Dict[
|
||||
str, Tuple[str, str, ContentFilterAction]
|
||||
] = {} # keyword -> (category, severity, action)
|
||||
|
||||
# Load categories if provided
|
||||
if categories:
|
||||
|
|
@ -170,7 +194,7 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
normalized_blocked_words.append(word)
|
||||
|
||||
# Compile regex patterns
|
||||
self.compiled_patterns: List[Tuple[Pattern, str, ContentFilterAction]] = []
|
||||
self.compiled_patterns: List[Dict[str, Any]] = []
|
||||
for pattern_config in normalized_patterns:
|
||||
self._add_pattern(pattern_config)
|
||||
|
||||
|
|
@ -323,11 +347,13 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
pattern_config: ContentFilterPattern configuration
|
||||
"""
|
||||
try:
|
||||
extra_config: Dict[str, Any] = {}
|
||||
if pattern_config.pattern_type == "prebuilt":
|
||||
if not pattern_config.pattern_name:
|
||||
raise ValueError("pattern_name is required for prebuilt patterns")
|
||||
compiled = get_compiled_pattern(pattern_config.pattern_name)
|
||||
pattern_name = pattern_config.pattern_name
|
||||
extra_config = PATTERN_EXTRA_CONFIG.get(pattern_name, {}) or {}
|
||||
elif pattern_config.pattern_type == "regex":
|
||||
if not pattern_config.pattern:
|
||||
raise ValueError("pattern is required for regex patterns")
|
||||
|
|
@ -336,8 +362,20 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
else:
|
||||
raise ValueError(f"Unknown pattern_type: {pattern_config.pattern_type}")
|
||||
|
||||
keyword_regex: Optional[Pattern] = None
|
||||
if extra_config.get("keyword_pattern"):
|
||||
keyword_regex = re.compile(
|
||||
extra_config["keyword_pattern"], re.IGNORECASE
|
||||
)
|
||||
|
||||
self.compiled_patterns.append(
|
||||
(compiled, pattern_name, pattern_config.action)
|
||||
{
|
||||
"regex": compiled,
|
||||
"pattern_name": pattern_name,
|
||||
"action": pattern_config.action,
|
||||
"keyword_regex": keyword_regex,
|
||||
"allow_word_numbers": bool(extra_config.get("allow_word_numbers")),
|
||||
}
|
||||
)
|
||||
verbose_proxy_logger.debug(
|
||||
f"Added pattern: {pattern_name} with action {pattern_config.action}"
|
||||
|
|
@ -395,6 +433,130 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
except Exception as e:
|
||||
raise Exception(f"Error loading blocked words file {file_path}: {str(e)}")
|
||||
|
||||
def _find_pattern_spans(
|
||||
self, text: str, pattern_entry: Dict[str, Any]
|
||||
) -> List[Tuple[int, int]]:
|
||||
"""Return all match spans for a pattern, applying contextual rules if required."""
|
||||
|
||||
regex: Pattern = pattern_entry["regex"]
|
||||
keyword_regex: Optional[Pattern] = pattern_entry.get("keyword_regex")
|
||||
allow_word_numbers: bool = pattern_entry.get("allow_word_numbers", False)
|
||||
|
||||
keyword_matches: Optional[List[re.Match]] = None
|
||||
if keyword_regex is not None:
|
||||
keyword_matches = list(keyword_regex.finditer(text))
|
||||
if not keyword_matches:
|
||||
return []
|
||||
|
||||
match_spans: List[Tuple[int, int]] = []
|
||||
|
||||
for match in regex.finditer(text):
|
||||
if keyword_matches is not None and not self._match_near_keyword(
|
||||
match.start(), match.end(), keyword_matches, text
|
||||
):
|
||||
continue
|
||||
match_spans.append((match.start(), match.end()))
|
||||
|
||||
if allow_word_numbers:
|
||||
for word_match in WORD_NUMBER_SEQUENCE_PATTERN.finditer(text):
|
||||
digits = self._convert_word_number_sequence(word_match.group())
|
||||
if not digits:
|
||||
continue
|
||||
if not regex.fullmatch(digits):
|
||||
continue
|
||||
if keyword_matches is not None and not self._match_near_keyword(
|
||||
word_match.start(), word_match.end(), keyword_matches, text
|
||||
):
|
||||
continue
|
||||
match_spans.append((word_match.start(), word_match.end()))
|
||||
|
||||
return self._merge_spans(match_spans)
|
||||
|
||||
def _match_near_keyword(
|
||||
self,
|
||||
value_start: int,
|
||||
value_end: int,
|
||||
keyword_matches: List[re.Match],
|
||||
text: str,
|
||||
) -> bool:
|
||||
"""Check if a value is separated from a keyword by an allowed gap."""
|
||||
|
||||
for keyword_match in keyword_matches:
|
||||
keyword_start = keyword_match.start()
|
||||
keyword_end = keyword_match.end()
|
||||
|
||||
if value_start >= keyword_end:
|
||||
gap_text = text[keyword_end:value_start]
|
||||
elif keyword_start >= value_end:
|
||||
gap_text = text[value_end:keyword_start]
|
||||
else:
|
||||
return True # overlapping
|
||||
|
||||
if self._gap_text_allowed(gap_text):
|
||||
return True
|
||||
return False
|
||||
|
||||
def _gap_text_allowed(self, gap_text: str) -> bool:
|
||||
"""Return True if the gap between keyword and value meets word-count rules."""
|
||||
|
||||
if not gap_text.strip():
|
||||
return True
|
||||
if any(char.isdigit() for char in gap_text):
|
||||
return False
|
||||
|
||||
words = GAP_WORD_TOKENIZER.findall(gap_text)
|
||||
return len(words) <= MAX_KEYWORD_VALUE_GAP_WORDS
|
||||
|
||||
def _merge_spans(self, spans: List[Tuple[int, int]]) -> List[Tuple[int, int]]:
|
||||
"""Merge overlapping spans to avoid double-masking."""
|
||||
|
||||
if not spans:
|
||||
return []
|
||||
|
||||
spans.sort(key=lambda item: item[0])
|
||||
merged: List[Tuple[int, int]] = [spans[0]]
|
||||
|
||||
for start, end in spans[1:]:
|
||||
last_start, last_end = merged[-1]
|
||||
if start <= last_end:
|
||||
merged[-1] = (last_start, max(last_end, end))
|
||||
else:
|
||||
merged.append((start, end))
|
||||
return merged
|
||||
|
||||
def _mask_spans(
|
||||
self, text: str, spans: List[Tuple[int, int]], redaction: str
|
||||
) -> str:
|
||||
"""Apply masking for the provided spans using the given redaction tag."""
|
||||
|
||||
if not spans:
|
||||
return text
|
||||
|
||||
result_parts: List[str] = []
|
||||
previous_end = 0
|
||||
for start, end in spans:
|
||||
result_parts.append(text[previous_end:start])
|
||||
result_parts.append(redaction)
|
||||
previous_end = end
|
||||
result_parts.append(text[previous_end:])
|
||||
return "".join(result_parts)
|
||||
|
||||
def _convert_word_number_sequence(self, sequence: str) -> Optional[str]:
|
||||
"""Convert a spelled-out digit sequence (e.g., 'One-Two') into digits."""
|
||||
|
||||
tokens = WORD_NUMBER_TOKEN_FINDER.findall(sequence)
|
||||
if not tokens:
|
||||
return None
|
||||
|
||||
digits: List[str] = []
|
||||
for token in tokens:
|
||||
digit = WORD_NUMBER_MAP.get(token.lower())
|
||||
if digit is None:
|
||||
return None
|
||||
digits.append(digit)
|
||||
|
||||
return "".join(digits) if digits else None
|
||||
|
||||
def _check_patterns(
|
||||
self, text: str
|
||||
) -> Optional[Tuple[str, str, ContentFilterAction]]:
|
||||
|
|
@ -407,10 +569,13 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
Returns:
|
||||
Tuple of (matched_text, pattern_name, action) if match found, None otherwise
|
||||
"""
|
||||
for compiled_pattern, pattern_name, action in self.compiled_patterns:
|
||||
match = compiled_pattern.search(text)
|
||||
if match:
|
||||
matched_text = match.group(0)
|
||||
for pattern_entry in self.compiled_patterns:
|
||||
spans = self._find_pattern_spans(text, pattern_entry)
|
||||
if spans:
|
||||
start, end = spans[0]
|
||||
matched_text = text[start:end]
|
||||
pattern_name = pattern_entry["pattern_name"]
|
||||
action = pattern_entry["action"]
|
||||
verbose_proxy_logger.debug(
|
||||
f"Pattern '{pattern_name}' matched: {matched_text[:20]}..."
|
||||
)
|
||||
|
|
@ -582,11 +747,13 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
)
|
||||
|
||||
# Check regex patterns - process ALL patterns, not just first match
|
||||
for compiled_pattern, pattern_name, action in self.compiled_patterns:
|
||||
match = compiled_pattern.search(text)
|
||||
if not match:
|
||||
for pattern_entry in self.compiled_patterns:
|
||||
spans = self._find_pattern_spans(text, pattern_entry)
|
||||
if not spans:
|
||||
continue
|
||||
|
||||
pattern_name = pattern_entry["pattern_name"]
|
||||
action = pattern_entry["action"]
|
||||
if detections is not None:
|
||||
# Don't log matched_text to avoid exposing sensitive content (emails, credit cards, etc.)
|
||||
pattern_detection: PatternDetection = {
|
||||
|
|
@ -604,11 +771,10 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
detail={"error": error_msg, "pattern": pattern_name},
|
||||
)
|
||||
elif action == ContentFilterAction.MASK:
|
||||
# Replace ALL matches of this pattern with redaction tag
|
||||
redaction_tag = self.pattern_redaction_format.format(
|
||||
pattern_name=pattern_name.upper()
|
||||
)
|
||||
text = compiled_pattern.sub(redaction_tag, text)
|
||||
text = self._mask_spans(text, spans, redaction_tag)
|
||||
verbose_proxy_logger.info(
|
||||
f"Masked all {pattern_name} matches in content"
|
||||
)
|
||||
|
|
@ -924,19 +1090,28 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
if pattern_match:
|
||||
matched_text, pattern_name, action = pattern_match
|
||||
if action == ContentFilterAction.BLOCK:
|
||||
error_msg = f"Content blocked: {pattern_name} pattern detected"
|
||||
error_msg = (
|
||||
f"Content blocked: {pattern_name} pattern detected"
|
||||
)
|
||||
verbose_proxy_logger.warning(error_msg)
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail={"error": error_msg, "pattern": pattern_name},
|
||||
detail={
|
||||
"error": error_msg,
|
||||
"pattern": pattern_name,
|
||||
},
|
||||
)
|
||||
|
||||
# Check blocked words
|
||||
blocked_word_match = self._check_blocked_words(accumulated_content)
|
||||
blocked_word_match = self._check_blocked_words(
|
||||
accumulated_content
|
||||
)
|
||||
if blocked_word_match:
|
||||
keyword, action, description = blocked_word_match
|
||||
if action == ContentFilterAction.BLOCK:
|
||||
error_msg = f"Content blocked: keyword '{keyword}' detected"
|
||||
error_msg = (
|
||||
f"Content blocked: keyword '{keyword}' detected"
|
||||
)
|
||||
if description:
|
||||
error_msg += f" ({description})"
|
||||
verbose_proxy_logger.warning(error_msg)
|
||||
|
|
|
|||
|
|
@ -120,11 +120,11 @@
|
|||
"description": "Detects URLs (http/https)"
|
||||
},
|
||||
{
|
||||
"name": "passport_us",
|
||||
"display_name": "Passport (US)",
|
||||
"pattern": "\\b[0-9]{9}\\b",
|
||||
"category": "PII Patterns",
|
||||
"description": "US passport numbers (9 digits)"
|
||||
"name": "passport_us",
|
||||
"display_name": "Passport (US)",
|
||||
"pattern": "\\b[0-9]{9}\\b",
|
||||
"category": "PII Patterns",
|
||||
"description": "US passport numbers (9 digits)"
|
||||
},
|
||||
{
|
||||
"name": "passport_uk",
|
||||
|
|
@ -203,7 +203,6 @@
|
|||
"category": "Protected Class - Fair Lending",
|
||||
"description": "Detects race, ethnicity and national origin terms - protected under ECOA and Fair Housing Act"
|
||||
},
|
||||
|
||||
{
|
||||
"name": "religion",
|
||||
"display_name": "Religion & Creed (Protected Class)",
|
||||
|
|
@ -236,7 +235,7 @@
|
|||
"name": "military_status",
|
||||
"display_name": "Military Status (Protected Class)",
|
||||
"pattern": "\\b(veteran|military|armed\\s+forces|army|navy|air\\s+force|marine(s|\\s+corps)?|coast\\s+guard|national\\s+guard|reserve(s|ist)?|active\\s+duty|deployment|deployed|enlisted|commissioned|honorable\\s+discharge|dishonorable\\s+discharge|VA\\s+benefits|GI\\s+bill|military\\s+service|service\\s+member|servicemember|SCRA|MLA|military\\s+lending)\\b",
|
||||
"category": "Protected Class - Fair Lending",
|
||||
"category": "Protected Class - Fair Lending",
|
||||
"description": "Detects military status terms - protected under SCRA and MLA"
|
||||
},
|
||||
{
|
||||
|
|
@ -245,7 +244,7 @@
|
|||
"pattern": "\\b(welfare|public\\s+assistance|food\\s+stamps|SNAP|WIC|TANF|medicaid|section\\s+8|housing\\s+voucher|subsidized\\s+housing|public\\s+housing|government\\s+benefits|social\\s+services|unemployment\\s+(benefits|insurance)|UI\\s+benefits|EBT|benefit\\s+recipient)\\b",
|
||||
"category": "Protected Class - Fair Lending",
|
||||
"description": "Detects public assistance terms - protected under ECOA"
|
||||
} ,
|
||||
},
|
||||
{
|
||||
"name": "weapons_firearms",
|
||||
"display_name": "Weapons & Firearms",
|
||||
|
|
@ -313,10 +312,12 @@
|
|||
{
|
||||
"name": "nl_bsn_contextual",
|
||||
"display_name": "BSN (Dutch Citizen Service Number)",
|
||||
"pattern": "\\b(?:BSN|B\\.S\\.N\\.|burgerservicenummer|burger\\s*service\\s*nummer|sofi\\s*nummer|sofinummer|persoonsnummer|identificatienummer|citizen\\s*service\\s*number)[:\\s]*[0-9]{9}\\b|\\b[0-9]{9}\\b(?=\\s*(?:BSN|burgerservicenummer|sofinummer))",
|
||||
"pattern": "\\b[0-9]{9}\\b",
|
||||
"category": "PII Patterns",
|
||||
"action": "MASK",
|
||||
"description": "Detects Dutch BSN numbers with contextual keywords"
|
||||
"description": "Detects Dutch BSN numbers with contextual keywords",
|
||||
"keyword_pattern": "(?:\\b(?:BSN|B\\.S\\.N\\.|burgerservicenummer|burger\\s*service\\s*nummer|sofi\\s*nummer|sofinummer|persoonsnummer|identificatienummer|citizen\\s*service\\s*number)\\b|8\\s*5\\s*\\|\\\\\\|)",
|
||||
"allow_word_numbers": true
|
||||
},
|
||||
{
|
||||
"name": "br_cpf",
|
||||
|
|
@ -369,5 +370,3 @@
|
|||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ import json
|
|||
import os
|
||||
import re
|
||||
from enum import Enum
|
||||
from typing import Dict, List, Pattern
|
||||
from typing import Any, Dict, List, Pattern
|
||||
|
||||
|
||||
def _load_patterns_from_json() -> Dict:
|
||||
|
|
@ -41,6 +41,26 @@ PREBUILT_PATTERNS: Dict[str, str] = {
|
|||
}
|
||||
|
||||
|
||||
# Capture any extra configuration declared per pattern (e.g., contextual keywords)
|
||||
KNOWN_PATTERN_KEYS = {
|
||||
"name",
|
||||
"display_name",
|
||||
"pattern",
|
||||
"category",
|
||||
"action",
|
||||
"description",
|
||||
}
|
||||
|
||||
PATTERN_EXTRA_CONFIG: Dict[str, Dict[str, Any]] = {}
|
||||
for pattern_data in _PATTERNS_DATA["patterns"]:
|
||||
extra_config = {
|
||||
key: value
|
||||
for key, value in pattern_data.items()
|
||||
if key not in KNOWN_PATTERN_KEYS
|
||||
}
|
||||
PATTERN_EXTRA_CONFIG[pattern_data["name"]] = extra_config
|
||||
|
||||
|
||||
def get_compiled_pattern(pattern_name: str) -> Pattern:
|
||||
"""
|
||||
Get a compiled regex pattern by name.
|
||||
|
|
|
|||
|
|
@ -79,8 +79,12 @@ class UnifiedLLMGuardrails(CustomLogger):
|
|||
endpoint_guardrail_translation_mappings = (
|
||||
load_guardrail_translation_mappings()
|
||||
)
|
||||
if CallTypes(call_type) not in endpoint_guardrail_translation_mappings:
|
||||
return data
|
||||
|
||||
try:
|
||||
if CallTypes(call_type) not in endpoint_guardrail_translation_mappings:
|
||||
return data
|
||||
except ValueError:
|
||||
return data # handle unmapped call types
|
||||
|
||||
endpoint_translation = endpoint_guardrail_translation_mappings[
|
||||
CallTypes(call_type)
|
||||
|
|
|
|||
|
|
@ -114,25 +114,25 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
) -> Optional[str]:
|
||||
"""
|
||||
Get priority from user_api_key_dict.
|
||||
|
||||
|
||||
Checks team metadata first (takes precedence), then falls back to key metadata.
|
||||
|
||||
|
||||
Args:
|
||||
user_api_key_dict: User authentication info
|
||||
|
||||
|
||||
Returns:
|
||||
Priority string if found, None otherwise
|
||||
"""
|
||||
priority: Optional[str] = None
|
||||
|
||||
|
||||
# Check team metadata first (takes precedence)
|
||||
if user_api_key_dict.team_metadata is not None:
|
||||
priority = user_api_key_dict.team_metadata.get("priority", None)
|
||||
|
||||
|
||||
# Fall back to key metadata
|
||||
if priority is None:
|
||||
priority = user_api_key_dict.metadata.get("priority", None)
|
||||
|
||||
|
||||
return priority
|
||||
|
||||
def _normalize_priority_weights(
|
||||
|
|
@ -299,10 +299,13 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
"""
|
||||
descriptors: List[RateLimitDescriptor] = []
|
||||
|
||||
if litellm.priority_reservation is None:
|
||||
return descriptors
|
||||
|
||||
# Get model group info
|
||||
model_group_info: Optional[ModelGroupInfo] = (
|
||||
self.llm_router.get_model_group_info(model_group=model)
|
||||
)
|
||||
model_group_info: Optional[
|
||||
ModelGroupInfo
|
||||
] = self.llm_router.get_model_group_info(model_group=model)
|
||||
if model_group_info is None:
|
||||
return descriptors
|
||||
|
||||
|
|
@ -577,9 +580,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
)
|
||||
|
||||
# Get model configuration
|
||||
model_group_info: Optional[ModelGroupInfo] = (
|
||||
self.llm_router.get_model_group_info(model_group=model)
|
||||
)
|
||||
model_group_info: Optional[
|
||||
ModelGroupInfo
|
||||
] = self.llm_router.get_model_group_info(model_group=model)
|
||||
if model_group_info is None:
|
||||
verbose_proxy_logger.debug(
|
||||
f"No model group info for {model}, allowing request"
|
||||
|
|
@ -703,7 +706,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
|
||||
# Get priority from user_api_key_auth_metadata in standard_logging_metadata
|
||||
# This is where user_api_key_dict.metadata is stored during pre-call
|
||||
user_api_key_auth_metadata = standard_logging_metadata.get("user_api_key_auth_metadata") or {}
|
||||
user_api_key_auth_metadata = (
|
||||
standard_logging_metadata.get("user_api_key_auth_metadata") or {}
|
||||
)
|
||||
key_priority: Optional[str] = user_api_key_auth_metadata.get("priority")
|
||||
|
||||
# Get total tokens from response
|
||||
|
|
@ -775,7 +780,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
|
||||
# Only log 'priority' if it's known safe; otherwise, redact.
|
||||
SAFE_PRIORITIES = {"low", "medium", "high", "default"}
|
||||
logged_priority = key_priority if key_priority in SAFE_PRIORITIES else "REDACTED"
|
||||
logged_priority = (
|
||||
key_priority if key_priority in SAFE_PRIORITIES else "REDACTED"
|
||||
)
|
||||
verbose_proxy_logger.debug(
|
||||
f"[Dynamic Rate Limiter] Incremented tokens by {total_tokens} for "
|
||||
f"model={model_group}, priority={logged_priority}"
|
||||
|
|
|
|||
|
|
@ -1236,7 +1236,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
return pipeline_operations
|
||||
|
||||
def _get_total_tokens_from_usage(
|
||||
self, usage: Any | None, rate_limit_type: Literal["output", "input", "total"]
|
||||
self, usage: Optional[Any], rate_limit_type: Literal["output", "input", "total"]
|
||||
) -> int:
|
||||
"""
|
||||
Get total tokens from response usage for rate limiting.
|
||||
|
|
|
|||
|
|
@ -1000,6 +1000,13 @@ async def add_litellm_data_to_request( # noqa: PLR0915
|
|||
"user_api_key_model_max_budget"
|
||||
] = user_api_key_dict.model_max_budget
|
||||
|
||||
# User spend, budget - used by prometheus.py
|
||||
# Follow same pattern as team and API key budgets
|
||||
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
|
||||
data[_metadata_variable_name][
|
||||
"user_api_key_user_max_budget"
|
||||
] = user_api_key_dict.user_max_budget
|
||||
|
||||
data[_metadata_variable_name]["user_api_key_metadata"] = user_api_key_dict.metadata
|
||||
_headers = dict(request.headers)
|
||||
_headers.pop(
|
||||
|
|
|
|||
|
|
@ -343,7 +343,7 @@ def _build_where_conditions(
|
|||
start_date: str,
|
||||
end_date: str,
|
||||
model: Optional[str],
|
||||
api_key: Optional[Union[str, List[str]]],
|
||||
api_key: Optional[str],
|
||||
exclude_entity_ids: Optional[List[str]] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Build prisma where clause for daily activity queries."""
|
||||
|
|
@ -357,10 +357,7 @@ def _build_where_conditions(
|
|||
if model:
|
||||
where_conditions["model"] = model
|
||||
if api_key:
|
||||
if isinstance(api_key, list):
|
||||
where_conditions["api_key"] = {"in": api_key}
|
||||
else:
|
||||
where_conditions["api_key"] = api_key
|
||||
where_conditions["api_key"] = api_key
|
||||
|
||||
if entity_id is not None:
|
||||
if isinstance(entity_id, list):
|
||||
|
|
@ -448,7 +445,7 @@ async def get_daily_activity(
|
|||
start_date: Optional[str],
|
||||
end_date: Optional[str],
|
||||
model: Optional[str],
|
||||
api_key: Optional[Union[str, List[str]]],
|
||||
api_key: Optional[str],
|
||||
page: int,
|
||||
page_size: int,
|
||||
exclude_entity_ids: Optional[List[str]] = None,
|
||||
|
|
|
|||
|
|
@ -412,6 +412,13 @@ async def new_user(
|
|||
status_code=403,
|
||||
detail="License is over limit. Please contact support@berri.ai to upgrade your license.",
|
||||
)
|
||||
|
||||
# Only proxy admins can create administrative users
|
||||
if data.user_role in [LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY] and user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN:
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail=f"Only proxy admins can create administrative users (proxy_admin, proxy_admin_viewer). Attempted to create user with role: {data.user_role}. Your role: {user_api_key_dict.user_role}"
|
||||
)
|
||||
|
||||
data_json = data.json() # type: ignore
|
||||
data_json = _update_internal_new_user_params(data_json, data)
|
||||
|
|
|
|||
|
|
@ -3601,7 +3601,7 @@ async def get_team_daily_activity(
|
|||
},
|
||||
)
|
||||
|
||||
## Fetch team aliases and check team admin status
|
||||
## Fetch team aliases
|
||||
where_condition = {}
|
||||
if team_ids_list:
|
||||
where_condition["team_id"] = {"in": list(team_ids_list)}
|
||||
|
|
@ -3612,36 +3612,6 @@ async def get_team_daily_activity(
|
|||
t.team_id: {"team_alias": t.team_alias} for t in team_aliases
|
||||
}
|
||||
|
||||
# Check if user is team admin for any requested teams
|
||||
# If not, filter by user's API keys
|
||||
user_api_keys: Optional[List[str]] = None
|
||||
if not _user_has_admin_view(user_api_key_dict) and team_ids_list and team_aliases:
|
||||
# Check if user is team admin for any of the teams
|
||||
is_team_admin_for_any = False
|
||||
for team_alias in team_aliases:
|
||||
team_obj = LiteLLM_TeamTable(**team_alias.model_dump())
|
||||
if _is_user_team_admin(
|
||||
user_api_key_dict=user_api_key_dict, team_obj=team_obj
|
||||
):
|
||||
is_team_admin_for_any = True
|
||||
break
|
||||
|
||||
# If user is not a team admin for any team, filter by their API keys
|
||||
if not is_team_admin_for_any:
|
||||
# Get all API keys for this user
|
||||
user_keys = await prisma_client.db.litellm_verificationtoken.find_many(
|
||||
where={"user_id": user_api_key_dict.user_id}
|
||||
)
|
||||
user_api_keys = [key.token for key in user_keys if key.token]
|
||||
# If user has no API keys, return empty result
|
||||
if not user_api_keys:
|
||||
user_api_keys = [""] # Use empty string to ensure no matches
|
||||
|
||||
# If api_key parameter is provided, use it; otherwise use user_api_keys if set
|
||||
final_api_key_filter: Optional[Union[str, List[str]]] = api_key
|
||||
if final_api_key_filter is None and user_api_keys is not None:
|
||||
final_api_key_filter = user_api_keys
|
||||
|
||||
return await get_daily_activity(
|
||||
prisma_client=prisma_client,
|
||||
table_name="litellm_dailyteamspend",
|
||||
|
|
@ -3652,7 +3622,7 @@ async def get_team_daily_activity(
|
|||
start_date=start_date,
|
||||
end_date=end_date,
|
||||
model=model,
|
||||
api_key=final_api_key_filter,
|
||||
api_key=api_key,
|
||||
page=page,
|
||||
page_size=page_size,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -761,7 +761,6 @@ async def handle_bedrock_passthrough_router_model(
|
|||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
|
||||
|
||||
async def handle_bedrock_count_tokens(
|
||||
endpoint: str,
|
||||
request: Request,
|
||||
|
|
@ -1555,6 +1554,7 @@ async def _base_vertex_proxy_route(
|
|||
from litellm.llms.vertex_ai.common_utils import (
|
||||
construct_target_url,
|
||||
get_vertex_location_from_url,
|
||||
get_vertex_model_id_from_url,
|
||||
get_vertex_project_id_from_url,
|
||||
)
|
||||
|
||||
|
|
@ -1584,6 +1584,25 @@ async def _base_vertex_proxy_route(
|
|||
vertex_location=vertex_location,
|
||||
)
|
||||
|
||||
if vertex_project is None or vertex_location is None:
|
||||
# Check if model is in router config
|
||||
model_id = get_vertex_model_id_from_url(endpoint)
|
||||
if model_id:
|
||||
from litellm.proxy.proxy_server import llm_router
|
||||
|
||||
if llm_router:
|
||||
try:
|
||||
# Use the dedicated pass-through deployment selection method to automatically filter use_in_pass_through=True
|
||||
deployment = llm_router.get_available_deployment_for_pass_through(model=model_id)
|
||||
if deployment:
|
||||
litellm_params = deployment.get("litellm_params", {})
|
||||
vertex_project = litellm_params.get("vertex_project")
|
||||
vertex_location = litellm_params.get("vertex_location")
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.debug(
|
||||
f"Error getting available deployment for model {model_id}: {e}"
|
||||
)
|
||||
|
||||
vertex_credentials = passthrough_endpoint_router.get_vertex_credentials(
|
||||
project_id=vertex_project,
|
||||
location=vertex_location,
|
||||
|
|
|
|||
|
|
@ -37,6 +37,12 @@ model_list:
|
|||
model_info:
|
||||
litellm_provider: bedrock_converse
|
||||
mode: chat
|
||||
- model_name: azure-claude-opus-4-5
|
||||
litellm_params:
|
||||
model: azure_ai/claude-opus-4-5
|
||||
api_base: https://krish-mh44t553-eastus2.services.ai.azure.com
|
||||
api_key: os.environ/AZURE_ANTHROPIC_API_KEY
|
||||
|
||||
|
||||
general_settings:
|
||||
store_prompts_in_spend_logs: true
|
||||
|
|
|
|||
|
|
@ -3253,20 +3253,22 @@ class ProxyConfig:
|
|||
) -> Optional[dict]:
|
||||
"""
|
||||
Get router_settings in priority order: Key > Team > Global
|
||||
|
||||
|
||||
Returns:
|
||||
dict: Combined router_settings, or None if no settings found
|
||||
"""
|
||||
if prisma_client is None:
|
||||
return None
|
||||
|
||||
|
||||
import json
|
||||
import yaml
|
||||
|
||||
|
||||
# 1. Try key-level router_settings
|
||||
if user_api_key_dict is not None:
|
||||
# Check if router_settings is available on the key object
|
||||
key_router_settings_value = getattr(user_api_key_dict, "router_settings", None)
|
||||
key_router_settings_value = getattr(
|
||||
user_api_key_dict, "router_settings", None
|
||||
)
|
||||
if key_router_settings_value is not None:
|
||||
key_router_settings = None
|
||||
if isinstance(key_router_settings_value, str):
|
||||
|
|
@ -3279,11 +3281,15 @@ class ProxyConfig:
|
|||
pass
|
||||
elif isinstance(key_router_settings_value, dict):
|
||||
key_router_settings = key_router_settings_value
|
||||
|
||||
|
||||
# If key has router_settings (non-empty dict), use it
|
||||
if key_router_settings is not None and isinstance(key_router_settings, dict) and key_router_settings:
|
||||
if (
|
||||
key_router_settings is not None
|
||||
and isinstance(key_router_settings, dict)
|
||||
and key_router_settings
|
||||
):
|
||||
return key_router_settings
|
||||
|
||||
|
||||
# 2. Try team-level router_settings
|
||||
if user_api_key_dict is not None and user_api_key_dict.team_id is not None:
|
||||
try:
|
||||
|
|
@ -3291,37 +3297,51 @@ class ProxyConfig:
|
|||
where={"team_id": user_api_key_dict.team_id}
|
||||
)
|
||||
if team_obj is not None:
|
||||
team_router_settings_value = getattr(team_obj, "router_settings", None)
|
||||
team_router_settings_value = getattr(
|
||||
team_obj, "router_settings", None
|
||||
)
|
||||
if team_router_settings_value is not None:
|
||||
team_router_settings = None
|
||||
if isinstance(team_router_settings_value, str):
|
||||
try:
|
||||
team_router_settings = yaml.safe_load(team_router_settings_value)
|
||||
team_router_settings = yaml.safe_load(
|
||||
team_router_settings_value
|
||||
)
|
||||
except (yaml.YAMLError, json.JSONDecodeError):
|
||||
try:
|
||||
team_router_settings = json.loads(team_router_settings_value)
|
||||
team_router_settings = json.loads(
|
||||
team_router_settings_value
|
||||
)
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
elif isinstance(team_router_settings_value, dict):
|
||||
team_router_settings = team_router_settings_value
|
||||
|
||||
|
||||
# If team has router_settings (non-empty dict), use it
|
||||
if team_router_settings is not None and isinstance(team_router_settings, dict) and team_router_settings:
|
||||
if (
|
||||
team_router_settings is not None
|
||||
and isinstance(team_router_settings, dict)
|
||||
and team_router_settings
|
||||
):
|
||||
return team_router_settings
|
||||
except Exception:
|
||||
# If team lookup fails, continue to global settings
|
||||
pass
|
||||
|
||||
|
||||
# 3. Try global router_settings
|
||||
try:
|
||||
db_router_settings = await prisma_client.db.litellm_config.find_first(
|
||||
where={"param_name": "router_settings"}
|
||||
)
|
||||
if db_router_settings is not None and isinstance(db_router_settings.param_value, dict) and db_router_settings.param_value:
|
||||
if (
|
||||
db_router_settings is not None
|
||||
and isinstance(db_router_settings.param_value, dict)
|
||||
and db_router_settings.param_value
|
||||
):
|
||||
return db_router_settings.param_value
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
return None
|
||||
|
||||
async def _add_router_settings_from_db_config(
|
||||
|
|
@ -4688,27 +4708,48 @@ class ProxyStartupEvent:
|
|||
### SPEND LOG CLEANUP ###
|
||||
if general_settings.get("maximum_spend_logs_retention_period") is not None:
|
||||
spend_log_cleanup = SpendLogCleanup()
|
||||
# Get the interval from config or default to 1 day
|
||||
retention_interval = general_settings.get(
|
||||
"maximum_spend_logs_retention_interval", "1d"
|
||||
)
|
||||
try:
|
||||
interval_seconds = duration_in_seconds(retention_interval)
|
||||
scheduler.add_job(
|
||||
spend_log_cleanup.cleanup_old_spend_logs,
|
||||
"interval",
|
||||
seconds=interval_seconds
|
||||
+ random.randint(0, 60), # Add small random offset
|
||||
# REMOVED jitter parameter - major cause of memory leak
|
||||
args=[prisma_client],
|
||||
id="spend_log_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.error(
|
||||
"Invalid maximum_spend_logs_retention_interval value"
|
||||
cleanup_cron = general_settings.get("maximum_spend_logs_cleanup_cron")
|
||||
|
||||
if cleanup_cron:
|
||||
from apscheduler.triggers.cron import CronTrigger
|
||||
|
||||
try:
|
||||
cron_trigger = CronTrigger.from_crontab(cleanup_cron)
|
||||
scheduler.add_job(
|
||||
spend_log_cleanup.cleanup_old_spend_logs,
|
||||
cron_trigger,
|
||||
args=[prisma_client],
|
||||
id="spend_log_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
verbose_proxy_logger.info(
|
||||
f"Spend log cleanup scheduled with cron: {cleanup_cron}"
|
||||
)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.error(
|
||||
f"Invalid maximum_spend_logs_cleanup_cron value: {cleanup_cron}"
|
||||
)
|
||||
else:
|
||||
# Interval-based scheduling (existing behavior)
|
||||
retention_interval = general_settings.get(
|
||||
"maximum_spend_logs_retention_interval", "1d"
|
||||
)
|
||||
try:
|
||||
interval_seconds = duration_in_seconds(retention_interval)
|
||||
scheduler.add_job(
|
||||
spend_log_cleanup.cleanup_old_spend_logs,
|
||||
"interval",
|
||||
seconds=interval_seconds + random.randint(0, 60),
|
||||
args=[prisma_client],
|
||||
id="spend_log_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.error(
|
||||
"Invalid maximum_spend_logs_retention_interval value"
|
||||
)
|
||||
### CHECK BATCH COST ###
|
||||
if llm_router is not None:
|
||||
try:
|
||||
|
|
@ -9922,7 +9963,9 @@ async def get_config(): # noqa: PLR0915
|
|||
|
||||
_success_callbacks = normalize_callback(_success_callbacks)
|
||||
_failure_callbacks = normalize_callback(_failure_callbacks)
|
||||
_success_and_failure_callbacks = normalize_callback(_success_and_failure_callbacks)
|
||||
_success_and_failure_callbacks = normalize_callback(
|
||||
_success_and_failure_callbacks
|
||||
)
|
||||
|
||||
_data_to_return = []
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -72,6 +72,11 @@ class UISettings(BaseModel):
|
|||
description="If true, internal users cannot add models from the UI",
|
||||
)
|
||||
|
||||
disable_team_admin_delete_team_user: bool = Field(
|
||||
default=False,
|
||||
description="Prevents Team Admins from deleting users from the teams they manage. Useful for SCIM provisioning where team membership is defined externally.",
|
||||
)
|
||||
|
||||
|
||||
class UISettingsResponse(SettingsResponse):
|
||||
"""Response model for UI settings"""
|
||||
|
|
@ -80,7 +85,7 @@ class UISettingsResponse(SettingsResponse):
|
|||
|
||||
|
||||
# Allowlist of UI settings that can be stored
|
||||
ALLOWED_UI_SETTINGS_FIELDS = {"disable_model_add_for_internal_users"}
|
||||
ALLOWED_UI_SETTINGS_FIELDS = {"disable_model_add_for_internal_users", "disable_team_admin_delete_team_user"}
|
||||
|
||||
|
||||
@router.get(
|
||||
|
|
|
|||
|
|
@ -61,6 +61,10 @@ async def _arealtime(
|
|||
api_key=api_key,
|
||||
)
|
||||
|
||||
# Ensure query params use the normalized provider model (no proxy aliases).
|
||||
if query_params is not None:
|
||||
query_params = {**query_params, "model": model}
|
||||
|
||||
litellm_logging_obj.update_environment_variables(
|
||||
model=model,
|
||||
user=user,
|
||||
|
|
|
|||
|
|
@ -2,127 +2,67 @@
|
|||
|
||||
from typing import (
|
||||
Any,
|
||||
Awaitable,
|
||||
Callable,
|
||||
Dict,
|
||||
Iterable,
|
||||
List,
|
||||
Optional,
|
||||
Union,
|
||||
cast,
|
||||
)
|
||||
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
from litellm.responses.utils import ResponsesAPIRequestUtils
|
||||
from litellm.types.llms.openai import ToolParam
|
||||
from litellm.types.utils import ModelResponse
|
||||
from litellm.utils import CustomStreamWrapper
|
||||
|
||||
CompletionCallable = Callable[..., Awaitable[Union[ModelResponse, CustomStreamWrapper]]]
|
||||
|
||||
_CHAT_COMPLETION_CALL_ARG_KEYS = [
|
||||
"model",
|
||||
"messages",
|
||||
"functions",
|
||||
"function_call",
|
||||
"timeout",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"n",
|
||||
"stream",
|
||||
"stream_options",
|
||||
"stop",
|
||||
"max_tokens",
|
||||
"max_completion_tokens",
|
||||
"modalities",
|
||||
"prediction",
|
||||
"audio",
|
||||
"presence_penalty",
|
||||
"frequency_penalty",
|
||||
"logit_bias",
|
||||
"user",
|
||||
"response_format",
|
||||
"seed",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"logprobs",
|
||||
"top_logprobs",
|
||||
"deployment_id",
|
||||
"reasoning_effort",
|
||||
"verbosity",
|
||||
"safety_identifier",
|
||||
"service_tier",
|
||||
"base_url",
|
||||
"api_version",
|
||||
"api_key",
|
||||
"model_list",
|
||||
"extra_headers",
|
||||
"thinking",
|
||||
"web_search_options",
|
||||
"shared_session",
|
||||
]
|
||||
|
||||
|
||||
def _build_call_args_from_context(call_context: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Build kwargs for `acompletion` from the `completion` call context."""
|
||||
|
||||
call_args = {
|
||||
key: call_context.get(key)
|
||||
for key in _CHAT_COMPLETION_CALL_ARG_KEYS
|
||||
if key in call_context
|
||||
}
|
||||
additional_kwargs = dict(call_context.get("kwargs") or {})
|
||||
call_args.update(additional_kwargs)
|
||||
return call_args
|
||||
|
||||
|
||||
async def _call_acompletion_internal(
|
||||
completion_callable: CompletionCallable, **call_args: Any
|
||||
async def acompletion_with_mcp(
|
||||
model: str,
|
||||
messages: List,
|
||||
tools: Optional[List] = None,
|
||||
**kwargs: Any,
|
||||
) -> Union[ModelResponse, CustomStreamWrapper]:
|
||||
"""Invoke `acompletion` while skipping MCP interception to avoid recursion."""
|
||||
"""
|
||||
Async completion with MCP integration.
|
||||
|
||||
safe_args = dict(call_args)
|
||||
safe_args["_skip_mcp_handler"] = True
|
||||
safe_args.pop("acompletion", None)
|
||||
return await completion_callable(**safe_args)
|
||||
This function handles MCP tool integration following the same pattern as aresponses_api_with_mcp.
|
||||
It's designed to be called from the synchronous completion() function and return a coroutine.
|
||||
|
||||
When MCP tools with server_url="litellm_proxy" are provided, this function will:
|
||||
1. Get available tools from the MCP server manager
|
||||
2. Transform them to OpenAI format
|
||||
3. Call acompletion with the transformed tools
|
||||
4. If require_approval="never" and tool calls are returned, automatically execute them
|
||||
5. Make a follow-up call with the tool results
|
||||
"""
|
||||
from litellm import acompletion as litellm_acompletion
|
||||
|
||||
async def handle_chat_completion_with_mcp(
|
||||
call_context: Dict[str, Any],
|
||||
completion_callable: CompletionCallable,
|
||||
) -> Optional[Union[ModelResponse, CustomStreamWrapper]]:
|
||||
"""Handle MCP-enabled tool execution for chat completion requests."""
|
||||
# Parse MCP tools and separate from other tools
|
||||
(
|
||||
mcp_tools_with_litellm_proxy,
|
||||
other_tools,
|
||||
) = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools)
|
||||
|
||||
call_args = _build_call_args_from_context(call_context)
|
||||
if not mcp_tools_with_litellm_proxy:
|
||||
# No MCP tools, proceed with regular completion
|
||||
return await litellm_acompletion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
tools = call_args.get("tools")
|
||||
if not tools:
|
||||
return None
|
||||
|
||||
tools_for_mcp = cast(Optional[Iterable[ToolParam]], tools)
|
||||
|
||||
if not LiteLLM_Proxy_MCP_Handler._should_use_litellm_mcp_gateway(
|
||||
tools=tools_for_mcp
|
||||
):
|
||||
return None
|
||||
|
||||
mcp_tools, _ = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools)
|
||||
if not mcp_tools:
|
||||
return None
|
||||
|
||||
base_call_args = dict(call_args)
|
||||
|
||||
user_api_key_auth = call_args.get("user_api_key_auth") or (
|
||||
(call_args.get("metadata", {}) or {}).get("user_api_key_auth")
|
||||
# Extract user_api_key_auth from metadata or kwargs
|
||||
user_api_key_auth = kwargs.get("user_api_key_auth") or (
|
||||
(kwargs.get("metadata", {}) or {}).get("user_api_key_auth")
|
||||
)
|
||||
|
||||
# Process MCP tools
|
||||
(
|
||||
deduplicated_mcp_tools,
|
||||
tool_server_map,
|
||||
) = await LiteLLM_Proxy_MCP_Handler._process_mcp_tools_without_openai_transform(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
)
|
||||
|
||||
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(
|
||||
|
|
@ -130,25 +70,43 @@ async def handle_chat_completion_with_mcp(
|
|||
target_format="chat",
|
||||
)
|
||||
|
||||
base_call_args["tools"] = openai_tools or None
|
||||
# Combine with other tools
|
||||
all_tools = openai_tools + other_tools if (openai_tools or other_tools) else None
|
||||
|
||||
# Determine if we should auto-execute tools
|
||||
should_auto_execute = LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
|
||||
mcp_tools_with_litellm_proxy=mcp_tools
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
|
||||
)
|
||||
|
||||
# Extract MCP auth headers
|
||||
(
|
||||
mcp_auth_header,
|
||||
mcp_server_auth_headers,
|
||||
oauth2_headers,
|
||||
raw_headers,
|
||||
) = ResponsesAPIRequestUtils.extract_mcp_headers_from_request(
|
||||
secret_fields=base_call_args.get("secret_fields"),
|
||||
secret_fields=kwargs.get("secret_fields"),
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
if not should_auto_execute:
|
||||
return await _call_acompletion_internal(completion_callable, **base_call_args)
|
||||
# Prepare call parameters
|
||||
# Remove keys that shouldn't be passed to acompletion
|
||||
clean_kwargs = {k: v for k, v in kwargs.items() if k not in ["acompletion"]}
|
||||
|
||||
base_call_args = {
|
||||
"model": model,
|
||||
"messages": messages,
|
||||
"tools": all_tools,
|
||||
"_skip_mcp_handler": True, # Prevent recursion
|
||||
**clean_kwargs,
|
||||
}
|
||||
|
||||
# If not auto-executing, just make the call with transformed tools
|
||||
if not should_auto_execute:
|
||||
return await litellm_acompletion(**base_call_args)
|
||||
|
||||
# For auto-execute: disable streaming for initial call
|
||||
stream = kwargs.get("stream", False)
|
||||
mock_tool_calls = base_call_args.pop("mock_tool_calls", None)
|
||||
|
||||
initial_call_args = dict(base_call_args)
|
||||
|
|
@ -156,23 +114,26 @@ async def handle_chat_completion_with_mcp(
|
|||
if mock_tool_calls is not None:
|
||||
initial_call_args["mock_tool_calls"] = mock_tool_calls
|
||||
|
||||
initial_response = await _call_acompletion_internal(
|
||||
completion_callable, **initial_call_args
|
||||
)
|
||||
# Make initial call
|
||||
initial_response = await litellm_acompletion(**initial_call_args)
|
||||
|
||||
if not isinstance(initial_response, ModelResponse):
|
||||
return initial_response
|
||||
|
||||
# Extract tool calls from response
|
||||
tool_calls = LiteLLM_Proxy_MCP_Handler._extract_tool_calls_from_chat_response(
|
||||
response=initial_response
|
||||
)
|
||||
|
||||
if not tool_calls:
|
||||
if base_call_args.get("stream"):
|
||||
# No tool calls, return response or retry with streaming if needed
|
||||
if stream:
|
||||
retry_args = dict(base_call_args)
|
||||
retry_args["stream"] = call_args.get("stream")
|
||||
return await _call_acompletion_internal(completion_callable, **retry_args)
|
||||
retry_args["stream"] = stream
|
||||
return await litellm_acompletion(**retry_args)
|
||||
return initial_response
|
||||
|
||||
# Execute tool calls
|
||||
tool_results = await LiteLLM_Proxy_MCP_Handler._execute_tool_calls(
|
||||
tool_server_map=tool_server_map,
|
||||
tool_calls=tool_calls,
|
||||
|
|
@ -186,14 +147,16 @@ async def handle_chat_completion_with_mcp(
|
|||
if not tool_results:
|
||||
return initial_response
|
||||
|
||||
# Create follow-up messages with tool results
|
||||
follow_up_messages = LiteLLM_Proxy_MCP_Handler._create_follow_up_messages_for_chat(
|
||||
original_messages=call_args.get("messages", []),
|
||||
original_messages=messages,
|
||||
response=initial_response,
|
||||
tool_results=tool_results,
|
||||
)
|
||||
|
||||
# Make follow-up call with original stream setting
|
||||
follow_up_call_args = dict(base_call_args)
|
||||
follow_up_call_args["messages"] = follow_up_messages
|
||||
follow_up_call_args["stream"] = call_args.get("stream")
|
||||
follow_up_call_args["stream"] = stream
|
||||
|
||||
return await _call_acompletion_internal(completion_callable, **follow_up_call_args)
|
||||
return await litellm_acompletion(**follow_up_call_args)
|
||||
|
|
|
|||
|
|
@ -8032,6 +8032,154 @@ class Router:
|
|||
)
|
||||
raise e
|
||||
|
||||
async def async_get_available_deployment_for_pass_through(
|
||||
self,
|
||||
model: str,
|
||||
request_kwargs: Dict,
|
||||
messages: Optional[List[Dict[str, str]]] = None,
|
||||
input: Optional[Union[str, List]] = None,
|
||||
specific_deployment: Optional[bool] = False,
|
||||
):
|
||||
"""
|
||||
Async version of get_available_deployment_for_pass_through
|
||||
|
||||
Only returns deployments configured with use_in_pass_through=True
|
||||
"""
|
||||
try:
|
||||
parent_otel_span = _get_parent_otel_span_from_kwargs(request_kwargs)
|
||||
|
||||
# 1. Execute pre-routing hook
|
||||
pre_routing_hook_response = await self.async_pre_routing_hook(
|
||||
model=model,
|
||||
request_kwargs=request_kwargs,
|
||||
messages=messages,
|
||||
input=input,
|
||||
specific_deployment=specific_deployment,
|
||||
)
|
||||
if pre_routing_hook_response is not None:
|
||||
model = pre_routing_hook_response.model
|
||||
messages = pre_routing_hook_response.messages
|
||||
|
||||
# 2. Get healthy deployments
|
||||
healthy_deployments = await self.async_get_healthy_deployments(
|
||||
model=model,
|
||||
request_kwargs=request_kwargs,
|
||||
messages=messages,
|
||||
input=input,
|
||||
specific_deployment=specific_deployment,
|
||||
parent_otel_span=parent_otel_span,
|
||||
)
|
||||
|
||||
# 3. If specific deployment returned, verify if it supports pass-through
|
||||
if isinstance(healthy_deployments, dict):
|
||||
litellm_params = healthy_deployments.get("litellm_params", {})
|
||||
if litellm_params.get("use_in_pass_through"):
|
||||
return healthy_deployments
|
||||
else:
|
||||
raise litellm.BadRequestError(
|
||||
message=f"Deployment {healthy_deployments.get('model_info', {}).get('id')} does not support pass-through endpoint (use_in_pass_through=False)",
|
||||
model=model,
|
||||
llm_provider="",
|
||||
)
|
||||
|
||||
# 4. Filter deployments that support pass-through
|
||||
pass_through_deployments = self._filter_pass_through_deployments(
|
||||
healthy_deployments=healthy_deployments
|
||||
)
|
||||
|
||||
if len(pass_through_deployments) == 0:
|
||||
raise litellm.BadRequestError(
|
||||
message=f"Model {model} has no deployments configured with use_in_pass_through=True. Please add use_in_pass_through: true to the deployment configuration",
|
||||
model=model,
|
||||
llm_provider="",
|
||||
)
|
||||
|
||||
# 5. Apply load balancing strategy
|
||||
start_time = time.perf_counter()
|
||||
if (
|
||||
self.routing_strategy == "usage-based-routing-v2"
|
||||
and self.lowesttpm_logger_v2 is not None
|
||||
):
|
||||
deployment = (
|
||||
await self.lowesttpm_logger_v2.async_get_available_deployments(
|
||||
model_group=model,
|
||||
healthy_deployments=pass_through_deployments, # type: ignore
|
||||
messages=messages,
|
||||
input=input,
|
||||
)
|
||||
)
|
||||
elif (
|
||||
self.routing_strategy == "latency-based-routing"
|
||||
and self.lowestlatency_logger is not None
|
||||
):
|
||||
deployment = (
|
||||
await self.lowestlatency_logger.async_get_available_deployments(
|
||||
model_group=model,
|
||||
healthy_deployments=pass_through_deployments, # type: ignore
|
||||
messages=messages,
|
||||
input=input,
|
||||
request_kwargs=request_kwargs,
|
||||
)
|
||||
)
|
||||
elif self.routing_strategy == "simple-shuffle":
|
||||
return simple_shuffle(
|
||||
llm_router_instance=self,
|
||||
healthy_deployments=pass_through_deployments,
|
||||
model=model,
|
||||
)
|
||||
elif (
|
||||
self.routing_strategy == "least-busy"
|
||||
and self.leastbusy_logger is not None
|
||||
):
|
||||
deployment = (
|
||||
await self.leastbusy_logger.async_get_available_deployments(
|
||||
model_group=model,
|
||||
healthy_deployments=pass_through_deployments, # type: ignore
|
||||
)
|
||||
)
|
||||
else:
|
||||
deployment = None
|
||||
|
||||
if deployment is None:
|
||||
exception = await async_raise_no_deployment_exception(
|
||||
litellm_router_instance=self,
|
||||
model=model,
|
||||
parent_otel_span=parent_otel_span,
|
||||
)
|
||||
raise exception
|
||||
|
||||
verbose_router_logger.info(
|
||||
f"async_get_available_deployment_for_pass_through model: {model}, selected deployment: {self.print_deployment(deployment)}"
|
||||
)
|
||||
|
||||
end_time = time.perf_counter()
|
||||
_duration = end_time - start_time
|
||||
asyncio.create_task(
|
||||
self.service_logger_obj.async_service_success_hook(
|
||||
service=ServiceTypes.ROUTER,
|
||||
duration=_duration,
|
||||
call_type="<routing_strategy>.async_get_available_deployments",
|
||||
parent_otel_span=parent_otel_span,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
)
|
||||
)
|
||||
|
||||
return deployment
|
||||
except Exception as e:
|
||||
traceback_exception = traceback.format_exc()
|
||||
if request_kwargs is not None:
|
||||
logging_obj = request_kwargs.get("litellm_logging_obj", None)
|
||||
if logging_obj is not None:
|
||||
threading.Thread(
|
||||
target=logging_obj.failure_handler,
|
||||
args=(e, traceback_exception),
|
||||
).start()
|
||||
asyncio.create_task(
|
||||
logging_obj.async_failure_handler(e, traceback_exception) # type: ignore
|
||||
)
|
||||
raise e
|
||||
|
||||
async def async_pre_routing_hook(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -8184,6 +8332,169 @@ class Router:
|
|||
)
|
||||
return deployment
|
||||
|
||||
def get_available_deployment_for_pass_through(
|
||||
self,
|
||||
model: str,
|
||||
messages: Optional[List[Dict[str, str]]] = None,
|
||||
input: Optional[Union[str, List]] = None,
|
||||
specific_deployment: Optional[bool] = False,
|
||||
request_kwargs: Optional[Dict] = None,
|
||||
):
|
||||
"""
|
||||
Returns deployments available for pass-through endpoints (based on load balancing strategy)
|
||||
|
||||
Similar to get_available_deployment, but only returns deployments with use_in_pass_through=True
|
||||
|
||||
Args:
|
||||
model: Model name
|
||||
messages: Optional list of messages
|
||||
input: Optional input data
|
||||
specific_deployment: Whether to find a specific deployment
|
||||
request_kwargs: Optional request parameters
|
||||
|
||||
Returns:
|
||||
Dict: Selected deployment configuration
|
||||
|
||||
Raises:
|
||||
BadRequestError: If no deployment is configured with use_in_pass_through=True
|
||||
RouterRateLimitError: If no pass-through deployments are available
|
||||
"""
|
||||
# 1. Perform common checks to get healthy deployments list
|
||||
model, healthy_deployments = self._common_checks_available_deployment(
|
||||
model=model,
|
||||
messages=messages,
|
||||
input=input,
|
||||
specific_deployment=specific_deployment,
|
||||
)
|
||||
|
||||
# 2. If the returned is a specific deployment (Dict), verify and return directly
|
||||
if isinstance(healthy_deployments, dict):
|
||||
litellm_params = healthy_deployments.get("litellm_params", {})
|
||||
if litellm_params.get("use_in_pass_through"):
|
||||
return healthy_deployments
|
||||
else:
|
||||
# Specific deployment does not support pass-through
|
||||
raise litellm.BadRequestError(
|
||||
message=f"Deployment {healthy_deployments.get('model_info', {}).get('id')} does not support pass-through endpoint (use_in_pass_through=False)",
|
||||
model=model,
|
||||
llm_provider="",
|
||||
)
|
||||
|
||||
# 3. Filter deployments that support pass-through
|
||||
pass_through_deployments = self._filter_pass_through_deployments(
|
||||
healthy_deployments=healthy_deployments
|
||||
)
|
||||
|
||||
if len(pass_through_deployments) == 0:
|
||||
# No deployments support pass-through
|
||||
raise litellm.BadRequestError(
|
||||
message=f"Model {model} has no deployment configured with use_in_pass_through=True. Please add use_in_pass_through: true in the deployment configuration",
|
||||
model=model,
|
||||
llm_provider="",
|
||||
)
|
||||
|
||||
# 4. Apply cooldown filtering
|
||||
parent_otel_span: Optional[Span] = _get_parent_otel_span_from_kwargs(
|
||||
request_kwargs
|
||||
)
|
||||
cooldown_deployments = _get_cooldown_deployments(
|
||||
litellm_router_instance=self, parent_otel_span=parent_otel_span
|
||||
)
|
||||
pass_through_deployments = self._filter_cooldown_deployments(
|
||||
healthy_deployments=pass_through_deployments,
|
||||
cooldown_deployments=cooldown_deployments,
|
||||
)
|
||||
|
||||
# 5. Apply pre-call checks (if enabled)
|
||||
if self.enable_pre_call_checks and messages is not None:
|
||||
pass_through_deployments = self._pre_call_checks(
|
||||
model=model,
|
||||
healthy_deployments=pass_through_deployments,
|
||||
messages=messages,
|
||||
request_kwargs=request_kwargs,
|
||||
)
|
||||
|
||||
if len(pass_through_deployments) == 0:
|
||||
model_ids = self.get_model_ids(model_name=model)
|
||||
_cooldown_time = self.cooldown_cache.get_min_cooldown(
|
||||
model_ids=model_ids, parent_otel_span=parent_otel_span
|
||||
)
|
||||
_cooldown_list = _get_cooldown_deployments(
|
||||
litellm_router_instance=self, parent_otel_span=parent_otel_span
|
||||
)
|
||||
raise RouterRateLimitError(
|
||||
model=model,
|
||||
cooldown_time=_cooldown_time,
|
||||
enable_pre_call_checks=self.enable_pre_call_checks,
|
||||
cooldown_list=_cooldown_list,
|
||||
)
|
||||
|
||||
# 6. Apply load balancing strategy
|
||||
if self.routing_strategy == "least-busy" and self.leastbusy_logger is not None:
|
||||
deployment = self.leastbusy_logger.get_available_deployments(
|
||||
model_group=model, healthy_deployments=pass_through_deployments # type: ignore
|
||||
)
|
||||
elif self.routing_strategy == "simple-shuffle":
|
||||
return simple_shuffle(
|
||||
llm_router_instance=self,
|
||||
healthy_deployments=pass_through_deployments,
|
||||
model=model,
|
||||
)
|
||||
elif (
|
||||
self.routing_strategy == "latency-based-routing"
|
||||
and self.lowestlatency_logger is not None
|
||||
):
|
||||
deployment = self.lowestlatency_logger.get_available_deployments(
|
||||
model_group=model,
|
||||
healthy_deployments=pass_through_deployments, # type: ignore
|
||||
request_kwargs=request_kwargs,
|
||||
)
|
||||
elif (
|
||||
self.routing_strategy == "usage-based-routing"
|
||||
and self.lowesttpm_logger is not None
|
||||
):
|
||||
deployment = self.lowesttpm_logger.get_available_deployments(
|
||||
model_group=model,
|
||||
healthy_deployments=pass_through_deployments, # type: ignore
|
||||
messages=messages,
|
||||
input=input,
|
||||
)
|
||||
elif (
|
||||
self.routing_strategy == "usage-based-routing-v2"
|
||||
and self.lowesttpm_logger_v2 is not None
|
||||
):
|
||||
deployment = self.lowesttpm_logger_v2.get_available_deployments(
|
||||
model_group=model,
|
||||
healthy_deployments=pass_through_deployments, # type: ignore
|
||||
messages=messages,
|
||||
input=input,
|
||||
)
|
||||
else:
|
||||
deployment = None
|
||||
|
||||
if deployment is None:
|
||||
verbose_router_logger.info(
|
||||
f"get_available_deployment_for_pass_through model: {model}, no available deployments"
|
||||
)
|
||||
model_ids = self.get_model_ids(model_name=model)
|
||||
_cooldown_time = self.cooldown_cache.get_min_cooldown(
|
||||
model_ids=model_ids, parent_otel_span=parent_otel_span
|
||||
)
|
||||
_cooldown_list = _get_cooldown_deployments(
|
||||
litellm_router_instance=self, parent_otel_span=parent_otel_span
|
||||
)
|
||||
raise RouterRateLimitError(
|
||||
model=model,
|
||||
cooldown_time=_cooldown_time,
|
||||
enable_pre_call_checks=self.enable_pre_call_checks,
|
||||
cooldown_list=_cooldown_list,
|
||||
)
|
||||
|
||||
verbose_router_logger.info(
|
||||
f"get_available_deployment_for_pass_through model: {model}, selected deployment: {self.print_deployment(deployment)}"
|
||||
)
|
||||
return deployment
|
||||
|
||||
def _filter_cooldown_deployments(
|
||||
self, healthy_deployments: List[Dict], cooldown_deployments: List[str]
|
||||
) -> List[Dict]:
|
||||
|
|
@ -8206,6 +8517,34 @@ class Router:
|
|||
if deployment["model_info"]["id"] not in cooldown_set
|
||||
]
|
||||
|
||||
def _filter_pass_through_deployments(
|
||||
self, healthy_deployments: List[Dict]
|
||||
) -> List[Dict]:
|
||||
"""
|
||||
Filter out deployments configured with use_in_pass_through=True
|
||||
|
||||
Args:
|
||||
healthy_deployments: List of healthy deployments
|
||||
|
||||
Returns:
|
||||
List[Dict]: Only includes a list of deployments that support pass-through
|
||||
"""
|
||||
verbose_router_logger.debug(
|
||||
f"Filter pass-through deployments from {len(healthy_deployments)} healthy deployments"
|
||||
)
|
||||
|
||||
pass_through_deployments = [
|
||||
deployment
|
||||
for deployment in healthy_deployments
|
||||
if deployment.get("litellm_params", {}).get("use_in_pass_through", False)
|
||||
]
|
||||
|
||||
verbose_router_logger.debug(
|
||||
f"Found {len(pass_through_deployments)} deployments with pass-through enabled"
|
||||
)
|
||||
|
||||
return pass_through_deployments
|
||||
|
||||
def _track_deployment_metrics(
|
||||
self, deployment, parent_otel_span: Optional[Span], response=None
|
||||
):
|
||||
|
|
|
|||
|
|
@ -175,6 +175,9 @@ DEFINED_PROMETHEUS_METRICS = Literal[
|
|||
"litellm_remaining_api_key_budget_metric",
|
||||
"litellm_api_key_max_budget_metric",
|
||||
"litellm_api_key_budget_remaining_hours_metric",
|
||||
"litellm_remaining_user_budget_metric",
|
||||
"litellm_user_max_budget_metric",
|
||||
"litellm_user_budget_remaining_hours_metric",
|
||||
"litellm_deployment_state",
|
||||
"litellm_deployment_failure_responses",
|
||||
"litellm_deployment_total_requests",
|
||||
|
|
@ -421,6 +424,18 @@ class PrometheusMetricLabels:
|
|||
litellm_remaining_api_key_budget_metric
|
||||
)
|
||||
|
||||
litellm_remaining_user_budget_metric = [
|
||||
UserAPIKeyLabelNames.USER.value,
|
||||
]
|
||||
|
||||
litellm_user_max_budget_metric = [
|
||||
UserAPIKeyLabelNames.USER.value,
|
||||
]
|
||||
|
||||
litellm_user_budget_remaining_hours_metric = [
|
||||
UserAPIKeyLabelNames.USER.value,
|
||||
]
|
||||
|
||||
# Add deployment metrics
|
||||
litellm_deployment_failure_responses = [
|
||||
UserAPIKeyLabelNames.REQUESTED_MODEL.value,
|
||||
|
|
|
|||
|
|
@ -636,8 +636,10 @@ class ANTHROPIC_BETA_HEADER_VALUES(str, Enum):
|
|||
ADVANCED_TOOL_USE_2025_11_20 = "advanced-tool-use-2025-11-20"
|
||||
|
||||
|
||||
# Tool search beta header constant
|
||||
# Tool search beta header constant (for Anthropic direct API and Microsoft Foundry)
|
||||
ANTHROPIC_TOOL_SEARCH_BETA_HEADER = "advanced-tool-use-2025-11-20"
|
||||
|
||||
# Effort beta header constant
|
||||
ANTHROPIC_EFFORT_BETA_HEADER = "effort-2025-11-24"
|
||||
|
||||
|
||||
|
|
|
|||
36
litellm/types/llms/anthropic_tool_search.py
Normal file
36
litellm/types/llms/anthropic_tool_search.py
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
"""
|
||||
Tool Search Beta Header Configuration
|
||||
|
||||
Reference: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool
|
||||
"""
|
||||
|
||||
from typing import Dict
|
||||
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
# Tool search beta header values
|
||||
TOOL_SEARCH_BETA_HEADER_ANTHROPIC = "advanced-tool-use-2025-11-20"
|
||||
TOOL_SEARCH_BETA_HEADER_VERTEX = "tool-search-tool-2025-10-19"
|
||||
TOOL_SEARCH_BETA_HEADER_BEDROCK = "tool-search-tool-2025-10-19"
|
||||
|
||||
|
||||
# Mapping of custom_llm_provider -> tool search beta header
|
||||
TOOL_SEARCH_BETA_HEADER_BY_PROVIDER: Dict[str, str] = {
|
||||
LlmProviders.ANTHROPIC.value: TOOL_SEARCH_BETA_HEADER_ANTHROPIC,
|
||||
LlmProviders.AZURE.value: TOOL_SEARCH_BETA_HEADER_ANTHROPIC,
|
||||
LlmProviders.AZURE_AI.value: TOOL_SEARCH_BETA_HEADER_ANTHROPIC,
|
||||
LlmProviders.VERTEX_AI.value: TOOL_SEARCH_BETA_HEADER_VERTEX,
|
||||
LlmProviders.VERTEX_AI_BETA.value: TOOL_SEARCH_BETA_HEADER_VERTEX,
|
||||
LlmProviders.BEDROCK.value: TOOL_SEARCH_BETA_HEADER_BEDROCK,
|
||||
}
|
||||
|
||||
|
||||
def get_tool_search_beta_header(custom_llm_provider: str) -> str:
|
||||
"""
|
||||
Get the tool search beta header for a given provider.
|
||||
"""
|
||||
return TOOL_SEARCH_BETA_HEADER_BY_PROVIDER.get(
|
||||
custom_llm_provider,
|
||||
TOOL_SEARCH_BETA_HEADER_ANTHROPIC
|
||||
)
|
||||
|
||||
|
|
@ -364,6 +364,11 @@ class CallTypes(str, Enum):
|
|||
asend_message = "asend_message"
|
||||
send_message = "send_message"
|
||||
|
||||
#########################################################
|
||||
# Claude Code Call Types
|
||||
#########################################################
|
||||
acreate_skill = "acreate_skill"
|
||||
|
||||
|
||||
CallTypesLiteral = Literal[
|
||||
"embedding",
|
||||
|
|
@ -420,6 +425,7 @@ CallTypesLiteral = Literal[
|
|||
"send_message",
|
||||
"aresponses",
|
||||
"responses",
|
||||
"acreate_skill",
|
||||
]
|
||||
|
||||
# Mapping of API routes to their corresponding call types
|
||||
|
|
|
|||
|
|
@ -8354,6 +8354,12 @@ class ProviderConfigManager:
|
|||
)
|
||||
|
||||
return get_vertex_ai_image_generation_config(model)
|
||||
elif LlmProviders.OPENROUTER == provider:
|
||||
from litellm.llms.openrouter.image_generation import (
|
||||
get_openrouter_image_generation_config,
|
||||
)
|
||||
|
||||
return get_openrouter_image_generation_config(model)
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -28782,13 +28782,13 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"vertex_ai/zai-org/glm-4.7-maas": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "vertex_ai-zai_models",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_token": 2.2e-06,
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[tool.poetry]
|
||||
name = "litellm"
|
||||
version = "1.80.16"
|
||||
version = "1.80.17"
|
||||
description = "Library to easily interface with LLM API providers"
|
||||
authors = ["BerriAI"]
|
||||
license = "MIT"
|
||||
|
|
@ -167,7 +167,7 @@ requires = ["poetry-core", "wheel"]
|
|||
build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "1.80.16"
|
||||
version = "1.80.17"
|
||||
version_files = [
|
||||
"pyproject.toml:^version"
|
||||
]
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue