Merge remote-tracking branch 'origin/add-digitalocean-provider' into add-digitalocean-provider

This commit is contained in:
Muhammad Sannan Nasir 2025-07-07 06:00:21 +05:00
commit 44ba3971d6
440 changed files with 19638 additions and 4151 deletions

View file

@ -950,14 +950,14 @@ jobs:
command: |
pwd
ls
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 4
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
no_output_timeout: 120m
- run:
name: Run enterprise tests
command: |
pwd
ls
python -m pytest -vv tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-enterprise.xml --durations=10 -n 4
python -m pytest -vv tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-enterprise.xml --durations=10 -n 8
no_output_timeout: 120m
- run:
name: Rename the coverage files
@ -1358,6 +1358,7 @@ jobs:
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
- run: python ./tests/code_coverage_tests/check_licenses.py
- run: python ./tests/code_coverage_tests/router_code_coverage.py
- run: python ./tests/code_coverage_tests/code_qa_check_tests.py
- run: python ./tests/code_coverage_tests/test_proxy_types_import.py
- run: python ./tests/code_coverage_tests/callback_manager_test.py
- run: python ./tests/code_coverage_tests/recursive_detector.py

View file

@ -30,6 +30,7 @@ jobs:
poetry install --with dev,proxy-dev --extras proxy
poetry run pip install "pytest-retry==1.6.3"
poetry run pip install pytest-xdist
poetry run pip install "google-genai==1.22.0"
- name: Setup litellm-enterprise as local package
run: |
cd enterprise

View file

@ -72,7 +72,7 @@ messages = [{ "content": "Hello, how are you?","role": "user"}]
response = completion(model="openai/gpt-4o", messages=messages)
# anthropic call
response = completion(model="anthropic/claude-3-sonnet-20240229", messages=messages)
response = completion(model="anthropic/claude-sonnet-4-20250514", messages=messages)
print(response)
```
@ -80,9 +80,9 @@ print(response)
```json
{
"id": "chatcmpl-565d891b-a42e-4c39-8d14-82a1f5208885",
"created": 1734366691,
"model": "claude-3-sonnet-20240229",
"id": "chatcmpl-1214900a-6cdd-4148-b663-b5e2f642b4de",
"created": 1751494488,
"model": "claude-sonnet-4-20250514",
"object": "chat.completion",
"system_fingerprint": null,
"choices": [
@ -90,7 +90,7 @@ print(response)
"finish_reason": "stop",
"index": 0,
"message": {
"content": "Hello! As an AI language model, I don't have feelings, but I'm operating properly and ready to assist you with any questions or tasks you may have. How can I help you today?",
"content": "Hello! I'm doing well, thank you for asking. I'm here and ready to help with whatever you'd like to discuss or work on. How are you doing today?",
"role": "assistant",
"tool_calls": null,
"function_call": null
@ -98,9 +98,9 @@ print(response)
}
],
"usage": {
"completion_tokens": 43,
"completion_tokens": 39,
"prompt_tokens": 13,
"total_tokens": 56,
"total_tokens": 52,
"completion_tokens_details": null,
"prompt_tokens_details": {
"audio_tokens": null,
@ -141,8 +141,8 @@ response = completion(model="openai/gpt-4o", messages=messages, stream=True)
for part in response:
print(part.choices[0].delta.content or "")
# claude 2
response = completion('anthropic/claude-3-sonnet-20240229', messages, stream=True)
# claude sonnet 4
response = completion('anthropic/claude-sonnet-4-20250514', messages, stream=True)
for part in response:
print(part)
```
@ -151,9 +151,9 @@ for part in response:
```json
{
"id": "chatcmpl-2be06597-eb60-4c70-9ec5-8cd2ab1b4697",
"created": 1734366925,
"model": "claude-3-sonnet-20240229",
"id": "chatcmpl-fe575c37-5004-4926-ae5e-bfbc31f356ca",
"created": 1751494808,
"model": "claude-sonnet-4-20250514",
"object": "chat.completion.chunk",
"system_fingerprint": null,
"choices": [
@ -161,6 +161,7 @@ for part in response:
"finish_reason": null,
"index": 0,
"delta": {
"provider_specific_fields": null,
"content": "Hello",
"role": "assistant",
"function_call": null,
@ -169,7 +170,10 @@ for part in response:
},
"logprobs": null
}
]
],
"provider_specific_fields": null,
"stream_options": null,
"citations": null
}
```

View file

@ -279,7 +279,7 @@ with run as run:
curl -X POST 'http://0.0.0.0:4000/threads/{thread_id}/runs' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"assistant_id": "asst_6xVZQFFy1Kw87NbnYeNebxTf",
"stream": true
}'

View file

@ -0,0 +1,236 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Google AI generateContent
Use LiteLLM to call Google AI's generateContent endpoints for text generation, multimodal interactions, and streaming responses.
## Overview
| Feature | Supported | Notes |
|-------|-------|-------|
| Cost Tracking | ✅ | |
| Logging | ✅ | works across all integrations |
| End-user Tracking | ✅ | |
| Streaming | ✅ | |
| Fallbacks | ✅ | between supported models |
| Loadbalancing | ✅ | between supported models |
## Usage
---
### LiteLLM Python SDK
<Tabs>
<TabItem value="basic" label="Basic Usage">
#### Non-streaming example
```python showLineNumbers title="Basic Text Generation"
from litellm.google_genai import agenerate_content
from google.genai.types import ContentDict, PartDict
import os
# Set API key
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
contents = ContentDict(
parts=[
PartDict(text="Hello, can you tell me a short joke?")
],
role="user",
)
response = await agenerate_content(
contents=contents,
model="gemini/gemini-2.0-flash",
max_tokens=100,
)
print(response)
```
#### Streaming example
```python showLineNumbers title="Streaming Text Generation"
from litellm.google_genai import agenerate_content_stream
from google.genai.types import ContentDict, PartDict
import os
# Set API key
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
contents = ContentDict(
parts=[
PartDict(text="Write a long story about space exploration")
],
role="user",
)
response = await agenerate_content_stream(
contents=contents,
model="gemini/gemini-2.0-flash",
max_tokens=500,
)
async for chunk in response:
print(chunk)
```
</TabItem>
<TabItem value="sync" label="Sync Usage">
#### Sync non-streaming example
```python showLineNumbers title="Sync Text Generation"
from litellm.google_genai import generate_content
from google.genai.types import ContentDict, PartDict
import os
# Set API key
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
contents = ContentDict(
parts=[
PartDict(text="Hello, can you tell me a short joke?")
],
role="user",
)
response = generate_content(
contents=contents,
model="gemini/gemini-2.0-flash",
max_tokens=100,
)
print(response)
```
#### Sync streaming example
```python showLineNumbers title="Sync Streaming Text Generation"
from litellm.google_genai import generate_content_stream
from google.genai.types import ContentDict, PartDict
import os
# Set API key
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
contents = ContentDict(
parts=[
PartDict(text="Write a long story about space exploration")
],
role="user",
)
response = generate_content_stream(
contents=contents,
model="gemini/gemini-2.0-flash",
max_tokens=500,
)
for chunk in response:
print(chunk)
```
</TabItem>
</Tabs>
### LiteLLM Proxy Server
1. Setup config.yaml
```yaml
model_list:
- model_name: gemini-flash
litellm_params:
model: gemini/gemini-2.0-flash
api_key: os.environ/GEMINI_API_KEY
```
2. Start proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
<Tabs>
<TabItem value="gemini-proxy" label="Google GenAI SDK">
```python showLineNumbers title="Google GenAI SDK with LiteLLM Proxy"
from google.genai import Client
import os
# Configure Google GenAI SDK to use LiteLLM proxy
os.environ["GOOGLE_GEMINI_BASE_URL"] = "http://localhost:4000"
os.environ["GEMINI_API_KEY"] = "sk-1234"
client = Client()
response = client.models.generate_content(
model="gemini-flash",
contents=[
{
"parts": [{"text": "Write a short story about AI"}],
"role": "user"
}
],
config={"max_output_tokens": 100}
)
```
</TabItem>
<TabItem value="curl-proxy" label="curl">
#### Generate Content
```bash showLineNumbers title="generateContent via LiteLLM Proxy"
curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:generateContent' \
-H 'content-type: application/json' \
-H 'authorization: Bearer sk-1234' \
-d '{
"contents": [
{
"parts": [
{
"text": "Write a short story about AI"
}
],
"role": "user"
}
],
"generationConfig": {
"maxOutputTokens": 100
}
}'
```
#### Stream Generate Content
```bash showLineNumbers title="streamGenerateContent via LiteLLM Proxy"
curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:streamGenerateContent' \
-H 'content-type: application/json' \
-H 'authorization: Bearer sk-1234' \
-d '{
"contents": [
{
"parts": [
{
"text": "Write a long story about space exploration"
}
],
"role": "user"
}
],
"generationConfig": {
"maxOutputTokens": 500
}
}'
```
</TabItem>
</Tabs>
## Related
- [Use LiteLLM with gemini-cli](../docs/tutorials/litellm_gemini_cli)

View file

@ -3,12 +3,43 @@ import TabItem from '@theme/TabItem';
# SSL, HTTP Proxy Security Settings
If you're in an environment using an older TTS bundle, with an older encryption, follow this guide.
If you're in an environment using an older TTS bundle, with an older encryption, follow this guide. By default
LiteLLM uses the certifi CA bundle for SSL verification, which is compatible with most modern servers.
However, if you need to disable SSL verification or use a custom CA bundle, you can do so by following the steps below.
Be aware that environmental variables take precedence over the settings in the SDK.
LiteLLM uses HTTPX for network requests, unless otherwise specified.
LiteLLM uses HTTPX for network requests, unless otherwise specified.
## 1. Disable SSL verification
## 1. Custom CA Bundle
You can set a custom CA bundle file path using the `SSL_CERT_FILE` environmental variable or passing a string to the the ssl_verify setting.
<Tabs>
<TabItem value="sdk" label="SDK">
```python
import litellm
litellm.ssl_verify = "client.pem"
```
</TabItem>
<TabItem value="proxy" label="PROXY">
```yaml
litellm_settings:
ssl_verify: "client.pem"
```
</TabItem>
<TabItem value="env_var" label="Environment Variables">
```bash
export SSL_CERT_FILE="client.pem"
```
</TabItem>
</Tabs>
## 2. Disable SSL verification
<Tabs>
@ -35,14 +66,42 @@ export SSL_VERIFY="False"
</TabItem>
</Tabs>
## 2. Lower security settings
## 3. Lower security settings
The `ssl_security_level` allows setting a lower security level for SSL connections.
<Tabs>
<TabItem value="sdk" label="SDK">
```python
import litellm
litellm.ssl_security_level = "DEFAULT@SECLEVEL=1"
```
</TabItem>
<TabItem value="proxy" label="PROXY">
```yaml
litellm_settings:
ssl_security_level: "DEFAULT@SECLEVEL=1"
```
</TabItem>
<TabItem value="env_var" label="Environment Variables">
```bash
export SSL_SECURITY_LEVEL="DEFAULT@SECLEVEL=1"
```
</TabItem>
</Tabs>
## 4. Certificate authentication
The `SSL_CERTIFICATE` environmental variable or `ssl_certificate` attribute allows setting a client side certificate to authenticate the client to the server.
<Tabs>
<TabItem value="sdk" label="SDK">
```python
import litellm
litellm.ssl_security_level = 1
litellm.ssl_certificate = "/path/to/certificate.pem"
```
</TabItem>
@ -50,20 +109,18 @@ litellm.ssl_certificate = "/path/to/certificate.pem"
```yaml
litellm_settings:
ssl_security_level: 1
ssl_certificate: "/path/to/certificate.pem"
```
</TabItem>
<TabItem value="env_var" label="Environment Variables">
```bash
export SSL_SECURITY_LEVEL="1"
export SSL_CERTIFICATE="/path/to/certificate.pem"
```
</TabItem>
</Tabs>
## 3. Use HTTP_PROXY environment variable
## 5. Use HTTP_PROXY environment variable
Both httpx and aiohttp libraries use `urllib.request.getproxies` from environment variables. Before client initialization, you may set proxy (and optional SSL_CERT_FILE) by setting the environment variables:

View file

@ -52,7 +52,7 @@ litellm --config /path/to/config.yaml
curl -X POST 'http://0.0.0.0:4000/v1/images/generations' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-D '{
-d '{
"model": "gpt-image-1",
"prompt": "A cute baby sea otter",
"n": 1,

View file

@ -83,7 +83,6 @@ mcp_servers:
</Tabs>
## Using your MCP
<Tabs>
@ -159,7 +158,7 @@ Use tools directly from Cursor IDE with LiteLLM MCP:
2. **Navigate to MCP Tools**: Go to the "MCP Tools" tab and click "New MCP Server"
3. **Add Configuration**: Copy and paste the JSON configuration below, then save with `Cmd+S` or `Ctrl+S`
```json title="Cursor MCP Configuration" showLineNumbers
```json title="Basic Cursor MCP Configuration" showLineNumbers
{
"mcpServers": {
"LiteLLM": {
@ -173,98 +172,100 @@ Use tools directly from Cursor IDE with LiteLLM MCP:
```
</TabItem>
</Tabs>
<TabItem value="http" label="Streamable HTTP">
## Segregating MCP Server Access
#### Connect via Streamable HTTP Transport
You can choose to access specific MCP servers and only list their tools using the `x-mcp-servers` header. This header allows you to:
- Limit tool access to one or more specific MCP servers
- Control which tools are available in different environments or use cases
Connect to LiteLLM MCP using HTTP transport. Compatible with any MCP client that supports HTTP streaming:
The header accepts a comma-separated list of server names: `"Zapier_Gmail,Server2,Server3"`
**Server URL:**
```text showLineNumbers
<your-litellm-proxy-base-url>/mcp
Notes:
- Server names with spaces should be replaced with underscores
- If the header is not provided, tools from all available MCP servers will be accessible
<Tabs>
<TabItem value="openai" label="OpenAI API">
```bash title="cURL Example with Server Segregation" showLineNumbers
curl --location 'https://api.openai.com/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer $OPENAI_API_KEY" \
--data '{
"model": "gpt-4o",
"tools": [
{
"type": "mcp",
"server_label": "litellm",
"server_url": "<your-litellm-proxy-base-url>/mcp",
"require_approval": "never",
"headers": {
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
"x-mcp-servers": "Zapier_Gmail"
}
}
],
"input": "Run available tools",
"tool_choice": "required"
}'
```
**Headers:**
```text showLineNumbers
x-litellm-api-key: Bearer YOUR_LITELLM_API_KEY
```
This URL can be used with any MCP client that supports HTTP transport. Refer to your client documentation to determine the appropriate transport method.
In this example, the request will only have access to tools from the "Zapier_Gmail" MCP server.
</TabItem>
<TabItem value="fastmcp" label="Python FastMCP">
<TabItem value="litellm" label="LiteLLM Proxy">
#### Connect via Python FastMCP Client
Use the Python FastMCP client to connect to your LiteLLM MCP server:
**Installation:**
```bash title="Install FastMCP" showLineNumbers
pip install fastmcp
```bash title="cURL Example with Server Segregation" showLineNumbers
curl --location '<your-litellm-proxy-base-url>/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer $LITELLM_API_KEY" \
--data '{
"model": "gpt-4o",
"tools": [
{
"type": "mcp",
"server_label": "litellm",
"server_url": "<your-litellm-proxy-base-url>/mcp",
"require_approval": "never",
"headers": {
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
"x-mcp-servers": "Zapier_Gmail,Server2"
}
}
],
"input": "Run available tools",
"tool_choice": "required"
}'
```
or with uv:
This configuration restricts the request to only use tools from the specified MCP servers.
```bash title="Install with uv" showLineNumbers
uv pip install fastmcp
```
</TabItem>
**Usage:**
<TabItem value="cursor" label="Cursor IDE">
```python title="Python FastMCP Example" showLineNumbers
import asyncio
import json
from fastmcp import Client
from fastmcp.client.transports import StreamableHttpTransport
# Create the transport with your LiteLLM MCP server URL
server_url = "<your-litellm-proxy-base-url>/mcp"
transport = StreamableHttpTransport(
server_url,
headers={
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
```json title="Cursor MCP Configuration with Server Segregation" showLineNumbers
{
"mcpServers": {
"LiteLLM": {
"url": "<your-litellm-proxy-base-url>/mcp",
"headers": {
"x-litellm-api-key": "Bearer $LITELLM_API_KEY",
"x-mcp-servers": "Zapier_Gmail,Server2"
}
}
)
# Initialize the client with the transport
client = Client(transport=transport)
async def main():
# Connection is established here
print("Connecting to LiteLLM MCP server...")
async with client:
print(f"Client connected: {client.is_connected()}")
# Make MCP calls within the context
print("Fetching available tools...")
tools = await client.list_tools()
print(f"Available tools: {json.dumps([t.name for t in tools], indent=2)}")
# Example: Call a tool (replace 'tool_name' with an actual tool name)
if tools:
tool_name = tools[0].name
print(f"Calling tool: {tool_name}")
# Call the tool with appropriate arguments
result = await client.call_tool(tool_name, arguments={})
print(f"Tool result: {result}")
# Run the example
if __name__ == "__main__":
asyncio.run(main())
}
}
```
This configuration in Cursor IDE settings will limit tool access to only the specified MCP server.
</TabItem>
</Tabs>
## Using your MCP with client side credentials
Use this if you want to pass a client side authentication token to LiteLLM to then pass to your MCP to auth to your MCP.
@ -715,4 +716,4 @@ async with stdio_client(server_params) as (read, write):
```
</TabItem>
</Tabs>
</Tabs>

View file

@ -11,6 +11,13 @@ Example trace in Langfuse using multiple models via LiteLLM:
<Image img={require('../../img/langfuse-example-trace-multiple-models-min.png')} />
:::info
For Langfuse v3, we recommend using the [Langfuse OTEL](./langfuse_otel_integration) integration.
:::
## Usage with LiteLLM Proxy (LLM Gateway)
👉 [**Follow this link to start sending logs to langfuse with LiteLLM Proxy server**](../proxy/logging)

View file

@ -1,7 +1,14 @@
# Langfuse OpenTelemetry Integration
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
import Image from '@theme/IdealImage';
# 🪢 Langfuse OpenTelemetry Integration
The Langfuse OpenTelemetry integration allows you to send LiteLLM traces and observability data to Langfuse using the OpenTelemetry protocol. This provides a standardized way to collect and analyze your LLM usage data.
<Image img={require('../../img/langfuse_otel.png')} />
## Features
- Automatic trace collection for all LiteLLM requests
@ -108,15 +115,26 @@ litellm.callbacks = ["langfuse_otel"]
Add the integration to your proxy configuration:
1. Add the credentials to your environment variables
```bash
export LANGFUSE_PUBLIC_KEY="pk-lf-..."
export LANGFUSE_SECRET_KEY="sk-lf-..."
export LANGFUSE_HOST="https://us.cloud.langfuse.com" # Default US region
```
2. Setup config.yaml
```yaml
# config.yaml
litellm_settings:
callbacks: ["langfuse_otel"]
```
environment_variables:
LANGFUSE_PUBLIC_KEY: "pk-lf-..."
LANGFUSE_SECRET_KEY: "sk-lf-..."
LANGFUSE_HOST: "https://us.cloud.langfuse.com" # Default US region
3. Run the proxy
```bash
litellm --config /path/to/config.yaml
```
## Data Collected
@ -163,11 +181,24 @@ This is automatically handled by the integration - you just need to provide the
Enable verbose logging to see detailed information:
<Tabs>
<TabItem value="sdk" label="SDK">
```python
import litellm
litellm.set_verbose = True
litellm._turn_on_debug()
```
</TabItem>
<TabItem value="proxy" label="PROXY">
```bash
export LITELLM_LOG="DEBUG"
```
</TabItem>
</Tabs>
This will show:
- Endpoint resolution logic
- Authentication header creation

View file

@ -104,4 +104,14 @@ for successful + failed requests
click under `litellm_request` in the trace
<Image img={require('../../img/otel_debug_trace.png')} />
<Image img={require('../../img/otel_debug_trace.png')} />
### Not seeing traces land on Integration
If you don't see traces landing on your integration, set `OTEL_DEBUG="True"` in your LiteLLM environment and try again.
```shell
export OTEL_DEBUG="True"
```
This will emit any logging issues to the console.

View file

@ -212,7 +212,7 @@ If you need to switch `pii_masking` off for an API Key set `"permissions": {"pii
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"permissions": {"pii_masking": true}
}'
```

View file

@ -339,7 +339,7 @@ documents = [
]
response = rerank(
model="azure_ai/rerank-english-v3.0",
model="azure_ai/cohere-rerank-v3.5",
query=query,
documents=documents,
top_n=3,
@ -362,9 +362,9 @@ model_list:
litellm_params:
model: together_ai/Salesforce/Llama-Rank-V1
api_key: os.environ/TOGETHERAI_API_KEY
- model_name: rerank-english-v3.0
- model_name: cohere-rerank-v3.5
litellm_params:
model: azure_ai/rerank-english-v3.0
model: azure_ai/cohere-rerank-v3.5
api_key: os.environ/AZURE_AI_API_KEY
api_base: os.environ/AZURE_AI_API_BASE
```
@ -384,7 +384,7 @@ curl http://0.0.0.0:4000/rerank \
-H "Authorization: Bearer sk-1234" \
-H "Content-Type: application/json" \
-d '{
"model": "rerank-english-v3.0",
"model": "cohere-rerank-v3.5",
"query": "What is the capital of the United States?",
"documents": [
"Carson City is the capital city of the American state of Nevada.",

View file

@ -196,7 +196,51 @@ for chunk in stream:
</TabItem>
</Tabs>
## Provider-specific Parameters
Any non-openai parameters will be passed to the agent as custom parameters.
<Tabs>
<TabItem value="sdk" label="SDK">
```python showLineNumbers title="Using custom parameters"
from litellm import completion
response = litellm.completion(
model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW",
messages=[
{
"role": "user",
"content": "Hi who is ishaan cto of litellm, tell me 10 things about him",
}
],
invocationId="my-test-invocation-id", # PROVIDER-SPECIFIC VALUE
)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
```yaml showLineNumbers title="LiteLLM Proxy Configuration"
model_list:
- model_name: bedrock-agent-1
litellm_params:
model: bedrock/agent/L1RT58GYRW/MFPSBCXYTW
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: us-west-2
invocationId: my-test-invocation-id
```
</TabItem>
</Tabs>
## Further Reading
- [AWS Bedrock Agents Documentation](https://aws.amazon.com/bedrock/agents/)
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)

View file

@ -0,0 +1,186 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# GitHub Copilot
https://docs.github.com/en/copilot
:::tip
**We support GitHub Copilot Chat API with automatic authentication handling**
:::
| Property | Details |
|-------|-------|
| Description | GitHub Copilot Chat API provides access to GitHub's AI-powered coding assistant. |
| Provider Route on LiteLLM | `github_copilot/` |
| Supported Endpoints | `/chat/completions` |
| API Reference | [GitHub Copilot docs](https://docs.github.com/en/copilot) |
## Authentication
GitHub Copilot uses OAuth device flow for authentication. On first use, you'll be prompted to authenticate via GitHub:
1. LiteLLM will display a device code and verification URL
2. Visit the URL and enter the code to authenticate
3. Your credentials will be stored locally for future use
## Usage - LiteLLM Python SDK
### Chat Completion
```python showLineNumbers title="GitHub Copilot Chat Completion"
from litellm import completion
response = completion(
model="github_copilot/gpt-4",
messages=[{"role": "user", "content": "Write a Python function to calculate fibonacci numbers"}],
extra_headers={
"editor-version": "vscode/1.85.1",
"Copilot-Integration-Id": "vscode-chat"
}
)
print(response)
```
```python showLineNumbers title="GitHub Copilot Chat Completion - Streaming"
from litellm import completion
stream = completion(
model="github_copilot/gpt-4",
messages=[{"role": "user", "content": "Explain async/await in Python"}],
stream=True,
extra_headers={
"editor-version": "vscode/1.85.1",
"Copilot-Integration-Id": "vscode-chat"
}
)
for chunk in stream:
if chunk.choices[0].delta.content is not None:
print(chunk.choices[0].delta.content, end="")
```
## Usage - LiteLLM Proxy
Add the following to your LiteLLM Proxy configuration file:
```yaml showLineNumbers title="config.yaml"
model_list:
- model_name: github_copilot/gpt-4
litellm_params:
model: github_copilot/gpt-4
```
Start your LiteLLM Proxy server:
```bash showLineNumbers title="Start LiteLLM Proxy"
litellm --config config.yaml
# RUNNING on http://0.0.0.0:4000
```
<Tabs>
<TabItem value="openai-sdk" label="OpenAI SDK">
```python showLineNumbers title="GitHub Copilot via Proxy - Non-streaming"
from openai import OpenAI
# Initialize client with your proxy URL
client = OpenAI(
base_url="http://localhost:4000", # Your proxy URL
api_key="your-proxy-api-key" # Your proxy API key
)
# Non-streaming response
response = client.chat.completions.create(
model="github_copilot/gpt-4",
messages=[{"role": "user", "content": "How do I optimize this SQL query?"}],
extra_headers={
"editor-version": "vscode/1.85.1",
"Copilot-Integration-Id": "vscode-chat"
}
)
print(response.choices[0].message.content)
```
</TabItem>
<TabItem value="litellm-sdk" label="LiteLLM SDK">
```python showLineNumbers title="GitHub Copilot via Proxy - LiteLLM SDK"
import litellm
# Configure LiteLLM to use your proxy
response = litellm.completion(
model="litellm_proxy/github_copilot/gpt-4",
messages=[{"role": "user", "content": "Review this code for bugs"}],
api_base="http://localhost:4000",
api_key="your-proxy-api-key",
extra_headers={
"editor-version": "vscode/1.85.1",
"Copilot-Integration-Id": "vscode-chat"
}
)
print(response.choices[0].message.content)
```
</TabItem>
<TabItem value="curl" label="cURL">
```bash showLineNumbers title="GitHub Copilot via Proxy - cURL"
curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer your-proxy-api-key" \
-H "editor-version: vscode/1.85.1" \
-H "Copilot-Integration-Id: vscode-chat" \
-d '{
"model": "github_copilot/gpt-4",
"messages": [{"role": "user", "content": "Explain this error message"}]
}'
```
</TabItem>
</Tabs>
## Getting Started
1. Ensure you have GitHub Copilot access (paid GitHub subscription required)
2. Run your first LiteLLM request - you'll be prompted to authenticate
3. Follow the device flow authentication process
4. Start making requests to GitHub Copilot through LiteLLM
## Configuration
### Environment Variables
You can customize token storage locations:
```bash showLineNumbers title="Environment Variables"
# Optional: Custom token directory
export GITHUB_COPILOT_TOKEN_DIR="~/.config/litellm/github_copilot"
# Optional: Custom access token file name
export GITHUB_COPILOT_ACCESS_TOKEN_FILE="access-token"
# Optional: Custom API key file name
export GITHUB_COPILOT_API_KEY_FILE="api-key.json"
```
### Headers
GitHub Copilot supports various editor-specific headers:
```python showLineNumbers title="Common Headers"
extra_headers = {
"editor-version": "vscode/1.85.1", # Editor version
"editor-plugin-version": "copilot/1.155.0", # Plugin version
"Copilot-Integration-Id": "vscode-chat", # Integration ID
"user-agent": "GithubCopilot/1.155.0" # User agent
}
```

View file

@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# VertexAI [Anthropic, Gemini, Model Garden]
# VertexAI [Gemini]
## Overview
@ -1208,534 +1208,6 @@ os.environ["VERTEXAI_LOCATION"] = "us-central1 # Your Location
# set directly on module
litellm.vertex_location = "us-central1 # Your Location
```
## Anthropic
| Model Name | Function Call |
|------------------|--------------------------------------|
| claude-3-opus@20240229 | `completion('vertex_ai/claude-3-opus@20240229', messages)` |
| claude-3-5-sonnet@20240620 | `completion('vertex_ai/claude-3-5-sonnet@20240620', messages)` |
| claude-3-sonnet@20240229 | `completion('vertex_ai/claude-3-sonnet@20240229', messages)` |
| claude-3-haiku@20240307 | `completion('vertex_ai/claude-3-haiku@20240307', messages)` |
| claude-3-7-sonnet@20250219 | `completion('vertex_ai/claude-3-7-sonnet@20250219', messages)` |
### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "claude-3-sonnet@20240229"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
temperature=0.7,
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: anthropic-vertex
litellm_params:
model: vertex_ai/claude-3-sonnet@20240229
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: anthropic-vertex
litellm_params:
model: vertex_ai/claude-3-sonnet@20240229
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "anthropic-vertex", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
### Usage - `thinking` / `reasoning_content`
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
resp = completion(
model="vertex_ai/claude-3-7-sonnet-20250219",
messages=[{"role": "user", "content": "What is the capital of France?"}],
thinking={"type": "enabled", "budget_tokens": 1024},
)
```
</TabItem>
<TabItem value="proxy" label="PROXY">
1. Setup config.yaml
```yaml
- model_name: claude-3-7-sonnet-20250219
litellm_params:
model: vertex_ai/claude-3-7-sonnet-20250219
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
2. Start proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
```bash
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
-d '{
"model": "claude-3-7-sonnet-20250219",
"messages": [{"role": "user", "content": "What is the capital of France?"}],
"thinking": {"type": "enabled", "budget_tokens": 1024}
}'
```
</TabItem>
</Tabs>
**Expected Response**
```python
ModelResponse(
id='chatcmpl-c542d76d-f675-4e87-8e5f-05855f5d0f5e',
created=1740470510,
model='claude-3-7-sonnet-20250219',
object='chat.completion',
system_fingerprint=None,
choices=[
Choices(
finish_reason='stop',
index=0,
message=Message(
content="The capital of France is Paris.",
role='assistant',
tool_calls=None,
function_call=None,
provider_specific_fields={
'citations': None,
'thinking_blocks': [
{
'type': 'thinking',
'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
'signature': 'EuYBCkQYAiJAy6...'
}
]
}
),
thinking_blocks=[
{
'type': 'thinking',
'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
'signature': 'EuYBCkQYAiJAy6AGB...'
}
],
reasoning_content='The capital of France is Paris. This is a very straightforward factual question.'
)
],
usage=Usage(
completion_tokens=68,
prompt_tokens=42,
total_tokens=110,
completion_tokens_details=None,
prompt_tokens_details=PromptTokensDetailsWrapper(
audio_tokens=None,
cached_tokens=0,
text_tokens=None,
image_tokens=None
),
cache_creation_input_tokens=0,
cache_read_input_tokens=0
)
)
```
## Meta/Llama API
| Model Name | Function Call |
|------------------|--------------------------------------|
| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "meta/llama3-405b-instruct-maas"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: anthropic-llama
litellm_params:
model: vertex_ai/meta/llama3-405b-instruct-maas
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: anthropic-llama
litellm_params:
model: vertex_ai/meta/llama3-405b-instruct-maas
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "anthropic-llama", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
## Mistral API
[**Supported OpenAI Params**](https://github.com/BerriAI/litellm/blob/e0f3cd580cb85066f7d36241a03c30aa50a8a31d/litellm/llms/openai.py#L137)
| Model Name | Function Call |
|------------------|--------------------------------------|
| mistral-large@latest | `completion('vertex_ai/mistral-large@latest', messages)` |
| mistral-large@2407 | `completion('vertex_ai/mistral-large@2407', messages)` |
| mistral-nemo@latest | `completion('vertex_ai/mistral-nemo@latest', messages)` |
| codestral@latest | `completion('vertex_ai/codestral@latest', messages)` |
| codestral@@2405 | `completion('vertex_ai/codestral@2405', messages)` |
### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "mistral-large@2407"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: vertex-mistral
litellm_params:
model: vertex_ai/mistral-large@2407
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: vertex-mistral
litellm_params:
model: vertex_ai/mistral-large@2407
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "vertex-mistral", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
### Usage - Codestral FIM
Call Codestral on VertexAI via the OpenAI [`/v1/completion`](https://platform.openai.com/docs/api-reference/completions/create) endpoint for FIM tasks.
Note: You can also call Codestral via `/chat/completion`.
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
# os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
# OR run `!gcloud auth print-access-token` in your terminal
model = "codestral@2405"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = text_completion(
model="vertex_ai/" + model,
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
prompt="def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
suffix="return True", # optional
temperature=0, # optional
top_p=1, # optional
max_tokens=10, # optional
min_tokens=10, # optional
seed=10, # optional
stop=["return"], # optional
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: vertex-codestral
litellm_params:
model: vertex_ai/codestral@2405
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: vertex-codestral
litellm_params:
model: vertex_ai/codestral@2405
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl -X POST 'http://0.0.0.0:4000/completions' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-d '{
"model": "vertex-codestral", # 👈 the 'model_name' in config
"prompt": "def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
"suffix":"return True", # optional
"temperature":0, # optional
"top_p":1, # optional
"max_tokens":10, # optional
"min_tokens":10, # optional
"seed":10, # optional
"stop":["return"], # optional
}'
```
</TabItem>
</Tabs>
## AI21 Models
| Model Name | Function Call |
|------------------|--------------------------------------|
| jamba-1.5-mini@001 | `completion(model='vertex_ai/jamba-1.5-mini@001', messages)` |
| jamba-1.5-large@001 | `completion(model='vertex_ai/jamba-1.5-large@001', messages)` |
### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "meta/jamba-1.5-mini@001"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: jamba-1.5-mini
litellm_params:
model: vertex_ai/jamba-1.5-mini@001
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: jamba-1.5-large
litellm_params:
model: vertex_ai/jamba-1.5-large@001
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "jamba-1.5-large",
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
## Gemini Pro
| Model Name | Function Call |
@ -1832,119 +1304,6 @@ curl --location 'https://0.0.0.0:4000/v1/chat/completions' \
</TabItem>
</Tabs>
## Model Garden
:::tip
All OpenAI compatible models from Vertex Model Garden are supported.
:::
#### Using Model Garden
**Almost all Vertex Model Garden models are OpenAI compatible.**
<Tabs>
<TabItem value="openai" label="OpenAI Compatible Models">
| Property | Details |
|----------|---------|
| Provider Route | `vertex_ai/openai/{MODEL_ID}` |
| Vertex Documentation | [Vertex Model Garden - OpenAI Chat Completions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gradio_streaming_chat_completions.ipynb), [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
| Supported Operations | `/chat/completions`, `/embeddings` |
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
## set ENV variables
os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
os.environ["VERTEXAI_LOCATION"] = "us-central1"
response = completion(
model="vertex_ai/openai/<your-endpoint-id>",
messages=[{ "content": "Hello, how are you?","role": "user"}]
)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: llama3-1-8b-instruct
litellm_params:
model: vertex_ai/openai/5464397967697903616
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "llama3-1-8b-instruct", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
</TabItem>
<TabItem value="non-openai" label="Non-OpenAI Compatible Models">
```python
from litellm import completion
import os
## set ENV variables
os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
os.environ["VERTEXAI_LOCATION"] = "us-central1"
response = completion(
model="vertex_ai/<your-endpoint-id>",
messages=[{ "content": "Hello, how are you?","role": "user"}]
)
```
</TabItem>
</Tabs>
## Gemini Pro Vision
| Model Name | Function Call |
|------------------|--------------------------------------|

View file

@ -0,0 +1,670 @@
import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Vertex AI - Anthropic, DeepSeek, Model Garden
## Supported Partner Providers
| Provider | LiteLLM Route | Vertex Documentation |
|----------|---------------|---------------|
| Anthropic (Claude) | `vertex_ai/claude-*` | [Vertex AI - Anthropic Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/use-claude) |
| DeepSeek | `vertex_ai/deepseek-ai/{MODEL}` | [Vertex AI - DeepSeek Models](https://cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek) |
| Meta/Llama | `vertex_ai/meta/{MODEL}` | [Vertex AI - Meta Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/llama) |
| Mistral | `vertex_ai/mistral-*` | [Vertex AI - Mistral Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/mistral) |
| AI21 (Jamba) | `vertex_ai/jamba-*` | [Vertex AI - AI21 Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/ai21) |
| Model Garden | `vertex_ai/openai/{MODEL_ID}` or `vertex_ai/{MODEL_ID}` | [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
## Vertex AI - Anthropic (Claude)
| Model Name | Function Call |
|------------------|--------------------------------------|
| claude-3-opus@20240229 | `completion('vertex_ai/claude-3-opus@20240229', messages)` |
| claude-3-5-sonnet@20240620 | `completion('vertex_ai/claude-3-5-sonnet@20240620', messages)` |
| claude-3-sonnet@20240229 | `completion('vertex_ai/claude-3-sonnet@20240229', messages)` |
| claude-3-haiku@20240307 | `completion('vertex_ai/claude-3-haiku@20240307', messages)` |
| claude-3-7-sonnet@20250219 | `completion('vertex_ai/claude-3-7-sonnet@20250219', messages)` |
#### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "claude-3-sonnet@20240229"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
temperature=0.7,
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: anthropic-vertex
litellm_params:
model: vertex_ai/claude-3-sonnet@20240229
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: anthropic-vertex
litellm_params:
model: vertex_ai/claude-3-sonnet@20240229
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "anthropic-vertex", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
#### Usage - `thinking` / `reasoning_content`
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
resp = completion(
model="vertex_ai/claude-3-7-sonnet-20250219",
messages=[{"role": "user", "content": "What is the capital of France?"}],
thinking={"type": "enabled", "budget_tokens": 1024},
)
```
</TabItem>
<TabItem value="proxy" label="PROXY">
1. Setup config.yaml
```yaml
- model_name: claude-3-7-sonnet-20250219
litellm_params:
model: vertex_ai/claude-3-7-sonnet-20250219
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
2. Start proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
```bash
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
-d '{
"model": "claude-3-7-sonnet-20250219",
"messages": [{"role": "user", "content": "What is the capital of France?"}],
"thinking": {"type": "enabled", "budget_tokens": 1024}
}'
```
</TabItem>
</Tabs>
**Expected Response**
```python
ModelResponse(
id='chatcmpl-c542d76d-f675-4e87-8e5f-05855f5d0f5e',
created=1740470510,
model='claude-3-7-sonnet-20250219',
object='chat.completion',
system_fingerprint=None,
choices=[
Choices(
finish_reason='stop',
index=0,
message=Message(
content="The capital of France is Paris.",
role='assistant',
tool_calls=None,
function_call=None,
provider_specific_fields={
'citations': None,
'thinking_blocks': [
{
'type': 'thinking',
'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
'signature': 'EuYBCkQYAiJAy6...'
}
]
}
),
thinking_blocks=[
{
'type': 'thinking',
'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
'signature': 'EuYBCkQYAiJAy6AGB...'
}
],
reasoning_content='The capital of France is Paris. This is a very straightforward factual question.'
)
],
usage=Usage(
completion_tokens=68,
prompt_tokens=42,
total_tokens=110,
completion_tokens_details=None,
prompt_tokens_details=PromptTokensDetailsWrapper(
audio_tokens=None,
cached_tokens=0,
text_tokens=None,
image_tokens=None
),
cache_creation_input_tokens=0,
cache_read_input_tokens=0
)
)
```
## VertexAI DeepSeek
| Property | Details |
|----------|---------|
| Provider Route | `vertex_ai/deepseek-ai/{MODEL}` |
| Vertex Documentation | [Vertex AI - DeepSeek Models](https://cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek) |
#### Usage
**LiteLLM Supports all Vertex AI DeepSeek Models.** Ensure you use the `vertex_ai/deepseek-ai/` prefix for all Vertex AI DeepSeek models.
| Model Name | Usage |
|------------------|------------------------------|
| vertex_ai/deepseek-ai/deepseek-r1-0528-maas | `completion('vertex_ai/deepseek-ai/deepseek-r1-0528-maas', messages)` |
## VertexAI Meta/Llama API
| Model Name | Function Call |
|------------------|--------------------------------------|
| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
#### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "meta/llama3-405b-instruct-maas"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: anthropic-llama
litellm_params:
model: vertex_ai/meta/llama3-405b-instruct-maas
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: anthropic-llama
litellm_params:
model: vertex_ai/meta/llama3-405b-instruct-maas
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "anthropic-llama", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
## VertexAI Mistral API
[**Supported OpenAI Params**](https://github.com/BerriAI/litellm/blob/e0f3cd580cb85066f7d36241a03c30aa50a8a31d/litellm/llms/openai.py#L137)
| Model Name | Function Call |
|------------------|--------------------------------------|
| mistral-large@latest | `completion('vertex_ai/mistral-large@latest', messages)` |
| mistral-large@2407 | `completion('vertex_ai/mistral-large@2407', messages)` |
| mistral-nemo@latest | `completion('vertex_ai/mistral-nemo@latest', messages)` |
| codestral@latest | `completion('vertex_ai/codestral@latest', messages)` |
| codestral@@2405 | `completion('vertex_ai/codestral@2405', messages)` |
#### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "mistral-large@2407"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: vertex-mistral
litellm_params:
model: vertex_ai/mistral-large@2407
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: vertex-mistral
litellm_params:
model: vertex_ai/mistral-large@2407
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "vertex-mistral", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
#### Usage - Codestral FIM
Call Codestral on VertexAI via the OpenAI [`/v1/completion`](https://platform.openai.com/docs/api-reference/completions/create) endpoint for FIM tasks.
Note: You can also call Codestral via `/chat/completion`.
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
# os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
# OR run `!gcloud auth print-access-token` in your terminal
model = "codestral@2405"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = text_completion(
model="vertex_ai/" + model,
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
prompt="def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
suffix="return True", # optional
temperature=0, # optional
top_p=1, # optional
max_tokens=10, # optional
min_tokens=10, # optional
seed=10, # optional
stop=["return"], # optional
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: vertex-codestral
litellm_params:
model: vertex_ai/codestral@2405
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: vertex-codestral
litellm_params:
model: vertex_ai/codestral@2405
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl -X POST 'http://0.0.0.0:4000/completions' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-d '{
"model": "vertex-codestral", # 👈 the 'model_name' in config
"prompt": "def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
"suffix":"return True", # optional
"temperature":0, # optional
"top_p":1, # optional
"max_tokens":10, # optional
"min_tokens":10, # optional
"seed":10, # optional
"stop":["return"], # optional
}'
```
</TabItem>
</Tabs>
## VertexAI AI21 Models
| Model Name | Function Call |
|------------------|--------------------------------------|
| jamba-1.5-mini@001 | `completion(model='vertex_ai/jamba-1.5-mini@001', messages)` |
| jamba-1.5-large@001 | `completion(model='vertex_ai/jamba-1.5-large@001', messages)` |
#### Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
model = "meta/jamba-1.5-mini@001"
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
response = completion(
model="vertex_ai/" + model,
messages=[{"role": "user", "content": "hi"}],
vertex_ai_project=vertex_ai_project,
vertex_ai_location=vertex_ai_location,
)
print("\nModel Response", response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: jamba-1.5-mini
litellm_params:
model: vertex_ai/jamba-1.5-mini@001
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
- model_name: jamba-1.5-large
litellm_params:
model: vertex_ai/jamba-1.5-large@001
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-west-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "jamba-1.5-large",
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
## Model Garden
:::tip
All OpenAI compatible models from Vertex Model Garden are supported.
:::
#### Using Model Garden
**Almost all Vertex Model Garden models are OpenAI compatible.**
<Tabs>
<TabItem value="openai" label="OpenAI Compatible Models">
| Property | Details |
|----------|---------|
| Provider Route | `vertex_ai/openai/{MODEL_ID}` |
| Vertex Documentation | [SDK for Deploy & OpenAI Chat Completions](https://github.com/GoogleCloudPlatform/generative-ai/blob/main/open-models/get_started_with_model_garden_sdk.ipynb), [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
| Supported Operations | `/chat/completions`, `/embeddings` |
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
## set ENV variables
os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
os.environ["VERTEXAI_LOCATION"] = "us-central1"
response = completion(
model="vertex_ai/openai/<your-endpoint-id>",
messages=[{ "content": "Hello, how are you?","role": "user"}]
)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
**1. Add to config**
```yaml
model_list:
- model_name: llama3-1-8b-instruct
litellm_params:
model: vertex_ai/openai/5464397967697903616
vertex_ai_project: "my-test-project"
vertex_ai_location: "us-east-1"
```
**2. Start proxy**
```bash
litellm --config /path/to/config.yaml
# RUNNING at http://0.0.0.0:4000
```
**3. Test it!**
```bash
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "llama3-1-8b-instruct", # 👈 the 'model_name' in config
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
</TabItem>
<TabItem value="non-openai" label="Non-OpenAI Compatible Models">
```python
from litellm import completion
import os
## set ENV variables
os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
os.environ["VERTEXAI_LOCATION"] = "us-central1"
response = completion(
model="vertex_ai/<your-endpoint-id>",
messages=[{ "content": "Hello, how are you?","role": "user"}]
)
```
</TabItem>
</Tabs>

View file

@ -0,0 +1,56 @@
# CLI Authentication
Use the litellm cli to authenticate to the LiteLLM Gateway. This is great if you're trying to give a large number of developers self-serve access to the LiteLLM Gateway.
## Demo
<iframe width="840" height="500" src="https://www.loom.com/embed/87c5d243cde642ff942783024ff037e3" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
## Usage
1. **Install the CLI**
If you have [uv](https://github.com/astral-sh/uv) installed, you can try this:
```shell
uv tool install 'litellm[proxy]'
```
If that works, you'll see something like this:
```shell
...
Installed 2 executables: litellm, litellm-proxy
```
and now you can use the tool by just typing `litellm-proxy` in your terminal:
```shell
litellm-proxy
```
2. **Set up environment variables**
```bash
export LITELLM_PROXY_URL=http://localhost:4000
```
*(Replace with your actual proxy URL)*
3. **Login**
```shell
litellm-proxy login
```
This will open a browser window to authenticate. If you have connected LiteLLM Proxy to your SSO provider, you should be able to login with your SSO credentials. Once logged in, you can use the CLI to make requests to the LiteLLM Gateway.
4. **Make a test request to view models**
```shell
litellm-proxy models list
```
This will list all the models available to you.

View file

@ -319,6 +319,7 @@ router_settings:
| ATHINA_API_KEY | API key for Athina service
| ATHINA_BASE_URL | Base URL for Athina service (defaults to `https://log.athina.ai`)
| AUTH_STRATEGY | Strategy used for authentication (e.g., OAuth, API key)
| ANTHROPIC_API_KEY | API key for Anthropic service
| AWS_ACCESS_KEY_ID | Access Key ID for AWS services
| AWS_PROFILE_NAME | AWS CLI profile name to be used
| AWS_REGION_NAME | Default AWS region for service interactions
@ -415,6 +416,8 @@ router_settings:
| DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND | Default price per second for Replicate GPU. Default is 0.001400
| DEFAULT_REPLICATE_POLLING_DELAY_SECONDS | Default delay in seconds for Replicate polling. Default is 1
| DEFAULT_REPLICATE_POLLING_RETRIES | Default number of retries for Replicate polling. Default is 5
| DEFAULT_SQS_BATCH_SIZE | Default batch size for SQS logging. Default is 512
| DEFAULT_SQS_FLUSH_INTERVAL_SECONDS | Default flush interval for SQS logging. Default is 10
| DEFAULT_S3_BATCH_SIZE | Default batch size for S3 logging. Default is 512
| DEFAULT_S3_FLUSH_INTERVAL_SECONDS | Default flush interval for S3 logging. Default is 10
| DEFAULT_SLACK_ALERTING_THRESHOLD | Default threshold for Slack alerting. Default is 300
@ -431,6 +434,9 @@ router_settings:
| DOCS_URL | The path to the Swagger API documentation. **By default this is "/"**
| EMAIL_LOGO_URL | URL for the logo used in emails
| EMAIL_SUPPORT_CONTACT | Support contact email address
| EMAIL_SIGNATURE | Custom HTML footer/signature for all emails. Can include HTML tags for formatting and links.
| EMAIL_SUBJECT_INVITATION | Custom subject template for invitation emails.
| EMAIL_SUBJECT_KEY_CREATED | Custom subject template for key creation emails.
| EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING | Flag to enable new multi-instance rate limiting. **Default is False**
| FIREWORKS_AI_4_B | Size parameter for Fireworks AI 4B model. Default is 4
| FIREWORKS_AI_16_B | Size parameter for Fireworks AI 16B model. Default is 16
@ -469,12 +475,16 @@ router_settings:
| GALILEO_PASSWORD | Password for Galileo authentication
| GALILEO_PROJECT_ID | Project ID for Galileo usage
| GALILEO_USERNAME | Username for Galileo authentication
| GITHUB_COPILOT_TOKEN_DIR | Directory to store GitHub Copilot token for `github_copilot` llm provider
| GITHUB_COPILOT_API_KEY_FILE | File to store GitHub Copilot API key for `github_copilot` llm provider
| GITHUB_COPILOT_ACCESS_TOKEN_FILE | File to store GitHub Copilot access token for `github_copilot` llm provider
| GREENSCALE_API_KEY | API key for Greenscale service
| GREENSCALE_ENDPOINT | Endpoint URL for Greenscale service
| GOOGLE_APPLICATION_CREDENTIALS | Path to Google Cloud credentials JSON file
| GOOGLE_CLIENT_ID | Client ID for Google OAuth
| GOOGLE_CLIENT_SECRET | Client secret for Google OAuth
| GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS
| GUARDRAILS_AI_API_BASE | Base URL for Guardrails AI API
| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60
| HF_API_BASE | Base URL for Hugging Face API
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
@ -513,6 +523,10 @@ router_settings:
| LANGSMITH_PROJECT | Project name for Langsmith integration
| LANGSMITH_SAMPLING_RATE | Sampling rate for Langsmith logging
| LANGTRACE_API_KEY | API key for Langtrace service
| LASSO_API_BASE | Base URL for Lasso API
| LASSO_API_KEY | API key for Lasso service
| LASSO_USER_ID | User ID for Lasso service
| LASSO_CONVERSATION_ID | Conversation ID for Lasso service
| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16
| LITERAL_API_KEY | API key for Literal integration
| LITERAL_API_URL | API URL for Literal service
@ -529,6 +543,7 @@ router_settings:
| LITELLM_LICENSE | License key for LiteLLM usage
| LITELLM_LOCAL_MODEL_COST_MAP | Local configuration for model cost mapping in LiteLLM
| LITELLM_LOG | Enable detailed logging for LiteLLM
| LITELLM_MASTER_KEY | Master key for proxy authentication
| LITELLM_MODE | Operating mode for LiteLLM (e.g., production, development)
| LITELLM_RATE_LIMIT_WINDOW_SIZE | Rate limit window size for LiteLLM. Default is 60
| LITELLM_SALT_KEY | Salt key for encryption in LiteLLM
@ -586,6 +601,8 @@ router_settings:
| OTEL_SERVICE_NAME | Service name identifier for OpenTelemetry
| OTEL_TRACER_NAME | Tracer name for OpenTelemetry tracing
| PAGERDUTY_API_KEY | API key for PagerDuty Alerting
| PANW_PRISMA_AIRS_API_KEY | API key for PANW Prisma AIRS service
| PANW_PRISMA_AIRS_API_BASE | Base URL for PANW Prisma AIRS service
| PHOENIX_API_KEY | API key for Arize Phoenix
| PHOENIX_COLLECTOR_ENDPOINT | API endpoint for Arize Phoenix
| PHOENIX_COLLECTOR_HTTP_ENDPOINT | API http endpoint for Arize Phoenix
@ -603,7 +620,6 @@ router_settings:
| PROXY_BUDGET_RESCHEDULER_MAX_TIME | Maximum time in seconds to wait before checking database for budget resets. Default is 605
| PROXY_BUDGET_RESCHEDULER_MIN_TIME | Minimum time in seconds to wait before checking database for budget resets. Default is 597
| PROXY_LOGOUT_URL | URL for logging out of the proxy service
| LITELLM_MASTER_KEY | Master key for proxy authentication
| QDRANT_API_BASE | Base URL for Qdrant API
| QDRANT_API_KEY | API key for Qdrant service
| QDRANT_SCALAR_QUANTILE | Scalar quantile for Qdrant operations. Default is 0.99
@ -638,6 +654,7 @@ router_settings:
| SSL_CERTIFICATE | Path to the SSL certificate file
| SSL_SECURITY_LEVEL | [BETA] Security level for SSL/TLS connections. E.g. `DEFAULT@SECLEVEL=1`
| SSL_VERIFY | Flag to enable or disable SSL certificate verification
| SSL_CERT_FILE | Path to the SSL certificate file for custom CA bundle
| SUPABASE_KEY | API key for Supabase service
| SUPABASE_URL | Base URL for Supabase instance
| STORE_MODEL_IN_DB | If true, enables storing model + credential information in the DB.

View file

@ -12,6 +12,9 @@ Requires v1.72.3 or higher.
:::
Limitations:
- This does not work in [litellm non-root](./deploy#non-root---without-internet-connection) images, as it requires write access to the UI files.
## Usage
### 1. Set `SERVER_ROOT_PATH` in your .env

View file

@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# 🙋‍♂️ Customers / End-User Budgets
# Customers / End-User Budgets
Track spend, set budgets for your customers.
@ -136,7 +136,7 @@ Create / Update a customer with budget
curl -X POST 'http://0.0.0.0:4000/customer/new'
-H 'Authorization: Bearer sk-1234'
-H 'Content-Type: application/json'
-D '{
-d '{
"user_id" : "my-customer-id",
"max_budget": "0", # 👈 CAN BE FLOAT
}'

View file

@ -237,6 +237,9 @@ spec:
containers:
- name: litellm
image: ghcr.io/berriai/litellm:main-stable # it is recommended to fix a version generally
args:
- "--config"
- "/app/proxy_server_config.yaml"
ports:
- containerPort: 4000
volumeMounts:
@ -386,7 +389,8 @@ spec:
- "/app/proxy_config.yaml" # Update the path to mount the config file
volumeMounts: # Define volume mount for proxy_config.yaml
- name: config-volume
mountPath: /app
mountPath: /app/proxy_config.yaml
subPath: config.yaml # Specify the field under data of the ConfigMap litellm-config
readOnly: true
livenessProbe:
httpGet:

View file

@ -124,9 +124,7 @@ On the Create Key Modal, Select Advanced Settings > Set Send Email to True.
/>
## Customizing Email Branding
## Email Customization
:::info
@ -134,13 +132,96 @@ Customizing Email Branding is an Enterprise Feature [Get in touch with us for a
:::
LiteLLM allows you to customize the:
- Logo on the Email
- Email support contact
LiteLLM allows you to customize various aspects of your email notifications. Below is a complete reference of all customizable fields:
Set the following in your env to customize your emails
| Field | Environment Variable | Type | Default Value | Example | Description |
|-------|-------------------|------|---------------|---------|-------------|
| Logo URL | `EMAIL_LOGO_URL` | string | LiteLLM logo | `"https://your-company.com/logo.png"` | Public URL to your company logo |
| Support Contact | `EMAIL_SUPPORT_CONTACT` | string | support@berri.ai | `"support@your-company.com"` | Email address for user support |
| Email Signature | `EMAIL_SIGNATURE` | string (HTML) | Standard LiteLLM footer | `"<p>Best regards,<br/>Your Team</p><p><a href='https://your-company.com'>Visit us</a></p>"` | HTML-formatted footer for all emails |
| Invitation Subject | `EMAIL_SUBJECT_INVITATION` | string | "LiteLLM: New User Invitation" | `"Welcome to Your Company!"` | Subject line for invitation emails |
| Key Creation Subject | `EMAIL_SUBJECT_KEY_CREATED` | string | "LiteLLM: API Key Created" | `"Your New API Key is Ready"` | Subject line for key creation emails |
```shell
EMAIL_LOGO_URL="https://litellm-listing.s3.amazonaws.com/litellm_logo.png" # public url to your logo
EMAIL_SUPPORT_CONTACT="support@berri.ai" # Your company support email
## HTML Support in Email Signature
The `EMAIL_SIGNATURE` field supports HTML formatting for rich, branded email footers. Here's an example of what you can include:
```html
<p>Best regards,<br/>The LiteLLM Team</p>
<p>
<a href='https://docs.litellm.ai'>Documentation</a> |
<a href='https://github.com/BerriAI/litellm'>GitHub</a>
</p>
<p style='font-size: 12px; color: #666;'>
This is an automated message from LiteLLM Proxy
</p>
```
Supported HTML features:
- Text formatting (bold, italic, etc.)
- Line breaks (`<br/>`)
- Links (`<a href='...'>`)
- Paragraphs (`<p>`)
- Basic inline styling
- Company information and social media links
- Legal disclaimers or terms of service links
## Environment Variables
You can customize the following aspects of emails through environment variables:
```bash
# Email Branding
EMAIL_LOGO_URL="https://your-company.com/logo.png" # Custom logo URL
EMAIL_SUPPORT_CONTACT="support@your-company.com" # Support contact email
EMAIL_SIGNATURE="<p>Best regards,<br/>Your Company Team</p><p><a href='https://your-company.com'>Visit our website</a></p>" # Custom HTML footer/signature
# Email Subject Lines
EMAIL_SUBJECT_INVITATION="Welcome to Your Company!" # Subject for invitation emails
EMAIL_SUBJECT_KEY_CREATED="Your API Key is Ready" # Subject for key creation emails
```
## HTML Support in Email Signature
The `EMAIL_SIGNATURE` environment variable supports HTML formatting, allowing you to create rich, branded email footers. You can include:
- Text formatting (bold, italic, etc.)
- Line breaks using `<br/>`
- Links using `<a href='...'>`
- Paragraphs using `<p>`
- Company information and social media links
- Legal disclaimers or terms of service links
Example HTML signature:
```html
<p>Best regards,<br/>The LiteLLM Team</p>
<p>
<a href='https://docs.litellm.ai'>Documentation</a> |
<a href='https://github.com/BerriAI/litellm'>GitHub</a>
</p>
<p style='font-size: 12px; color: #666;'>
This is an automated message from LiteLLM Proxy
</p>
```
## Default Templates
If environment variables are not set, LiteLLM will use default templates:
- Default logo: LiteLLM logo
- Default support contact: support@berri.ai
- Default signature: Standard LiteLLM footer
- Default subjects: "LiteLLM: \{event_message\}" (replaced with actual event message)
## Template Variables
When setting custom email subjects, you can use template variables that will be replaced with actual values:
```bash
# Examples of template variable usage
EMAIL_SUBJECT_INVITATION="Welcome to \{company_name\}!"
EMAIL_SUBJECT_KEY_CREATED="Your \{company_name\} API Key"
```
The system will automatically replace `\{event_message\}` and other template variables with their actual values when sending emails.

View file

@ -216,7 +216,7 @@ If you need to switch `pii_masking` off for an API Key set `"permissions": {"pii
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"permissions": {"pii_masking": true}
}'
```

View file

@ -155,7 +155,7 @@ Use this to control what guardrails run per project. In this tutorial we only wa
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"guardrails": ["aporia-pre-guard", "aporia-post-guard"]
}
}'

View file

@ -74,7 +74,7 @@ Use this to control what guardrails run per project. In this tutorial we only wa
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"guardrails": ["guardrails_ai-guard"]
}
}'

View file

@ -126,3 +126,30 @@ curl -i http://localhost:4000/v1/chat/completions \
</Tabs>
## Supported Params
```yaml
guardrails:
- guardrail_name: "lakera-guard"
litellm_params:
guardrail: lakera_v2 # supported values: "aporia", "bedrock", "lakera"
mode: "during_call"
api_key: os.environ/LAKERA_API_KEY
api_base: os.environ/LAKERA_API_BASE
### OPTIONAL ###
# project_id: Optional[str] = None,
# payload: Optional[bool] = True,
# breakdown: Optional[bool] = True,
# metadata: Optional[Dict] = None,
# dev_info: Optional[bool] = True,
```
- `api_base`: (Optional[str]) The base of the Lakera integration. Defaults to `https://api.lakera.ai`
- `api_key`: (str) The API Key for the Lakera integration.
- `project_id`: (Optional[str]) ID of the relevant project
- `payload`: (Optional[bool]) When true the response will return a payload object containing any PII, profanity or custom detector regex matches detected, along with their location within the contents.
- `breakdown`: (Optional[bool]) When true the response will return a breakdown list of the detectors that were run, as defined in the policy, and whether each of them detected something or not.
- `metadata`: (Optional[Dict]) Metadata tags can be attached to screening requests as an object that can contain any arbitrary key-value pairs.
- `dev_info`: (Optional[bool]) When true the response will return an object with developer information about the build of Lakera Guard.

View file

@ -421,7 +421,7 @@ Use this to control what guardrails run per API Key. In this tutorial we only wa
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"guardrails": ["aporia-pre-guard", "aporia-post-guard"]
}
}'

View file

@ -9,6 +9,7 @@ Log Proxy input, output, and exceptions using:
- Langfuse
- OpenTelemetry
- GCS, s3, Azure (Blob) Buckets
- AWS SQS
- Lunary
- MLflow
- Deepeval
@ -1384,6 +1385,75 @@ litellm_settings:
On s3 bucket, you will see the object key as `my-test-path/my-team-alias/...`
## AWS SQS
| Property | Details |
|----------|---------|
| Description | Log LLM Input/Output to AWS SQS Queue |
| AWS Docs on SQS | [AWS SQS](https://aws.amazon.com/sqs/) |
| Fields Logged to SQS | LiteLLM [Standard Logging Payload is logged for each LLM call](../proxy/logging_spec) |
Log LLM Logs to [AWS Simple Queue Service (SQS)](https://aws.amazon.com/sqs/)
We will use the litellm `--config` to set
- `litellm.callbacks = ["aws_sqs"]`
This will log all successful LLM calls to AWS SQS Queue
**Step 1** Set AWS Credentials in .env
```shell
AWS_ACCESS_KEY_ID = ""
AWS_SECRET_ACCESS_KEY = ""
AWS_REGION_NAME = ""
```
**Step 2**: Create a `config.yaml` file and set `litellm_settings`: `callbacks`
```yaml
model_list:
- model_name: gpt-4o
litellm_params:
model: gpt-4o
litellm_settings:
callbacks: ["aws_sqs"]
aws_sqs_callback_params:
sqs_queue_url: https://sqs.us-west-2.amazonaws.com/123456789012/my-queue # AWS SQS Queue URL
sqs_region_name: us-west-2 # AWS Region Name for SQS
sqs_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # use os.environ/<variable name> to pass environment variables. This is AWS Access Key ID for SQS
sqs_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for SQS
sqs_batch_size: 10 # [OPTIONAL] Number of messages to batch before sending (default: 10)
sqs_flush_interval: 30 # [OPTIONAL] Time in seconds to wait before flushing batch (default: 30)
```
**Step 3**: Start the proxy, make a test request
Start proxy
```shell
litellm --config config.yaml --debug
```
Test Request
```shell
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
"model": "gpt-4o",
"messages": [
{
"role": "user",
"content": "what llm are you"
}
]
}'
```
## Azure Blob Storage
Log LLM Logs to [Azure Data Lake Storage](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction)

View file

@ -147,7 +147,7 @@ print(file_response.text)
```python showLineNumbers title="create_batch.py"
...
client.batches.list(limit=10, extra_body={"target_model_names": "gpt-4o-batch"})
client.batches.list(limit=10, extra_query={"target_model_names": "gpt-4o-batch"})
```
### [Coming Soon] Cancel a batch

View file

@ -57,6 +57,42 @@ and more, as well as making chat and HTTP requests to the proxy server.
- If you see an error, check your environment variables and proxy server status.
## Authentication using CLI
You can use the CLI to authenticate to the LiteLLM Gateway. This is great if you're trying to give a large number of developers self-serve access to the LiteLLM Gateway.
:::info
For an indepth guide, see [CLI Authentication](./cli_sso).
:::
1. **Set up the proxy URL**
```bash
export LITELLM_PROXY_URL=http://localhost:4000
```
*(Replace with your actual proxy URL)*
2. **Login**
```bash
litellm-proxy login
```
This will open a browser window to authenticate. If you have connected LiteLLM Proxy to your SSO provider, you can login with your SSO credentials. Once logged in, you can use the CLI to make requests to the LiteLLM Gateway.
3. **Test your authentication**
```bash
litellm-proxy models list
```
This will list all the models available to you.
## Main Commands
### Models Management

View file

@ -64,9 +64,9 @@ Use this for for tracking per [user, key, team, etc.](virtual_keys)
| Metric Name | Description |
|----------------------|--------------------------------------|
| `litellm_spend_metric` | Total Spend, per `"user", "key", "model", "team", "end-user"` |
| `litellm_total_tokens` | input + output tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` |
| `litellm_input_tokens` | input tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` |
| `litellm_output_tokens` | output tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` |
| `litellm_total_tokens_metric` | input + output tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` |
| `litellm_input_tokens_metric` | input tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` |
| `litellm_output_tokens_metric` | output tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` |
### Team - Budget
@ -288,10 +288,11 @@ Control which labels are included for each metric to reduce cardinality:
litellm_settings:
callbacks: ["prometheus"]
prometheus_metrics_config:
- group: "spend_and_tokens"
- group: "token_consumption"
metrics:
- "litellm_spend_metric"
- "litellm_total_tokens"
- "litellm_input_tokens_metric"
- "litellm_output_tokens_metric"
- "litellm_total_tokens_metric"
include_labels:
- "model"
- "team"
@ -324,7 +325,6 @@ litellm_settings:
# Budget metrics with full label set
- group: "budget_tracking"
metrics:
- "litellm_spend_metric"
- "litellm_remaining_team_budget_metric"
include_labels:
- "team"
@ -385,7 +385,7 @@ Use these metrics to monitor the health of the DB Transaction Queue. Eg. Monitor
## **🔥 LiteLLM Maintained Grafana Dashboards **
## 🔥 LiteLLM Maintained Grafana Dashboards
Link to Grafana Dashboards maintained by LiteLLM

View file

@ -210,6 +210,7 @@ These are the params you can pass to the `litellm.completion` function in SDK an
```
prompt_id: str # required
prompt_variables: Optional[dict] # optional
prompt_version: Optional[int] # optional
langfuse_public_key: Optional[str] # optional
langfuse_secret: Optional[str] # optional
langfuse_secret_key: Optional[str] # optional

View file

@ -6,8 +6,27 @@ import Image from '@theme/IdealImage';
Use this if you want to create Virtual Keys that are not owned by a specific user but instead created for production projects
Why use a service account key?
- Prevent key from being deleted when user is deleted.
- Apply team limits, not team member limits to key.
## Usage
Use the `/key/service-account/generate` endpoint to generate a service account key.
```bash
curl -L -X POST 'http://localhost:4000/key/service-account/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-d '{
"team_id": "my-unique-team"
}'
```
## Example - require `user` param for all service account requests
### 1. Set settings for Service Accounts
Set `service_account_settings` if you want to create settings that only apply to service account keys

View file

@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# 💰 Setting Team Budgets
# Setting Team Budgets
Track spend, set budgets for your Internal Team
@ -318,7 +318,7 @@ curl -X POST 'http://0.0.0.0:4000/key/generate' \
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: sk-...' \ # 👈 key from step 2.
-D '{
-d '{
"model": "gpt-3.5-turbo",
"messages": [
{

View file

@ -4,52 +4,25 @@ import TabItem from '@theme/TabItem';
# Team/Key Based Logging
Allow each key/team to use their own Langfuse Project / custom callbacks
## Overview
**This allows you to do the following**
```
Allow each key/team to use their own Langfuse Project / custom callbacks. This enables granular control over logging and compliance requirements.
**Example Use Cases:**
```showLineNumbers title="Team Based Logging"
Team 1 -> Logs to Langfuse Project 1
Team 2 -> Logs to Langfuse Project 2
Team 3 -> Disabled Logging (for GDPR compliance)
```
## Team Based Logging
## Supported Logging Integrations
- `langfuse`
- `gcs_bucket`
- `langsmith`
- `arize`
### Setting Team Logging via `config.yaml`
Turn on/off logging and caching for a specific team id.
**Example:**
This config would send langfuse logs to 2 different langfuse projects, based on the team id
```yaml
litellm_settings:
default_team_settings:
- team_id: "dbe2f686-a686-4896-864a-4c3924458709"
success_callback: ["langfuse"]
langfuse_public_key: os.environ/LANGFUSE_PUB_KEY_1 # Project 1
langfuse_secret: os.environ/LANGFUSE_PRIVATE_KEY_1 # Project 1
- team_id: "06ed1e01-3fa7-4b9e-95bc-f2e59b74f3a8"
success_callback: ["langfuse"]
langfuse_public_key: os.environ/LANGFUSE_PUB_KEY_2 # Project 2
langfuse_secret: os.environ/LANGFUSE_SECRET_2 # Project 2
```
Now, when you [generate keys](./virtual_keys.md) for this team-id
```bash
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-d '{"team_id": "06ed1e01-3fa7-4b9e-95bc-f2e59b74f3a8"}'
```
All requests made with these keys will log data to their team-specific logging. -->
## [BETA] Team Logging via API
## [BETA] Team Logging
:::info
@ -57,7 +30,54 @@ All requests made with these keys will log data to their team-specific logging.
:::
### UI Usage
1. Create a Team with Logging Settings
Create a team called "AI Agents"
<Image
img={require('../../img/team_logging1.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
2. Create a Key for the Team
We will create a key for the team "AI Agents". The team logging settings will be used for all keys created for the team.
<Image
img={require('../../img/team_logging2.png')}
style={{width: '80%', display: 'block', margin: '2rem auto', border: '1px solid #E5E7EB'}}
/>
<br />
3. Make a test LLM API Request
Use the new key to make a test LLM API Request, we expect to see the logs on your logging provider configured in step 1.
<Image
img={require('../../img/team_logging3.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
4. Check Logs on your Logging Provider
Navigate to your configured logging provider and check if you received the logs from step 2.
<Image
img={require('../../img/team_logging4.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
### API Usage
### Set Callbacks Per Team
#### 1. Set callback for team
@ -189,6 +209,37 @@ curl -X GET 'http://localhost:4000/team/dbe2f686-a686-4896-864a-4c3924458709/cal
## Team Logging - `config.yaml`
Turn on/off logging and caching for a specific team id.
**Example:**
This config would send langfuse logs to 2 different langfuse projects, based on the team id
```yaml
litellm_settings:
default_team_settings:
- team_id: "dbe2f686-a686-4896-864a-4c3924458709"
success_callback: ["langfuse"]
langfuse_public_key: os.environ/LANGFUSE_PUB_KEY_1 # Project 1
langfuse_secret: os.environ/LANGFUSE_PRIVATE_KEY_1 # Project 1
- team_id: "06ed1e01-3fa7-4b9e-95bc-f2e59b74f3a8"
success_callback: ["langfuse"]
langfuse_public_key: os.environ/LANGFUSE_PUB_KEY_2 # Project 2
langfuse_secret: os.environ/LANGFUSE_SECRET_2 # Project 2
```
Now, when you [generate keys](./virtual_keys.md) for this team-id
```bash
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-d '{"team_id": "06ed1e01-3fa7-4b9e-95bc-f2e59b74f3a8"}'
```
All requests made with these keys will log data to their team-specific logging.
## [BETA] Key Based Logging
@ -201,11 +252,51 @@ Use the `/key/generate` or `/key/update` endpoints to add logging callbacks to a
:::
### How key based logging works:
**How key based logging works:**
- If **Key has no callbacks** configured, it will use the default callbacks specified in the config.yaml file
- If **Key has callbacks** configured, it will use the callbacks specified in the key
### UI Usage
1. Create a Key with Logging Settings
When creating a key, you can configure the specific logging settings for the key. These logging settings will be used for all requests made with this key.
<Image
img={require('../../img/key_logging.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
2. Make a test LLM API Request
Use the new key to make a test LLM API Request, we expect to see the logs on your logging provider configured in step 1.
<Image
img={require('../../img/key_logging2.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
3. Check Logs on your Logging Provider
Navigate to your configured logging provider and check if you received the logs from step 2.
<Image
img={require('../../img/key_logging_arize.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
### API Usage
<Tabs>
<TabItem label="Langfuse" value="langfuse">

View file

@ -501,6 +501,145 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
}'
```
## [BETA] Sync User Roles and Teams with IDP
Automatically sync user roles and team memberships from your Identity Provider (IDP) to LiteLLM's database. This ensures that user permissions and team memberships in LiteLLM stay in sync with your IDP.
**Note:** This is in beta and might change unexpectedly.
### Use Cases
- **Role Synchronization**: Automatically update user roles in LiteLLM when they change in your IDP
- **Team Membership Sync**: Keep team memberships in sync between your IDP and LiteLLM
- **Centralized Access Management**: Manage all user permissions through your IDP while maintaining LiteLLM functionality
### Setup
#### 1. Configure JWT Role Mapping
Map roles from your JWT token to LiteLLM user roles:
```yaml
general_settings:
enable_jwt_auth: True
litellm_jwtauth:
user_id_jwt_field: "sub"
team_ids_jwt_field: "groups"
roles_jwt_field: "roles"
user_id_upsert: true
sync_user_role_and_teams: true # 👈 Enable sync functionality
jwt_litellm_role_map: # 👈 Map JWT roles to LiteLLM roles
- jwt_role: "ADMIN"
litellm_role: "proxy_admin"
- jwt_role: "USER"
litellm_role: "internal_user"
- jwt_role: "VIEWER"
litellm_role: "internal_user"
```
#### 2. JWT Role Mapping Spec
- `jwt_role`: The role name as it appears in your JWT token. Supports wildcard patterns using `fnmatch` (e.g., `"ADMIN_*"` matches `"ADMIN_READ"`, `"ADMIN_WRITE"`, etc.)
- `litellm_role`: The corresponding LiteLLM user role
**Supported LiteLLM Roles:**
- `proxy_admin`: Full administrative access
- `internal_user`: Standard user access
- `internal_user_view_only`: Read-only access
#### 3. Example JWT Token
```json
{
"sub": "user-123",
"roles": ["ADMIN"],
"groups": ["team-alpha", "team-beta"],
"iat": 1234567890,
"exp": 1234567890
}
```
### How It Works
When a user makes a request with a JWT token:
1. **Role Sync**:
- LiteLLM checks if the user's role in the JWT matches their role in the database
- If different, the user's role is updated in LiteLLM's database
- Uses the `jwt_litellm_role_map` to convert JWT roles to LiteLLM roles
2. **Team Membership Sync**:
- Compares team memberships from the JWT token with the user's current teams in LiteLLM
- Adds the user to new teams found in the JWT
- Removes the user from teams not present in the JWT
3. **Database Updates**:
- Updates happen automatically during the authentication process
- No manual intervention required
### Configuration Options
```yaml
general_settings:
enable_jwt_auth: True
litellm_jwtauth:
# Required fields
user_id_jwt_field: "sub"
team_ids_jwt_field: "groups"
roles_jwt_field: "roles"
# Sync configuration
sync_user_role_and_teams: true
user_id_upsert: true
# Role mapping
jwt_litellm_role_map:
- jwt_role: "AI_ADMIN_*" # Wildcard pattern
litellm_role: "proxy_admin"
- jwt_role: "AI_USER"
litellm_role: "internal_user"
```
### Important Notes
- **Performance**: Sync operations happen during authentication, which may add slight latency
- **Database Access**: Requires database access for user and team updates
- **Team Creation**: Teams mentioned in JWT tokens must exist in LiteLLM before sync can assign users to them
- **Wildcard Support**: JWT role patterns support wildcard matching using `fnmatch`
### Testing the Sync Feature
1. **Create a test user with initial role**:
```bash
curl -X POST 'http://0.0.0.0:4000/user/new' \
-H 'Authorization: Bearer <PROXY_MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"user_id": "user-123",
"user_role": "internal_user"
}'
```
2. **Make a request with JWT containing different role**:
```bash
curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer <JWT_WITH_ADMIN_ROLE>' \
-d '{
"model": "claude-sonnet-4-20250514",
"messages": [{"role": "user", "content": "Hello"}]
}'
```
3. **Verify the role was updated**:
```bash
curl -X GET 'http://0.0.0.0:4000/user/info?user_id=user-123' \
-H 'Authorization: Bearer <PROXY_MASTER_KEY>'
```
## All JWT Params
[**See Code**](https://github.com/BerriAI/litellm/blob/b204f0c01c703317d812a1553363ab0cb989d5b6/litellm/proxy/_types.py#L95)

View file

@ -1,7 +1,7 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# 💰 Budgets, Rate Limits
# Budgets, Rate Limits
Requirements:

View file

@ -113,7 +113,7 @@ curl http://0.0.0.0:4000/rerank \
|-------------|--------------------|
| Cohere (v1 + v2 clients) | [Usage](#quick-start) |
| Together AI| [Usage](../docs/providers/togetherai) |
| Azure AI| [Usage](../docs/providers/azure_ai) |
| Azure AI| [Usage](../docs/providers/azure_ai#rerank-endpoint) |
| Jina AI| [Usage](../docs/providers/jina_ai) |
| AWS Bedrock| [Usage](../docs/providers/bedrock#rerank-api) |
| HuggingFace| [Usage](../docs/providers/huggingface_rerank) |

View file

@ -150,7 +150,7 @@ Use this to control what guardrails run per project. In this tutorial we only wa
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
-D '{
-d '{
"guardrails": ["aporia-pre-guard", "aporia-post-guard"]
}
}'

Binary file not shown.

After

Width:  |  Height:  |  Size: 136 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 298 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 333 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 626 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 467 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 128 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 79 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 248 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 320 KiB

View file

@ -1,5 +1,5 @@
---
title: "[PRE-RELEASE] v1.73.6-stable"
title: "v1.73.6-stable"
slug: "v1-73-6-stable"
date: 2025-06-28T10:00:00
authors:
@ -20,18 +20,27 @@ import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
:::warning
## Known Issues
The `non-root` docker image has a known issue around the UI not loading. If you use the `non-root` docker image we recommend waiting before upgrading to this version. We will post a patch fix for this.
:::
## Deploy this version
This release is not out yet. The pre-release will be live on Sunday and the stable release will be live on Wednesday.
<Tabs>
<TabItem value="docker" label="Docker">
``` showLineNumbers title="docker run litellm"
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
ghcr.io/berriai/litellm:v1.73.6-stable
```
</TabItem>
<TabItem value="pip" label="Pip">
``` showLineNumbers title="pip install litellm"
pip install litellm==1.73.6.post1
```
</TabItem>
</Tabs>
---

View file

@ -0,0 +1,354 @@
---
title: "[Pre-Release] v1.74.0"
slug: "v1-74-0-stable"
date: 2025-07-05T10:00:00
authors:
- name: Krrish Dholakia
title: CEO, LiteLLM
url: https://www.linkedin.com/in/krish-d/
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
- name: Ishaan Jaffer
title: CTO, LiteLLM
url: https://www.linkedin.com/in/reffajnaahsi/
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
hide_table_of_contents: false
---
import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
## Deploy this version
<Tabs>
<TabItem value="docker" label="Docker">
``` showLineNumbers title="docker run litellm"
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
ghcr.io/berriai/litellm:v1.74.0.rc
```
</TabItem>
<TabItem value="pip" label="Pip">
``` showLineNumbers title="pip install litellm"
pip install litellm==1.74.0.post1
```
</TabItem>
</Tabs>
---
## Key Highlights
### Azure Content Safety Guardrails
### MCP Gateway: Segregate MCP tools
MCP Server Segregation is now supported on LiteLLM. This means you can specify the `x-mcp-servers` header to specify which servers to list tools from. This is useful when you want to request tools from only a subset of configured servers — enabling curated toolsets and cleaner control.
#### Usage
<Tabs>
<TabItem value="openai" label="OpenAI API">
```bash title="cURL Example with Server Segregation" showLineNumbers
curl --location 'https://api.openai.com/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer $OPENAI_API_KEY" \
--data '{
"model": "gpt-4o",
"tools": [
{
"type": "mcp",
"server_label": "litellm",
"server_url": "<your-litellm-proxy-base-url>/mcp",
"require_approval": "never",
"headers": {
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
"x-mcp-servers": "Zapier_Gmail"
}
}
],
"input": "Run available tools",
"tool_choice": "required"
}'
```
In this example, the request will only have access to tools from the "Zapier_Gmail" MCP server.
</TabItem>
<TabItem value="litellm" label="LiteLLM Proxy">
```bash title="cURL Example with Server Segregation" showLineNumbers
curl --location '<your-litellm-proxy-base-url>/v1/responses' \
--header 'Content-Type: application/json' \
--header "Authorization: Bearer $LITELLM_API_KEY" \
--data '{
"model": "gpt-4o",
"tools": [
{
"type": "mcp",
"server_label": "litellm",
"server_url": "<your-litellm-proxy-base-url>/mcp",
"require_approval": "never",
"headers": {
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
"x-mcp-servers": "Zapier_Gmail,Server2"
}
}
],
"input": "Run available tools",
"tool_choice": "required"
}'
```
This configuration restricts the request to only use tools from the specified MCP servers.
</TabItem>
<TabItem value="cursor" label="Cursor IDE">
```json title="Cursor MCP Configuration with Server Segregation" showLineNumbers
{
"mcpServers": {
"LiteLLM": {
"url": "<your-litellm-proxy-base-url>/mcp",
"headers": {
"x-litellm-api-key": "Bearer $LITELLM_API_KEY",
"x-mcp-servers": "Zapier_Gmail,Server2"
}
}
}
}
```
This configuration in Cursor IDE settings will limit tool access to only the specified MCP server.
</TabItem>
</Tabs>
### Team / Key Based Logging on UI
<Image
img={require('../../img/release_notes/team_key_logging.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br />
This release brings support for Proxy Admins to configure Team/Key Based Logging Settings on the UI. This allows routing LLM request/response logs to different Langfuse/Arize projects based on the team or key.
For developers using LiteLLM, their logs are automatically routed to their specific Arize/Langfuse projects. On this release, we support the following integrations for key/team based logging:
- `langfuse`
- `arize`
- `langsmith`
### Python SDK: 2.3 Second Faster Import Times
This release brings significant performance improvements to the Python SDK with 2.3 seconds faster import times. We've refactored the initialization process to reduce startup overhead, making LiteLLM more efficient for applications that need quick initialization. This is a major improvement for applications that need to initialize LiteLLM quickly.
---
## New Models / Updated Models
#### Pricing / Context Window Updates
| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Type |
| ----------- | -------------------------------------- | -------------- | ------------------- | -------------------- | ---- |
| Watsonx | `watsonx/mistralai/mistral-large` | 131k | $3.00 | $10.00 | New |
| Azure AI | `azure_ai/cohere-rerank-v3.5` | 4k | $2.00/1k queries | - | New (Rerank) |
#### Features
- **[🆕 GitHub Copilot](../../docs/providers/github_copilot)** - Use GitHub Copilot API with LiteLLM - [PR](https://github.com/BerriAI/litellm/pull/12325), [Get Started](../../docs/providers/github_copilot)
- **[🆕 VertexAI DeepSeek](../../docs/providers/vertex)** - Add support for VertexAI DeepSeek models - [PR](https://github.com/BerriAI/litellm/pull/12312), [Get Started](../../docs/providers/vertex_partner#vertexai-deepseek)
- **[Azure AI](../../docs/providers/azure_ai)**
- Add azure_ai cohere rerank v3.5 - [PR](https://github.com/BerriAI/litellm/pull/12283), [Get Started](../../docs/providers/azure_ai#rerank-endpoint)
- **[Vertex AI](../../docs/providers/vertex)**
- Add size parameter support for image generation - [PR](https://github.com/BerriAI/litellm/pull/12292), [Get Started](../../docs/providers/vertex_image)
- **[Custom LLM](../../docs/providers/custom_llm_server)**
- Pass through extra_ properties on "custom" llm provider - [PR](https://github.com/BerriAI/litellm/pull/12185)
#### Bugs
- **[Mistral](../../docs/providers/mistral)**
- Fix transform_response handling for empty string content - [PR](https://github.com/BerriAI/litellm/pull/12202)
- Turn Mistral to use llm_http_handler - [PR](https://github.com/BerriAI/litellm/pull/12245)
- **[Gemini](../../docs/providers/gemini)**
- Fix tool call sequence - [PR](https://github.com/BerriAI/litellm/pull/11999)
- Fix custom api_base path preservation - [PR](https://github.com/BerriAI/litellm/pull/12215)
- **[Anthropic](../../docs/providers/anthropic)**
- Fix user_id validation logic - [PR](https://github.com/BerriAI/litellm/pull/11432)
- **[Bedrock](../../docs/providers/bedrock)**
- Support optional args for bedrock - [PR](https://github.com/BerriAI/litellm/pull/12287)
- **[Ollama](../../docs/providers/ollama)**
- Fix default parameters for ollama-chat - [PR](https://github.com/BerriAI/litellm/pull/12201)
- **[VLLM](../../docs/providers/vllm)**
- Add 'audio_url' message type support - [PR](https://github.com/BerriAI/litellm/pull/12270)
---
## LLM API Endpoints
#### Features
- **[/batches](../../docs/batches)**
- Support batch retrieve with target model Query Param - [PR](https://github.com/BerriAI/litellm/pull/12228)
- Anthropic completion bridge improvements - [PR](https://github.com/BerriAI/litellm/pull/12228)
- **[/responses](../../docs/response_api)**
- Azure responses api bridge improvements - [PR](https://github.com/BerriAI/litellm/pull/12224)
- Fix responses api error handling - [PR](https://github.com/BerriAI/litellm/pull/12225)
- **[/mcp (MCP Gateway)](../../docs/mcp)**
- Add MCP url masking on frontend - [PR](https://github.com/BerriAI/litellm/pull/12247)
- Add MCP servers header to scope - [PR](https://github.com/BerriAI/litellm/pull/12266)
- Litellm mcp tool prefix - [PR](https://github.com/BerriAI/litellm/pull/12289)
- Segregate MCP tools on connections using headers - [PR](https://github.com/BerriAI/litellm/pull/12296)
- Added changes to mcp url wrapping - [PR](https://github.com/BerriAI/litellm/pull/12207)
#### Bugs
- **[/v1/messages](../../docs/anthropic_unified)**
- Remove hardcoded model name on streaming - [PR](https://github.com/BerriAI/litellm/pull/12131)
- Support lowest latency routing - [PR](https://github.com/BerriAI/litellm/pull/12180)
- Non-anthropic models token usage returned - [PR](https://github.com/BerriAI/litellm/pull/12184)
- **[/chat/completions](../../docs/providers/anthropic_unified)**
- Support Cursor IDE tool_choice format `{"type": "auto"}` - [PR](https://github.com/BerriAI/litellm/pull/12168)
- **[/generateContent](../../docs/generate_content)**
- Allow passing litellm_params - [PR](https://github.com/BerriAI/litellm/pull/12177)
- Only pass supported params when using OpenAI models - [PR](https://github.com/BerriAI/litellm/pull/12297)
- Fix using gemini-cli with Vertex Anthropic Models - [PR](https://github.com/BerriAI/litellm/pull/12246)
- **Streaming**
- Fix Error code: 307 for LlamaAPI Streaming Chat - [PR](https://github.com/BerriAI/litellm/pull/11946)
- Store finish reason even if is_finished - [PR](https://github.com/BerriAI/litellm/pull/12250)
---
## Spend Tracking / Budget Improvements
#### Bugs
- Fix allow strings in calculate cost - [PR](https://github.com/BerriAI/litellm/pull/12200)
- VertexAI Anthropic streaming cost tracking with prompt caching fixes - [PR](https://github.com/BerriAI/litellm/pull/12188)
---
## Management Endpoints / UI
#### Bugs
- **Team Management**
- Prevent team model reset on model add - [PR](https://github.com/BerriAI/litellm/pull/12144)
- Return team-only models on /v2/model/info - [PR](https://github.com/BerriAI/litellm/pull/12144)
- Render team member budget correctly - [PR](https://github.com/BerriAI/litellm/pull/12144)
- **UI Rendering**
- Fix rendering ui on non-root images - [PR](https://github.com/BerriAI/litellm/pull/12226)
- Correctly display 'Internal Viewer' user role - [PR](https://github.com/BerriAI/litellm/pull/12284)
- **Configuration**
- Handle empty config.yaml - [PR](https://github.com/BerriAI/litellm/pull/12189)
- Fix gemini /models - replace models/ as expected - [PR](https://github.com/BerriAI/litellm/pull/12189)
#### Features
- **Team Management**
- Allow adding team specific logging callbacks - [PR](https://github.com/BerriAI/litellm/pull/12261)
- Add Arize Team Based Logging - [PR](https://github.com/BerriAI/litellm/pull/12264)
- Allow Viewing/Editing Team Based Callbacks - [PR](https://github.com/BerriAI/litellm/pull/12265)
- **UI Improvements**
- Comma separated spend and budget display - [PR](https://github.com/BerriAI/litellm/pull/12317)
- Add logos to callback list - [PR](https://github.com/BerriAI/litellm/pull/12244)
- **CLI**
- Add litellm-proxy cli login for starting to use litellm proxy - [PR](https://github.com/BerriAI/litellm/pull/12216)
- **Email Templates**
- Customizable Email template - Subject and Signature - [PR](https://github.com/BerriAI/litellm/pull/12218)
---
## Logging / Guardrail Integrations
#### Features
- **[Azure Content Safety](../../docs/guardrails/azure_content_safety)**
- Add Azure Content Safety Guardrails to LiteLLM proxy - [PR](https://github.com/BerriAI/litellm/pull/12268)
- Add azure content safety guardrails to the UI - [PR](https://github.com/BerriAI/litellm/pull/12309)
- **[DeepEval](../../docs/observability/deepeval_integration)**
- Fix DeepEval logging format for failure events - [PR](https://github.com/BerriAI/litellm/pull/12303)
- **[Arize](../../docs/proxy/logging#arize)**
- Add Arize Team Based Logging - [PR](https://github.com/BerriAI/litellm/pull/12264)
- **[Langfuse](../../docs/proxy/logging#langfuse)**
- Langfuse prompt_version support - [PR](https://github.com/BerriAI/litellm/pull/12301)
- **[Sentry Integration](../../docs/observability/sentry)**
- Add sentry scrubbing - [PR](https://github.com/BerriAI/litellm/pull/12210)
- **[AWS SQS Logging](../../docs/proxy/logging#aws-sqs)**
- New AWS SQS Logging Integration - [PR](https://github.com/BerriAI/litellm/pull/12176)
- **[S3 Logger](../../docs/proxy/logging#s3-buckets)**
- Add failure logging support - [PR](https://github.com/BerriAI/litellm/pull/12299)
- **[Prometheus Metrics](../../docs/proxy/prometheus)**
- Add better error validation for prometheus metrics and labels - [PR](https://github.com/BerriAI/litellm/pull/12182)
#### Bugs
- **Security**
- Ensure only LLM API route fails get logged on Langfuse - [PR](https://github.com/BerriAI/litellm/pull/12308)
- **OpenMeter**
- Integration error handling fix - [PR](https://github.com/BerriAI/litellm/pull/12147)
- **Message Redaction**
- Ensure message redaction works for responses API logging - [PR](https://github.com/BerriAI/litellm/pull/12291)
- **Bedrock Guardrails**
- Fix bedrock guardrails post_call for streaming responses - [PR](https://github.com/BerriAI/litellm/pull/12252)
---
## Performance / Loadbalancing / Reliability improvements
#### Features
- **Python SDK**
- 2 second faster import times - [PR](https://github.com/BerriAI/litellm/pull/12135)
- Reduce python sdk import time by .3s - [PR](https://github.com/BerriAI/litellm/pull/12140)
- **Error Handling**
- Add error handling for MCP tools not found or invalid server - [PR](https://github.com/BerriAI/litellm/pull/12223)
- **SSL/TLS**
- Fix SSL certificate error - [PR](https://github.com/BerriAI/litellm/pull/12327)
- Fix custom ca bundle support in aiohttp transport - [PR](https://github.com/BerriAI/litellm/pull/12281)
---
## General Proxy Improvements
- **Startup**
- Add new banner on startup - [PR](https://github.com/BerriAI/litellm/pull/12328)
- **Dependencies**
- Update pydantic version - [PR](https://github.com/BerriAI/litellm/pull/12213)
---
## New Contributors
* @wildcard made their first contribution in https://github.com/BerriAI/litellm/pull/12157
* @colesmcintosh made their first contribution in https://github.com/BerriAI/litellm/pull/12168
* @seyeong-han made their first contribution in https://github.com/BerriAI/litellm/pull/11946
* @dinggh made their first contribution in https://github.com/BerriAI/litellm/pull/12162
* @raz-alon made their first contribution in https://github.com/BerriAI/litellm/pull/11432
* @tofarr made their first contribution in https://github.com/BerriAI/litellm/pull/12200
* @szafranek made their first contribution in https://github.com/BerriAI/litellm/pull/12179
* @SamBoyd made their first contribution in https://github.com/BerriAI/litellm/pull/12147
* @lizzij made their first contribution in https://github.com/BerriAI/litellm/pull/12219
* @cipri-tom made their first contribution in https://github.com/BerriAI/litellm/pull/12201
* @zsimjee made their first contribution in https://github.com/BerriAI/litellm/pull/12185
* @jroberts2600 made their first contribution in https://github.com/BerriAI/litellm/pull/12175
* @njbrake made their first contribution in https://github.com/BerriAI/litellm/pull/12202
* @NANDINI-star made their first contribution in https://github.com/BerriAI/litellm/pull/12244
* @utsumi-fj made their first contribution in https://github.com/BerriAI/litellm/pull/12230
* @dcieslak19973 made their first contribution in https://github.com/BerriAI/litellm/pull/12283
* @hanouticelina made their first contribution in https://github.com/BerriAI/litellm/pull/12286
* @lowjiansheng made their first contribution in https://github.com/BerriAI/litellm/pull/11999
* @JoostvDoorn made their first contribution in https://github.com/BerriAI/litellm/pull/12281
* @takashiishida made their first contribution in https://github.com/BerriAI/litellm/pull/12239
## **[Git Diff](https://github.com/BerriAI/litellm/compare/v1.73.6-stable...v1.74.0-stable)**

View file

@ -142,6 +142,7 @@ const sidebars = {
"proxy/token_auth",
"proxy/service_accounts",
"proxy/access_control",
"proxy/cli_sso",
"proxy/custom_auth",
"proxy/ip_address",
"proxy/email",
@ -255,6 +256,7 @@ const sidebars = {
"embedding/supported_embedding",
"anthropic_unified",
"mcp",
"generateContent",
{
type: "category",
label: "/images",
@ -354,12 +356,12 @@ const sidebars = {
]
},
"providers/azure_ai",
"providers/aiml",
{
type: "category",
label: "Vertex AI",
items: [
"providers/vertex",
"providers/vertex_partner",
"providers/vertex_image",
]
},
@ -414,7 +416,6 @@ const sidebars = {
"providers/galadriel",
"providers/topaz",
"providers/groq",
"providers/github",
"providers/deepseek",
"providers/elevenlabs",
"providers/fireworks_ai",
@ -423,8 +424,11 @@ const sidebars = {
"providers/llamafile",
"providers/infinity",
"providers/xinference",
"providers/aiml",
"providers/cloudflare_workers",
"providers/deepinfra",
"providers/github",
"providers/github_copilot",
"providers/ai21",
"providers/nlp_cloud",
"providers/replicate",

Binary file not shown.

Binary file not shown.

View file

@ -29,6 +29,10 @@ from litellm.types.integrations.slack_alerting import LITELLM_LOGO_URL
class BaseEmailLogger(CustomLogger):
DEFAULT_LITELLM_EMAIL = "notifications@alerts.litellm.ai"
DEFAULT_SUPPORT_EMAIL = "support@berri.ai"
DEFAULT_SUBJECT_TEMPLATES = {
EmailEvent.new_user_invitation: "LiteLLM: {event_message}",
EmailEvent.virtual_key_created: "LiteLLM: {event_message}",
}
async def send_user_invitation_email(self, event: WebhookEvent):
"""
@ -38,8 +42,8 @@ class BaseEmailLogger(CustomLogger):
email_event=EmailEvent.new_user_invitation,
user_id=event.user_id,
user_email=getattr(event, "user_email", None),
event_message=event.event_message,
)
# Implement invitation email logic using email_params
verbose_proxy_logger.debug(
f"send_user_invitation_email_event: {json.dumps(event, indent=4, default=str)}"
@ -50,13 +54,13 @@ class BaseEmailLogger(CustomLogger):
recipient_email=email_params.recipient_email,
base_url=email_params.base_url,
email_support_contact=email_params.support_contact,
email_footer=EMAIL_FOOTER,
email_footer=email_params.signature,
)
await self.send_email(
from_email=self.DEFAULT_LITELLM_EMAIL,
to_email=[email_params.recipient_email],
subject=f"LiteLLM: {event.event_message}",
subject=email_params.subject,
html_body=email_html_content,
)
@ -68,11 +72,11 @@ class BaseEmailLogger(CustomLogger):
"""
Send email to user after creating key for the user
"""
email_params = await self._get_email_params(
user_id=send_key_created_email_event.user_id,
user_email=send_key_created_email_event.user_email,
email_event=EmailEvent.virtual_key_created,
event_message=send_key_created_email_event.event_message,
)
verbose_proxy_logger.debug(
@ -86,13 +90,13 @@ class BaseEmailLogger(CustomLogger):
key_token=send_key_created_email_event.virtual_key,
base_url=email_params.base_url,
email_support_contact=email_params.support_contact,
email_footer=EMAIL_FOOTER,
email_footer=email_params.signature,
)
await self.send_email(
from_email=self.DEFAULT_LITELLM_EMAIL,
to_email=[email_params.recipient_email],
subject=f"LiteLLM: {send_key_created_email_event.event_message}",
subject=email_params.subject,
html_body=email_html_content,
)
pass
@ -102,16 +106,63 @@ class BaseEmailLogger(CustomLogger):
email_event: EmailEvent,
user_id: Optional[str] = None,
user_email: Optional[str] = None,
event_message: Optional[str] = None,
) -> EmailParams:
"""
Get common email parameters used across different email sending methods
Args:
email_event: Type of email event
user_id: Optional user ID to look up email
user_email: Optional direct email address
event_message: Optional message to include in email subject
Returns:
EmailParams object containing logo_url, support_contact, base_url, and recipient_email
EmailParams object containing logo_url, support_contact, base_url, recipient_email, subject, and signature
"""
logo_url = os.getenv("EMAIL_LOGO_URL", None) or LITELLM_LOGO_URL
support_contact = os.getenv("EMAIL_SUPPORT_CONTACT", self.DEFAULT_SUPPORT_EMAIL)
base_url = os.getenv("PROXY_BASE_URL", "http://0.0.0.0:4000")
# Get email parameters with premium check for custom values
custom_logo = os.getenv("EMAIL_LOGO_URL", None)
custom_support = os.getenv("EMAIL_SUPPORT_CONTACT", None)
custom_signature = os.getenv("EMAIL_SIGNATURE", None)
custom_subject_invitation = os.getenv("EMAIL_SUBJECT_INVITATION", None)
custom_subject_key_created = os.getenv("EMAIL_SUBJECT_KEY_CREATED", None)
# Track which custom values were not applied
unused_custom_fields = []
# Function to safely get custom value or default
def get_custom_or_default(custom_value: Optional[str], default_value: str, field_name: str) -> str:
if custom_value is not None: # Only check premium if trying to use custom value
from litellm.proxy.proxy_server import premium_user
if premium_user is not True:
unused_custom_fields.append(field_name)
return default_value
return custom_value
return default_value
# Get parameters, falling back to defaults if custom values aren't allowed
logo_url = get_custom_or_default(custom_logo, LITELLM_LOGO_URL, "logo URL")
support_contact = get_custom_or_default(custom_support, self.DEFAULT_SUPPORT_EMAIL, "support contact")
base_url = os.getenv("PROXY_BASE_URL", "http://0.0.0.0:4000") # Not a premium feature
signature = get_custom_or_default(custom_signature, EMAIL_FOOTER, "email signature")
# Get custom subject template based on email event type
if email_event == EmailEvent.new_user_invitation:
subject_template = get_custom_or_default(
custom_subject_invitation,
self.DEFAULT_SUBJECT_TEMPLATES[EmailEvent.new_user_invitation],
"invitation subject template"
)
elif email_event == EmailEvent.virtual_key_created:
subject_template = get_custom_or_default(
custom_subject_key_created,
self.DEFAULT_SUBJECT_TEMPLATES[EmailEvent.virtual_key_created],
"key created subject template"
)
else:
subject_template = "LiteLLM: {event_message}"
subject = subject_template.format(event_message=event_message) if event_message else "LiteLLM Notification"
recipient_email: Optional[
str
@ -127,11 +178,25 @@ class BaseEmailLogger(CustomLogger):
user_id=user_id, base_url=base_url
)
# If any custom fields were not applied, log a warning
if unused_custom_fields:
fields_str = ", ".join(unused_custom_fields)
warning_msg = (
f"Email sent with default values instead of custom values for: {fields_str}. "
"This is an Enterprise feature. To use custom email fields, please upgrade to LiteLLM Enterprise. "
"Schedule a meeting here: https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat"
)
verbose_proxy_logger.warning(
f"{warning_msg}"
)
return EmailParams(
logo_url=logo_url,
support_contact=support_contact,
base_url=base_url,
recipient_email=recipient_email,
subject=subject,
signature=signature,
)
def _format_key_budget(self, max_budget: Optional[float]) -> str:

View file

@ -5,19 +5,19 @@ from pydantic import BaseModel, Field
from litellm.proxy._types import WebhookEvent
class EmailParams(BaseModel):
logo_url: str
support_contact: str
base_url: str
recipient_email: str
subject: str
signature: str
class SendKeyCreatedEmailEvent(WebhookEvent):
virtual_key: str
"""
The virtual key that was created
this will be sk-123xxx, since we will be emailing this to the user to start using the key
"""
@ -26,35 +26,25 @@ class EmailEvent(str, enum.Enum):
virtual_key_created = "Virtual Key Created"
new_user_invitation = "New User Invitation"
class EmailEventSettings(BaseModel):
event: EmailEvent
enabled: bool
class EmailEventSettingsUpdateRequest(BaseModel):
settings: List[EmailEventSettings]
class EmailEventSettingsResponse(BaseModel):
settings: List[EmailEventSettings]
class DefaultEmailSettings(BaseModel):
"""Default settings for email events"""
settings: Dict[EmailEvent, bool] = Field(
default_factory=lambda: {
EmailEvent.virtual_key_created: False, # Off by default
EmailEvent.new_user_invitation: True, # On by default
}
)
def to_dict(self) -> Dict[str, bool]:
"""Convert to dictionary with string keys for storage"""
return {event.value: enabled for event, enabled in self.settings.items()}
@classmethod
def get_defaults(cls) -> Dict[str, bool]:
"""Get the default settings as a dictionary with string keys"""
return cls().to_dict()
return cls().to_dict()

View file

@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm-enterprise"
version = "0.1.10"
version = "0.1.11"
description = "Package for LiteLLM Enterprise features"
authors = ["BerriAI"]
readme = "README.md"
@ -22,7 +22,7 @@ requires = ["poetry-core"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "0.1.10"
version = "0.1.11"
version_files = [
"pyproject.toml:version",
"../requirements.txt:litellm-enterprise==",

View file

@ -2,7 +2,7 @@
import warnings
warnings.filterwarnings("ignore", message=".*conflict with protected namespace.*")
### INIT VARIABLES ############
### INIT VARIABLES ################
import threading
import os
from typing import Callable, List, Optional, Dict, Union, Any, Literal, get_args
@ -118,6 +118,7 @@ _custom_logger_compatible_callbacks_literal = Literal[
"smtp_email",
"deepeval",
"s3_v2",
"aws_sqs",
]
logged_real_time_event_types: Optional[Union[List[str], Literal["*"]]] = None
_known_custom_logger_compatible_callbacks: List = list(
@ -214,6 +215,7 @@ use_litellm_proxy: bool = (
)
use_client: bool = False
ssl_verify: Union[str, bool] = True
ssl_security_level: Optional[str] = None
ssl_certificate: Optional[str] = None
disable_streaming_logging: bool = False
disable_token_counter: bool = False
@ -292,6 +294,7 @@ model_cost_map_url: str = (
suppress_debug_info = False
dynamodb_table_name: Optional[str] = None
s3_callback_params: Optional[Dict] = None
aws_sqs_callback_params: Optional[Dict] = None
generic_logger_headers: Optional[Dict] = None
default_key_generate_params: Optional[Dict] = None
upperbound_key_generate_params: Optional[LiteLLM_UpperboundKeyGenerateParams] = None
@ -1050,7 +1053,7 @@ from .llms.groq.chat.transformation import GroqChatConfig
from .llms.voyage.embedding.transformation import VoyageEmbeddingConfig
from .llms.infinity.embedding.transformation import InfinityEmbeddingConfig
from .llms.azure_ai.chat.transformation import AzureAIStudioConfig
from .llms.mistral.mistral_chat_transformation import MistralConfig
from .llms.mistral.chat.transformation import MistralConfig
from .llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from .llms.azure.responses.transformation import AzureOpenAIResponsesAPIConfig
from .llms.openai.chat.o_series_transformation import (
@ -1122,6 +1125,7 @@ from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config
from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig
from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig
from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig
from .llms.github_copilot.chat.transformation import GithubCopilotConfig
from .llms.nebius.chat.transformation import NebiusConfig
from .main import * # type: ignore
from .integrations import *

View file

@ -75,9 +75,7 @@ class ResponsesToCompletionBridgeHandler:
custom_llm_provider=custom_llm_provider,
)
def completion(
self, *args, **kwargs
) -> Union[
def completion(self, *args, **kwargs) -> Union[
Coroutine[Any, Any, Union["ModelResponse", "CustomStreamWrapper"]],
"ModelResponse",
"CustomStreamWrapper",
@ -106,6 +104,7 @@ class ResponsesToCompletionBridgeHandler:
litellm_params=litellm_params,
headers=headers,
litellm_logging_obj=logging_obj,
client=kwargs.get("client"),
)
result = responses(

View file

@ -121,6 +121,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
litellm_params: dict,
headers: dict,
litellm_logging_obj: "LiteLLMLoggingObj",
client: Optional[Any] = None,
) -> dict:
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
@ -186,6 +187,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
"input": input_items,
"litellm_logging_obj": litellm_logging_obj,
**litellm_params,
"client": client,
}
verbose_logger.debug(

View file

@ -8,6 +8,12 @@ DEFAULT_S3_FLUSH_INTERVAL_SECONDS = int(
os.getenv("DEFAULT_S3_FLUSH_INTERVAL_SECONDS", 10)
)
DEFAULT_S3_BATCH_SIZE = int(os.getenv("DEFAULT_S3_BATCH_SIZE", 512))
DEFAULT_SQS_FLUSH_INTERVAL_SECONDS = int(
os.getenv("DEFAULT_SQS_FLUSH_INTERVAL_SECONDS", 10)
)
DEFAULT_SQS_BATCH_SIZE = int(os.getenv("DEFAULT_SQS_BATCH_SIZE", 512))
SQS_SEND_MESSAGE_ACTION = "SendMessage"
SQS_API_VERSION = "2012-11-05"
DEFAULT_MAX_RETRIES = int(os.getenv("DEFAULT_MAX_RETRIES", 2))
DEFAULT_MAX_RECURSE_DEPTH = int(os.getenv("DEFAULT_MAX_RECURSE_DEPTH", 100))
DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER = int(
@ -256,6 +262,7 @@ LITELLM_CHAT_PROVIDERS = [
"lm_studio",
"galadriel",
"gradient_ai",
"github_copilot", # GitHub Copilot Chat API
"novita",
"meta_llama",
"featherless_ai",
@ -392,7 +399,6 @@ openai_compatible_endpoints: List = [
openai_compatible_providers: List = [
"anyscale",
"mistral",
"groq",
"nvidia_nim",
"cerebras",
@ -417,6 +423,7 @@ openai_compatible_providers: List = [
"llamafile",
"lm_studio",
"galadriel",
"github_copilot", # GitHub Copilot Chat API
"novita",
"meta_llama",
"featherless_ai",
@ -713,6 +720,8 @@ MAXIMUM_TRACEBACK_LINES_TO_LOG = int(os.getenv("MAXIMUM_TRACEBACK_LINES_TO_LOG",
# Headers to control callbacks
X_LITELLM_DISABLE_CALLBACKS = "x-litellm-disable-callbacks"
LITELLM_METADATA_FIELD = "litellm_metadata"
OLD_LITELLM_METADATA_FIELD = "metadata"
########################### LiteLLM Proxy Specific Constants ###########################
########################################################################################
@ -751,6 +760,10 @@ HEALTH_CHECK_TIMEOUT_SECONDS = int(
UI_SESSION_TOKEN_TEAM_ID = "litellm-dashboard"
LITELLM_PROXY_ADMIN_NAME = "default_user_id"
########################### CLI SSO AUTHENTICATION CONSTANTS ###########################
LITELLM_CLI_SOURCE_IDENTIFIER = "litellm-cli"
LITELLM_CLI_SESSION_TOKEN_PREFIX = "litellm-session-token"
########################### DB CRON JOB NAMES ###########################
DB_SPEND_UPDATE_JOB_NAME = "db_spend_update_job"
PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME = "prometheus_emit_budget_metrics"
@ -792,3 +805,29 @@ SPECIAL_LITELLM_AUTH_TOKEN = ["ui-token"]
DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL = int(
os.getenv("DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL", 60)
)
# Sentry Scrubbing Configuration
SENTRY_DENYLIST = [
# API Keys and Tokens
"api_key", "token", "key", "secret", "password", "auth", "credential",
"OPENAI_API_KEY", "ANTHROPIC_API_KEY", "AZURE_API_KEY", "COHERE_API_KEY",
"REPLICATE_API_KEY", "HUGGINGFACE_API_KEY", "TOGETHERAI_API_KEY",
"CLOUDFLARE_API_KEY", "BASETEN_KEY", "OPENROUTER_KEY", "DATAROBOT_API_TOKEN",
"FIREWORKS_API_KEY", "FIREWORKS_AI_API_KEY", "FIREWORKSAI_API_KEY",
# Database and Connection Strings
"database_url", "redis_url", "connection_string",
# Authentication and Security
"master_key", "LITELLM_MASTER_KEY", "auth_token", "jwt_token", "private_key",
"SLACK_WEBHOOK_URL", "webhook_url", "LANGFUSE_SECRET_KEY",
# Email Configuration
"SMTP_PASSWORD", "SMTP_USERNAME", "email_password",
# Cloud Provider Credentials
"aws_access_key", "aws_secret_key", "gcp_credentials",
"azure_credentials", "HCP_VAULT_TOKEN", "CIRCLE_OIDC_TOKEN",
# Proxy and Environment Settings
"proxy_url", "proxy_key", "environment_variables"
]
SENTRY_PII_DENYLIST = [
"user_id", "email", "phone", "address", "ip_address",
"SMTP_SENDER_EMAIL", "TEST_EMAIL_ADDRESS"
]

View file

@ -4,6 +4,7 @@ LiteLLM Proxy uses this MCP Client to connnect to other MCP servers.
import base64
from datetime import timedelta
from typing import List, Optional
import asyncio
from mcp import ClientSession
from mcp.client.sse import sse_client
@ -46,6 +47,7 @@ class MCPClient:
self._transport_ctx = None
self._transport = None
self._session_ctx = None
self._task: Optional[asyncio.Task] = None
# handle the basic auth value if provided
if auth_value:
@ -56,8 +58,12 @@ class MCPClient:
Enable async context manager support.
Initializes the transport and session.
"""
await self.connect()
return self
try:
await self.connect()
return self
except Exception:
await self.disconnect()
raise
async def connect(self):
"""Initialize the transport and session."""
@ -66,47 +72,63 @@ class MCPClient:
headers = self._get_auth_headers()
if self.transport_type == MCPTransport.sse:
self._transport_ctx = sse_client(
url=self.server_url,
timeout=self.timeout,
headers=headers,
)
self._transport = await self._transport_ctx.__aenter__()
self._session_ctx = ClientSession(self._transport[0], self._transport[1])
self._session = await self._session_ctx.__aenter__()
await self._session.initialize()
else:
self._transport_ctx = streamablehttp_client(
url=self.server_url,
timeout=timedelta(seconds=self.timeout),
headers=headers,
)
self._transport = await self._transport_ctx.__aenter__()
self._session_ctx = ClientSession(self._transport[0], self._transport[1])
self._session = await self._session_ctx.__aenter__()
await self._session.initialize()
try:
if self.transport_type == MCPTransport.sse:
self._transport_ctx = sse_client(
url=self.server_url,
timeout=self.timeout,
headers=headers,
)
self._transport = await self._transport_ctx.__aenter__()
self._session_ctx = ClientSession(self._transport[0], self._transport[1])
self._session = await self._session_ctx.__aenter__()
await self._session.initialize()
else:
self._transport_ctx = streamablehttp_client(
url=self.server_url,
timeout=timedelta(seconds=self.timeout),
headers=headers,
)
self._transport = await self._transport_ctx.__aenter__()
self._session_ctx = ClientSession(self._transport[0], self._transport[1])
self._session = await self._session_ctx.__aenter__()
await self._session.initialize()
except Exception:
await self.disconnect()
raise
async def __aexit__(self, exc_type, exc_val, exc_tb):
"""Cleanup when exiting context manager."""
if self._session:
await self._session_ctx.__aexit__(exc_type, exc_val, exc_tb) # type: ignore
if self._transport_ctx:
await self._transport_ctx.__aexit__(exc_type, exc_val, exc_tb)
await self.disconnect()
async def disconnect(self):
"""Clean up session and connections."""
if self._task and not self._task.done():
self._task.cancel()
try:
await self._task
except asyncio.CancelledError:
pass
if self._session:
try:
# Ensure session is properly closed
await self._session.close() # type: ignore
await self._session_ctx.__aexit__(None, None, None) # type: ignore
except Exception:
pass
self._session = None
self._session_ctx = None
if self._transport_ctx:
try:
await self._transport_ctx.__aexit__(None, None, None)
except Exception:
pass
self._transport_ctx = None
self._transport = None
if self._context:
try:
await self._context.__aexit__(None, None, None) # type: ignore
await self._context.__aexit__(None, None, None) # type: ignore
except Exception:
pass
self._context = None
@ -140,8 +162,15 @@ class MCPClient:
if self._session is None:
raise ValueError("Session is not initialized")
result = await self._session.list_tools()
return result.tools
try:
result = await self._session.list_tools()
return result.tools
except asyncio.CancelledError:
await self.disconnect()
raise
except Exception:
await self.disconnect()
raise
async def call_tool(
self, call_tool_request_params: MCPCallToolRequestParams
@ -155,10 +184,17 @@ class MCPClient:
if self._session is None:
raise ValueError("Session is not initialized")
tool_result = await self._session.call_tool(
name=call_tool_request_params.name,
arguments=call_tool_request_params.arguments,
)
return tool_result
try:
tool_result = await self._session.call_tool(
name=call_tool_request_params.name,
arguments=call_tool_request_params.arguments,
)
return tool_result
except asyncio.CancelledError:
await self.disconnect()
raise
except Exception:
await self.disconnect()
raise

View file

@ -1,6 +1,7 @@
from typing import Any, AsyncIterator, Coroutine, Dict, List, Optional, Union, cast
import litellm
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import ModelResponse
from .transformation import GoogleGenAIAdapter
@ -18,6 +19,7 @@ class GenerateContentToCompletionHandler:
contents: Union[List[Dict[str, Any]], Dict[str, Any]],
config: Optional[Dict[str, Any]] = None,
stream: bool = False,
litellm_params: Optional[GenericLiteLLMParams] = None,
extra_kwargs: Optional[Dict[str, Any]] = None,
) -> Dict[str, Any]:
"""Prepare kwargs for litellm.completion/acompletion"""
@ -27,6 +29,7 @@ class GenerateContentToCompletionHandler:
model=model,
contents=contents,
config=config,
litellm_params=litellm_params,
**(extra_kwargs or {})
)
@ -41,6 +44,7 @@ class GenerateContentToCompletionHandler:
async def async_generate_content_handler(
model: str,
contents: Union[List[Dict[str, Any]], Dict[str, Any]],
litellm_params: GenericLiteLLMParams,
config: Optional[Dict[str, Any]] = None,
stream: bool = False,
**kwargs,
@ -52,6 +56,7 @@ class GenerateContentToCompletionHandler:
contents=contents,
config=config,
stream=stream,
litellm_params=litellm_params,
extra_kwargs=kwargs,
)
@ -82,6 +87,7 @@ class GenerateContentToCompletionHandler:
def generate_content_handler(
model: str,
contents: Union[List[Dict[str, Any]], Dict[str, Any]],
litellm_params: GenericLiteLLMParams,
config: Optional[Dict[str, Any]] = None,
stream: bool = False,
_is_async: bool = False,
@ -95,6 +101,7 @@ class GenerateContentToCompletionHandler:
contents=contents,
config=config,
stream=stream,
litellm_params=litellm_params,
**kwargs,
)
@ -103,6 +110,7 @@ class GenerateContentToCompletionHandler:
contents=contents,
config=config,
stream=stream,
litellm_params=litellm_params,
extra_kwargs=kwargs,
)

View file

@ -13,6 +13,7 @@ from litellm.types.llms.openai import (
ChatCompletionToolParam,
ChatCompletionUserMessage,
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import (
AdapterCompletionStreamWrapper,
Choices,
@ -107,8 +108,9 @@ class GoogleGenAIAdapter:
model: str,
contents: Union[List[Dict[str, Any]], Dict[str, Any]],
config: Optional[Dict[str, Any]] = None,
litellm_params: Optional[GenericLiteLLMParams] = None,
**kwargs,
) -> ChatCompletionRequest:
) -> Dict[str, Any]:
"""
Transform generate_content request to litellm completion format
@ -119,7 +121,7 @@ class GoogleGenAIAdapter:
**kwargs: Additional parameters
Returns:
ChatCompletionRequest in OpenAI format
Dict in OpenAI format
"""
# Normalize contents to list format
@ -131,11 +133,11 @@ class GoogleGenAIAdapter:
# Transform contents to OpenAI messages format
messages = self._transform_contents_to_messages(contents_list)
# Create base request
completion_request: ChatCompletionRequest = ChatCompletionRequest(
model=model,
messages=messages,
)
# Create base request as dict (which is compatible with ChatCompletionRequest)
completion_request: ChatCompletionRequest = {
"model": model,
"messages": messages,
}
#########################################################
# Supported OpenAI chat completion params
@ -182,8 +184,40 @@ class GoogleGenAIAdapter:
)
if tool_choice:
completion_request["tool_choice"] = tool_choice
#########################################################
# forward any litellm specific params
#########################################################
completion_request_dict = dict(completion_request)
if litellm_params:
completion_request_dict = self._add_generic_litellm_params_to_request(
completion_request_dict=completion_request_dict,
litellm_params=litellm_params
)
return completion_request
return completion_request_dict
def _add_generic_litellm_params_to_request(
self,
completion_request_dict: Dict[str, Any],
litellm_params: Optional[GenericLiteLLMParams] = None
) -> dict:
"""Add generic litellm params to request. e.g add api_base, api_key, api_version, etc.
Args:
completion_request_dict: Dict[str, Any]
litellm_params: GenericLiteLLMParams
Returns:
Dict[str, Any]
"""
allowed_fields = GenericLiteLLMParams.model_fields.keys()
if litellm_params:
litellm_dict = litellm_params.model_dump(exclude_none=True)
for key, value in litellm_dict.items():
if key in allowed_fields:
completion_request_dict[key] = value
return completion_request_dict
def translate_completion_output_params_streaming(
self, completion_stream: Any

View file

@ -29,7 +29,7 @@ else:
GenerateContentConfigDict = Any
GenerateContentContentListUnionDict = Any
GenerateContentResponse = Any
####### ENVIRONMENT VARIABLES ###################
# Initialize any necessary instances or variables here
base_llm_http_handler = BaseLLMHTTPHandler()
@ -38,6 +38,7 @@ base_llm_http_handler = BaseLLMHTTPHandler()
class GenerateContentSetupResult(BaseModel):
"""Internal Type - Result of setting up a generate content call"""
model: str
request_body: Dict[str, Any]
custom_llm_provider: str
@ -53,7 +54,7 @@ class GenerateContentSetupResult(BaseModel):
class GenerateContentHelper:
"""Helper class for Google GenAI generate content operations"""
@staticmethod
def mock_generate_content_response(
mock_response: str = "This is a mock response from Google GenAI generate_content.",
@ -63,20 +64,17 @@ class GenerateContentHelper:
"text": mock_response,
"candidates": [
{
"content": {
"parts": [{"text": mock_response}],
"role": "model"
},
"content": {"parts": [{"text": mock_response}], "role": "model"},
"finishReason": "STOP",
"index": 0,
"safetyRatings": []
"safetyRatings": [],
}
],
"usageMetadata": {
"promptTokenCount": 10,
"candidatesTokenCount": 20,
"totalTokenCount": 30
}
"totalTokenCount": 30,
},
}
@staticmethod
@ -86,11 +84,11 @@ class GenerateContentHelper:
config: Optional[GenerateContentConfigDict] = None,
custom_llm_provider: Optional[str] = None,
stream: bool = False,
**kwargs
**kwargs,
) -> GenerateContentSetupResult:
"""
Common setup logic for generate_content calls
Args:
model: The model name
contents: The content to generate from
@ -99,18 +97,24 @@ class GenerateContentHelper:
stream: Whether this is a streaming call
local_vars: Local variables from the calling function
**kwargs: Additional keyword arguments
Returns:
GenerateContentSetupResult containing all setup information
"""
litellm_logging_obj: Optional[LiteLLMLoggingObj] = kwargs.get("litellm_logging_obj")
litellm_logging_obj: Optional[LiteLLMLoggingObj] = kwargs.get(
"litellm_logging_obj"
)
litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
# get llm provider logic
litellm_params = GenericLiteLLMParams(**kwargs)
## MOCK RESPONSE LOGIC (only for non-streaming)
if not stream and litellm_params.mock_response and isinstance(litellm_params.mock_response, str):
if (
not stream
and litellm_params.mock_response
and isinstance(litellm_params.mock_response, str)
):
raise ValueError("Mock response should be handled by caller")
(
@ -126,11 +130,11 @@ class GenerateContentHelper:
)
# get provider config
generate_content_provider_config: Optional[BaseGoogleGenAIGenerateContentConfig] = (
ProviderConfigManager.get_provider_google_genai_generate_content_config(
model=model,
provider=litellm.LlmProviders(custom_llm_provider),
)
generate_content_provider_config: Optional[
BaseGoogleGenAIGenerateContentConfig
] = ProviderConfigManager.get_provider_google_genai_generate_content_config(
model=model,
provider=litellm.LlmProviders(custom_llm_provider),
)
if generate_content_provider_config is None:
@ -146,28 +150,31 @@ class GenerateContentHelper:
generate_content_config_dict=dict(config or {}),
litellm_params=litellm_params,
litellm_logging_obj=litellm_logging_obj,
litellm_call_id=litellm_call_id
litellm_call_id=litellm_call_id,
)
#########################################################################################
# Construct request body
#########################################################################################
# Create Google Optional Params Config
generate_content_config_dict = generate_content_provider_config.map_generate_content_optional_params(
generate_content_config_dict=config or {},
model=model,
generate_content_config_dict = (
generate_content_provider_config.map_generate_content_optional_params(
generate_content_config_dict=config or {},
model=model,
)
)
request_body = generate_content_provider_config.transform_generate_content_request(
model=model,
contents=contents,
generate_content_config_dict=generate_content_config_dict,
request_body = (
generate_content_provider_config.transform_generate_content_request(
model=model,
contents=contents,
generate_content_config_dict=generate_content_config_dict,
)
)
# Pre Call logging
if litellm_logging_obj is None:
raise ValueError("litellm_logging_obj is required, but got None")
litellm_logging_obj.update_environment_variables(
model=model,
optional_params=dict(generate_content_config_dict),
@ -185,7 +192,7 @@ class GenerateContentHelper:
generate_content_config_dict=generate_content_config_dict,
litellm_params=litellm_params,
litellm_logging_obj=litellm_logging_obj,
litellm_call_id=litellm_call_id
litellm_call_id=litellm_call_id,
)
@ -202,7 +209,7 @@ async def agenerate_content(
timeout: Optional[Union[float, httpx.Timeout]] = None,
# LiteLLM specific params,
custom_llm_provider: Optional[str] = None,
**kwargs
**kwargs,
) -> Any:
"""
Async: Generate content using Google GenAI
@ -273,10 +280,12 @@ def generate_content(
local_vars = locals()
try:
_is_async = kwargs.pop("agenerate_content", False) is True
# Check for mock response first
litellm_params = GenericLiteLLMParams(**kwargs)
if litellm_params.mock_response and isinstance(litellm_params.mock_response, str):
if litellm_params.mock_response and isinstance(
litellm_params.mock_response, str
):
return GenerateContentHelper.mock_generate_content_response(
mock_response=litellm_params.mock_response
)
@ -288,19 +297,20 @@ def generate_content(
config=config,
custom_llm_provider=custom_llm_provider,
stream=False,
**kwargs
**kwargs,
)
# Check if we should use the adapter (when provider config is None)
if setup_result.generate_content_provider_config is None:
# Use the adapter to convert to completion format
return GenerateContentToCompletionHandler.generate_content_handler(
model=setup_result.model,
model=model,
contents=contents, # type: ignore
config=setup_result.generate_content_config_dict,
stream=False,
_is_async=_is_async,
**kwargs
litellm_params=setup_result.litellm_params,
**kwargs,
)
# Call the standard handler
@ -345,7 +355,7 @@ async def agenerate_content_stream(
timeout: Optional[Union[float, httpx.Timeout]] = None,
# LiteLLM specific params,
custom_llm_provider: Optional[str] = None,
**kwargs
**kwargs,
) -> Any:
"""
Async: Generate content using Google GenAI with streaming response
@ -353,7 +363,7 @@ async def agenerate_content_stream(
local_vars = locals()
try:
kwargs["agenerate_content_stream"] = True
# get custom llm provider so we can use this for mapping exceptions
if custom_llm_provider is None:
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
@ -362,21 +372,24 @@ async def agenerate_content_stream(
# Setup the call
setup_result = GenerateContentHelper.setup_generate_content_call(
model=model,
contents=contents,
config=config,
custom_llm_provider=custom_llm_provider,
stream=True,
**kwargs
**{
"model": model,
"contents": contents,
"config": config,
"custom_llm_provider": custom_llm_provider,
"stream": True,
**kwargs,
}
)
# Check if we should use the adapter (when provider config is None)
if setup_result.generate_content_provider_config is None:
# Use the adapter to convert to completion format
return await GenerateContentToCompletionHandler.async_generate_content_handler(
model=setup_result.model,
model=model,
contents=contents, # type: ignore
config=setup_result.generate_content_config_dict,
litellm_params=setup_result.litellm_params,
stream=True,
**kwargs
)
@ -399,7 +412,7 @@ async def agenerate_content_stream(
stream=True,
litellm_metadata=kwargs.get("litellm_metadata", {}),
)
except Exception as e:
raise litellm.exception_type(
model=model,
@ -440,19 +453,20 @@ def generate_content_stream(
config=config,
custom_llm_provider=custom_llm_provider,
stream=True,
**kwargs
**kwargs,
)
# Check if we should use the adapter (when provider config is None)
if setup_result.generate_content_provider_config is None:
# Use the adapter to convert to completion format
return GenerateContentToCompletionHandler.generate_content_handler(
model=setup_result.model,
model=model,
contents=contents, # type: ignore
config=setup_result.generate_content_config_dict,
stream=True,
_is_async=_is_async,
**kwargs
litellm_params=setup_result.litellm_params,
**kwargs,
)
# Call the handler with streaming enabled (sync version)
@ -481,4 +495,3 @@ def generate_content_stream(
completion_kwargs=local_vars,
extra_kwargs=kwargs,
)

View file

@ -29,6 +29,7 @@ class AnthropicCacheControlHook(CustomPromptManagement):
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
"""
Apply cache control directives based on specified injection points.
@ -80,10 +81,10 @@ class AnthropicCacheControlHook(CustomPromptManagement):
# Case 1: Target by specific index
if targetted_index is not None:
if 0 <= targetted_index < len(messages):
messages[
targetted_index
] = AnthropicCacheControlHook._safe_insert_cache_control_in_message(
messages[targetted_index], control
messages[targetted_index] = (
AnthropicCacheControlHook._safe_insert_cache_control_in_message(
messages[targetted_index], control
)
)
# Case 2: Target by role
elif targetted_role is not None:

View file

@ -12,6 +12,7 @@ from litellm.integrations.arize import _utils
from litellm.integrations.opentelemetry import OpenTelemetry
from litellm.types.integrations.arize import ArizeConfig
from litellm.types.services import ServiceLoggerPayload
from litellm.types.utils import StandardCallbackDynamicParams
if TYPE_CHECKING:
from opentelemetry.trace import Span as _Span
@ -102,3 +103,41 @@ class ArizeLogger(OpenTelemetry):
):
"""Arize is used mainly for LLM I/O tracing, sending Proxy Server Request adds bloat to arize logs"""
pass
def construct_dynamic_otel_headers(
self,
standard_callback_dynamic_params: StandardCallbackDynamicParams
) -> Optional[dict]:
"""
Construct dynamic Arize headers from standard callback dynamic params
This is used for team/key based logging.
Returns:
dict: A dictionary of dynamic Arize headers
"""
dynamic_headers = {}
#########################################################
# `arize-space-id` handling
# the suggested param is `arize_space_key`
#########################################################
if standard_callback_dynamic_params.get("arize_space_id"):
dynamic_headers["arize-space-id"] = standard_callback_dynamic_params.get(
"arize_space_id"
)
if standard_callback_dynamic_params.get("arize_space_key"):
dynamic_headers["arize-space-id"] = standard_callback_dynamic_params.get(
"arize_space_key"
)
#########################################################
# `api_key` handling
#########################################################
if standard_callback_dynamic_params.get("arize_api_key"):
dynamic_headers["api_key"] = standard_callback_dynamic_params.get(
"arize_api_key"
)
return dynamic_headers

View file

@ -1,5 +1,5 @@
from datetime import datetime
from typing import Dict, List, Literal, Optional, Union
from typing import Dict, List, Literal, Optional, Type, Union
from litellm._logging import verbose_logger
from litellm.integrations.custom_logger import CustomLogger
@ -9,6 +9,7 @@ from litellm.types.guardrails import (
LitellmParams,
PiiEntityType,
)
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
from litellm.types.utils import StandardLoggingGuardrailInformation
@ -46,19 +47,34 @@ class CustomGuardrail(CustomLogger):
self.mask_response_content: bool = mask_response_content
if supported_event_hooks:
## validate event_hook is in supported_event_hooks
self._validate_event_hook(event_hook, supported_event_hooks)
super().__init__(**kwargs)
@staticmethod
def get_config_model() -> Optional[Type["GuardrailConfigModel"]]:
"""
Returns the config model for the guardrail
This is used to render the config model in the UI.
"""
return None
def _validate_event_hook(
self,
event_hook: Optional[Union[GuardrailEventHooks, List[GuardrailEventHooks]]],
supported_event_hooks: List[GuardrailEventHooks],
) -> None:
if event_hook is None:
return
if isinstance(event_hook, str):
event_hook = GuardrailEventHooks(event_hook)
if isinstance(event_hook, list):
for hook in event_hook:
if isinstance(hook, str):
hook = GuardrailEventHooks(hook)
if hook not in supported_event_hooks:
raise ValueError(
f"Event hook {hook} is not in the supported event hooks {supported_event_hooks}"
@ -86,10 +102,13 @@ class CustomGuardrail(CustomLogger):
for _guardrail in requested_guardrails:
if isinstance(_guardrail, dict):
if self.guardrail_name in _guardrail:
return True
elif isinstance(_guardrail, str):
if self.guardrail_name == _guardrail:
return True
return False
def should_run_guardrail(self, data, event_type: GuardrailEventHooks) -> bool:
@ -336,14 +355,11 @@ def log_guardrail_information(func):
import asyncio
import functools
start_time = datetime.now()
@functools.wraps(func)
async def async_wrapper(*args, **kwargs):
start_time = datetime.now() # Move start_time inside the wrapper
self: CustomGuardrail = args[0]
request_data: Optional[dict] = (
kwargs.get("data") or kwargs.get("request_data") or {}
)
request_data: dict = kwargs.get("data") or kwargs.get("request_data") or {}
try:
response = await func(*args, **kwargs)
return self._process_response(
@ -364,10 +380,9 @@ def log_guardrail_information(func):
@functools.wraps(func)
def sync_wrapper(*args, **kwargs):
start_time = datetime.now() # Move start_time inside the wrapper
self: CustomGuardrail = args[0]
request_data: Optional[dict] = (
kwargs.get("data") or kwargs.get("request_data") or {}
)
request_data: dict = kwargs.get("data") or kwargs.get("request_data") or {}
try:
response = func(*args, **kwargs)
return self._process_response(

View file

@ -89,6 +89,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
litellm_logging_obj: LiteLLMLoggingObj,
tools: Optional[List[Dict]] = None,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
"""
Returns:
@ -107,6 +108,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
"""
Returns:
@ -408,3 +410,20 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
if len(text) > max_length
else text
)
def _select_metadata_field(
self, request_kwargs: Optional[Dict] = None
) -> Optional[str]:
"""
Select the metadata field to use for logging
1. If `litellm_metadata` is in the request kwargs, use it
2. Otherwise, use `metadata`
"""
from litellm.constants import LITELLM_METADATA_FIELD, OLD_LITELLM_METADATA_FIELD
if request_kwargs is None:
return None
if LITELLM_METADATA_FIELD in request_kwargs:
return LITELLM_METADATA_FIELD
return OLD_LITELLM_METADATA_FIELD

View file

@ -19,6 +19,7 @@ class CustomPromptManagement(CustomLogger, PromptManagementBase):
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
"""
Returns:
@ -45,6 +46,7 @@ class CustomPromptManagement(CustomLogger, PromptManagementBase):
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> PromptManagementClient:
raise NotImplementedError(
"Custom prompt management does not support compile prompt helper"

View file

@ -576,4 +576,4 @@ class DataDogLogger(
start_time_utc: Optional[datetimeObj],
end_time_utc: Optional[datetimeObj],
) -> Optional[dict]:
pass
pass

View file

@ -100,7 +100,7 @@ class DeepEvalLogger(CustomLogger):
except Exception as e:
raise e
verbose_logger.debug(
"DeepEvalLogger: sync_log_failure_event: Api response", response
"DeepEvalLogger: sync_log_failure_event: Api response %s", response
)
async def _async_event_handler(
@ -116,7 +116,7 @@ class DeepEvalLogger(CustomLogger):
)
verbose_logger.debug(
"DeepEvalLogger: async_event_handler: Api response", response
"DeepEvalLogger: async_event_handler: Api response %s", response
)
def _create_base_api_span(

View file

@ -156,7 +156,12 @@ class HumanloopLogger(CustomLogger):
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
) -> Tuple[str, List[AllMessageValues], dict,]:
prompt_version: Optional[int] = None,
) -> Tuple[
str,
List[AllMessageValues],
dict,
]:
humanloop_api_key = dynamic_callback_params.get(
"humanloop_api_key"
) or get_secret_str("HUMANLOOP_API_KEY")

View file

@ -1,6 +1,7 @@
import base64
import os
from typing import TYPE_CHECKING, Any, Union
from urllib.parse import quote
from litellm._logging import verbose_logger
from litellm.integrations.arize import _utils
@ -9,10 +10,11 @@ from litellm.types.integrations.langfuse_otel import LangfuseOtelConfig
if TYPE_CHECKING:
from opentelemetry.trace import Span as _Span
from litellm.integrations.opentelemetry import (
OpenTelemetryConfig as _OpenTelemetryConfig,
)
from litellm.types.integrations.arize import Protocol as _Protocol
from litellm.integrations.opentelemetry import OpenTelemetryConfig as _OpenTelemetryConfig
Protocol = _Protocol
OpenTelemetryConfig = _OpenTelemetryConfig
Span = Union[_Span, Any]
@ -54,7 +56,7 @@ class LangfuseOtelLogger:
"""
public_key = os.environ.get("LANGFUSE_PUBLIC_KEY", None)
secret_key = os.environ.get("LANGFUSE_SECRET_KEY", None)
if not public_key or not secret_key:
raise ValueError(
"LANGFUSE_PUBLIC_KEY and LANGFUSE_SECRET_KEY must be set for Langfuse OpenTelemetry integration."
@ -62,7 +64,7 @@ class LangfuseOtelLogger:
# Determine endpoint - default to US cloud
langfuse_host = os.environ.get("LANGFUSE_HOST", None)
if langfuse_host:
# If LANGFUSE_HOST is provided, construct OTEL endpoint from it
if not langfuse_host.startswith("http"):
@ -77,13 +79,13 @@ class LangfuseOtelLogger:
# Create Basic Auth header
auth_string = f"{public_key}:{secret_key}"
auth_header = base64.b64encode(auth_string.encode()).decode()
otlp_auth_headers = f"Authorization=Basic {auth_header}"
# URL encode the entire header value as required by OpenTelemetry specification
otlp_auth_headers = f"Authorization={quote(f'Basic {auth_header}')}"
# Set standard OTEL environment variables
os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = endpoint
os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = otlp_auth_headers
return LangfuseOtelConfig(
otlp_auth_headers=otlp_auth_headers,
protocol="otlp_http"
)
otlp_auth_headers=otlp_auth_headers, protocol="otlp_http"
)

View file

@ -134,8 +134,14 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
langfuse_prompt_id: str,
langfuse_client: LangfuseClass,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> PROMPT_CLIENT:
return langfuse_client.get_prompt(langfuse_prompt_id, label=prompt_label)
prompt_client = langfuse_client.get_prompt(
langfuse_prompt_id, label=prompt_label, version=prompt_version
)
return prompt_client
def _compile_prompt(
self,
@ -180,7 +186,12 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
litellm_logging_obj: LiteLLMLoggingObj,
tools: Optional[List[Dict]] = None,
prompt_label: Optional[str] = None,
) -> Tuple[str, List[AllMessageValues], dict,]:
prompt_version: Optional[int] = None,
) -> Tuple[
str,
List[AllMessageValues],
dict,
]:
return self.get_chat_completion_prompt(
model,
messages,
@ -189,6 +200,7 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
prompt_variables,
dynamic_callback_params,
prompt_label=prompt_label,
prompt_version=prompt_version,
)
def should_run_prompt_management(
@ -203,7 +215,8 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
langfuse_host=dynamic_callback_params.get("langfuse_host"),
)
langfuse_prompt_client = self._get_prompt_from_id(
langfuse_prompt_id=prompt_id, langfuse_client=langfuse_client
langfuse_prompt_id=prompt_id,
langfuse_client=langfuse_client,
)
return langfuse_prompt_client is not None
@ -213,6 +226,7 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> PromptManagementClient:
langfuse_client = langfuse_client_init(
langfuse_public_key=dynamic_callback_params.get("langfuse_public_key"),
@ -224,6 +238,7 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
langfuse_prompt_id=prompt_id,
langfuse_client=langfuse_client,
prompt_label=prompt_label,
prompt_version=prompt_version,
)
## SET PROMPT

View file

@ -65,9 +65,12 @@ class OpenMeterLogger(CustomLogger):
"total_tokens": response_obj["usage"].get("total_tokens"),
}
subject = (kwargs.get("user", None),) # end-user passed in via 'user' param
if not subject:
user_param = kwargs.get("user", None) # end-user passed in via 'user' param
if user_param is None:
raise Exception("OpenMeter: user is required")
# Ensure subject is always a string for OpenMeter API
subject = str(user_param)
return {
"specversion": "1.0",

View file

@ -19,6 +19,7 @@ if TYPE_CHECKING:
from opentelemetry.sdk.trace.export import SpanExporter as _SpanExporter
from opentelemetry.trace import Context as _Context
from opentelemetry.trace import Span as _Span
from opentelemetry.trace import Tracer as _Tracer
from litellm.proxy._types import (
ManagementEndpointLoggingPayload as _ManagementEndpointLoggingPayload,
@ -26,12 +27,14 @@ if TYPE_CHECKING:
from litellm.proxy.proxy_server import UserAPIKeyAuth as _UserAPIKeyAuth
Span = Union[_Span, Any]
Tracer = Union[_Tracer, Any]
Context = Union[_Context, Any]
SpanExporter = Union[_SpanExporter, Any]
UserAPIKeyAuth = Union[_UserAPIKeyAuth, Any]
ManagementEndpointLoggingPayload = Union[_ManagementEndpointLoggingPayload, Any]
else:
Span = Any
Tracer = Any
SpanExporter = Any
UserAPIKeyAuth = Any
ManagementEndpointLoggingPayload = Any
@ -313,6 +316,71 @@ class OpenTelemetry(CustomLogger):
# End Parent OTEL Sspan
parent_otel_span.end(end_time=self._to_ns(datetime.now()))
#########################################################
# Team/Key Based Logging Control Flow
#########################################################
def get_tracer_to_use_for_request(self, kwargs: dict) -> Tracer:
"""
Get the tracer to use for this request
If dynamic headers are present, a temporary tracer is created with the dynamic headers.
Otherwise, the default tracer is used.
Returns:
Tracer: The tracer to use for this request
"""
dynamic_headers = self._get_dynamic_otel_headers_from_kwargs(kwargs)
if dynamic_headers is not None:
# Create spans using a temporary tracer with dynamic headers
tracer_to_use = self._get_tracer_with_dynamic_headers(dynamic_headers)
verbose_logger.debug("Using dynamic headers for this request: %s", dynamic_headers)
else:
tracer_to_use = self.tracer
return tracer_to_use
def _get_dynamic_otel_headers_from_kwargs(self, kwargs) -> Optional[dict]:
"""Extract dynamic headers from kwargs if available."""
standard_callback_dynamic_params: Optional[
StandardCallbackDynamicParams
] = kwargs.get("standard_callback_dynamic_params")
if not standard_callback_dynamic_params:
return None
dynamic_headers = self.construct_dynamic_otel_headers(
standard_callback_dynamic_params=standard_callback_dynamic_params
)
return dynamic_headers if dynamic_headers else None
def _get_tracer_with_dynamic_headers(self, dynamic_headers: dict):
"""Create a temporary tracer with dynamic headers for this request only."""
from opentelemetry.sdk.resources import Resource
from opentelemetry.sdk.trace import TracerProvider
# Create a temporary tracer provider with dynamic headers
temp_provider = TracerProvider(resource=Resource(attributes=LITELLM_RESOURCE))
temp_provider.add_span_processor(self._get_span_processor(dynamic_headers=dynamic_headers))
return temp_provider.get_tracer(LITELLM_TRACER_NAME)
def construct_dynamic_otel_headers(self, standard_callback_dynamic_params: StandardCallbackDynamicParams) -> Optional[dict]:
"""
Construct dynamic headers from standard callback dynamic params
Note: You just need to override this method in Arize, Langfuse Otel if you want to allow team/key based logging.
Returns:
dict: A dictionary of dynamic headers
"""
return None
#########################################################
# End of Team/Key Based Logging Control Flow
#########################################################
def _handle_sucess(self, kwargs, response_obj, start_time, end_time):
from opentelemetry import trace
@ -323,12 +391,11 @@ class OpenTelemetry(CustomLogger):
kwargs,
self.config,
)
_parent_context, parent_otel_span = self._get_span_context(kwargs)
self._add_dynamic_span_processor_if_needed(kwargs)
# Span 1: Requst sent to litellm SDK
span = self.tracer.start_span(
# Span 1: Request sent to litellm SDK
otel_tracer: Tracer = self.get_tracer_to_use_for_request(kwargs)
span = otel_tracer.start_span(
name=self._get_span_name(kwargs),
start_time=self._to_ns(start_time),
context=_parent_context,
@ -342,7 +409,7 @@ class OpenTelemetry(CustomLogger):
pass
else:
# Span 2: Raw Request / Response to LLM
raw_request_span = self.tracer.start_span(
raw_request_span = otel_tracer.start_span(
name=RAW_REQUEST_SPAN_NAME,
start_time=self._to_ns(start_time),
context=trace.set_span_in_context(span),
@ -387,7 +454,8 @@ class OpenTelemetry(CustomLogger):
if end_time_float is not None:
end_time_datetime = datetime.fromtimestamp(end_time_float)
guardrail_span = self.tracer.start_span(
otel_tracer: Tracer = self.get_tracer_to_use_for_request(kwargs)
guardrail_span = otel_tracer.start_span(
name="guardrail",
start_time=self._to_ns(start_time_datetime),
context=context,
@ -420,44 +488,6 @@ class OpenTelemetry(CustomLogger):
guardrail_span.end(end_time=self._to_ns(end_time_datetime))
def _add_dynamic_span_processor_if_needed(self, kwargs):
"""
Helper method to add a span processor with dynamic headers if needed.
This allows for per-request configuration of telemetry exporters by
extracting headers from standard_callback_dynamic_params.
"""
from opentelemetry import trace
standard_callback_dynamic_params: Optional[
StandardCallbackDynamicParams
] = kwargs.get("standard_callback_dynamic_params")
if not standard_callback_dynamic_params:
return
# Extract headers from dynamic params
dynamic_headers = {}
# Handle Arize headers
if standard_callback_dynamic_params.get("arize_space_key"):
dynamic_headers["space_key"] = standard_callback_dynamic_params.get(
"arize_space_key"
)
if standard_callback_dynamic_params.get("arize_api_key"):
dynamic_headers["api_key"] = standard_callback_dynamic_params.get(
"arize_api_key"
)
# Only create a span processor if we have headers to use
if len(dynamic_headers) > 0:
from opentelemetry.sdk.trace import TracerProvider
provider = trace.get_tracer_provider()
if isinstance(provider, TracerProvider):
span_processor = self._get_span_processor(
dynamic_headers=dynamic_headers
)
provider.add_span_processor(span_processor)
def _handle_failure(self, kwargs, response_obj, start_time, end_time):
from opentelemetry.trace import Status, StatusCode
@ -470,7 +500,8 @@ class OpenTelemetry(CustomLogger):
_parent_context, parent_otel_span = self._get_span_context(kwargs)
# Span 1: Requst sent to litellm SDK
span = self.tracer.start_span(
otel_tracer: Tracer = self.get_tracer_to_use_for_request(kwargs)
span = otel_tracer.start_span(
name=self._get_span_name(kwargs),
start_time=self._to_ns(start_time),
context=_parent_context,
@ -579,7 +610,9 @@ class OpenTelemetry(CustomLogger):
)
return
elif self.callback_name == "langfuse_otel":
from litellm.integrations.langfuse.langfuse_otel import LangfuseOtelLogger
from litellm.integrations.langfuse.langfuse_otel import (
LangfuseOtelLogger,
)
LangfuseOtelLogger.set_langfuse_otel_attributes(
span, kwargs, response_obj

View file

@ -353,29 +353,289 @@ class PrometheusLogger(CustomLogger):
verbose_logger.debug(f"prometheus config: {config}")
label_filters = {}
# Parse and validate all configuration groups
parsed_configs = []
self.enabled_metrics = set()
# Parse each configuration group
for group_config in config:
# Validate configuration using Pydantic
if isinstance(group_config, dict):
parsed_config = PrometheusMetricsConfig(**group_config)
else:
parsed_config = group_config
# Add enabled metrics to the set
parsed_configs.append(parsed_config)
self.enabled_metrics.update(parsed_config.metrics)
# Set label filters for each metric in this group
for metric_name in parsed_config.metrics:
if parsed_config.include_labels:
label_filters[metric_name] = parsed_config.include_labels
# Validate all configurations
validation_results = self._validate_all_configurations(parsed_configs)
if validation_results.has_errors:
self._pretty_print_validation_errors(validation_results)
error_message = "Configuration validation failed:\n" + "\n".join(validation_results.all_error_messages)
raise ValueError(error_message)
# Build label filters from valid configurations
label_filters = self._build_label_filters(parsed_configs)
# Pretty print the processed configuration
self._pretty_print_prometheus_config(label_filters)
return label_filters
def _validate_all_configurations(self, parsed_configs: List) -> ValidationResults:
"""Validate all metric configurations and return collected errors"""
metric_errors = []
label_errors = []
for config in parsed_configs:
for metric_name in config.metrics:
# Validate metric name
metric_error = self._validate_single_metric_name(metric_name)
if metric_error:
metric_errors.append(metric_error)
continue # Skip label validation if metric name is invalid
# Validate labels if provided
if config.include_labels:
label_error = self._validate_single_metric_labels(metric_name, config.include_labels)
if label_error:
label_errors.append(label_error)
return ValidationResults(metric_errors=metric_errors, label_errors=label_errors)
def _validate_single_metric_name(self, metric_name: str) -> Optional[MetricValidationError]:
"""Validate a single metric name"""
from typing import get_args
if metric_name not in set(get_args(DEFINED_PROMETHEUS_METRICS)):
return MetricValidationError(
metric_name=metric_name,
valid_metrics=get_args(DEFINED_PROMETHEUS_METRICS)
)
return None
def _validate_single_metric_labels(self, metric_name: str, labels: List[str]) -> Optional[LabelValidationError]:
"""Validate labels for a single metric"""
from typing import cast
# Get valid labels for this metric from PrometheusMetricLabels
valid_labels = PrometheusMetricLabels.get_labels(cast(DEFINED_PROMETHEUS_METRICS, metric_name))
# Find invalid labels
invalid_labels = [label for label in labels if label not in valid_labels]
if invalid_labels:
return LabelValidationError(
metric_name=metric_name,
invalid_labels=invalid_labels,
valid_labels=valid_labels
)
return None
def _build_label_filters(self, parsed_configs: List) -> Dict[str, List[str]]:
"""Build label filters from validated configurations"""
label_filters = {}
for config in parsed_configs:
for metric_name in config.metrics:
if config.include_labels:
# Only add if metric name is valid (validation already passed)
if self._validate_single_metric_name(metric_name) is None:
label_filters[metric_name] = config.include_labels
return label_filters
def _validate_configured_metric_labels(self, metric_name: str, labels: List[str]):
"""
Ensure that all the configured labels are valid for the metric
Raises ValueError if the metric labels are invalid and pretty prints the error
"""
label_error = self._validate_single_metric_labels(metric_name, labels)
if label_error:
self._pretty_print_invalid_labels_error(
metric_name=label_error.metric_name,
invalid_labels=label_error.invalid_labels,
valid_labels=label_error.valid_labels
)
raise ValueError(label_error.message)
return True
#########################################################
# Pretty print functions
#########################################################
def _pretty_print_validation_errors(self, validation_results: ValidationResults) -> None:
"""Pretty print all validation errors using rich"""
try:
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text
console = Console()
# Create error panel title
title = Text("🚨🚨 Configuration Validation Errors", style="bold red")
# Print main error panel
console.print("\n")
console.print(Panel(title, border_style="red"))
# Show invalid metric names if any
if validation_results.metric_errors:
invalid_metrics = [e.metric_name for e in validation_results.metric_errors]
valid_metrics = validation_results.metric_errors[0].valid_metrics # All should have same valid metrics
metrics_error_text = Text(
f"Invalid Metric Names: {', '.join(invalid_metrics)}",
style="bold red"
)
console.print(Panel(metrics_error_text, border_style="red"))
metrics_table = Table(
title="📊 Valid Metric Names",
show_header=True,
header_style="bold green",
title_justify="left",
border_style="green",
)
metrics_table.add_column("Available Metrics", style="cyan", no_wrap=True)
for metric in sorted(valid_metrics):
metrics_table.add_row(metric)
console.print(metrics_table)
# Show invalid labels if any
if validation_results.label_errors:
for error in validation_results.label_errors:
labels_error_text = Text(
f"Invalid Labels for '{error.metric_name}': {', '.join(error.invalid_labels)}",
style="bold red"
)
console.print(Panel(labels_error_text, border_style="red"))
labels_table = Table(
title=f"🏷️ Valid Labels for '{error.metric_name}'",
show_header=True,
header_style="bold green",
title_justify="left",
border_style="green",
)
labels_table.add_column("Valid Labels", style="cyan", no_wrap=True)
for label in sorted(error.valid_labels):
labels_table.add_row(label)
console.print(labels_table)
console.print("\n")
except ImportError:
# Fallback to simple logging if rich is not available
for metric_error in validation_results.metric_errors:
verbose_logger.error(metric_error.message)
for label_error in validation_results.label_errors:
verbose_logger.error(label_error.message)
def _pretty_print_invalid_labels_error(
self, metric_name: str, invalid_labels: List[str], valid_labels: List[str]
) -> None:
"""Pretty print error message for invalid labels using rich"""
try:
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text
console = Console()
# Create error panel title
title = Text(
f"🚨🚨 Invalid Labels for Metric: '{metric_name}'\nInvalid labels: {', '.join(invalid_labels)}\nPlease specify only valid labels below",
style="bold red"
)
# Create valid labels table
labels_table = Table(
title="🏷️ Valid Labels for this Metric",
show_header=True,
header_style="bold green",
title_justify="left",
border_style="green",
)
labels_table.add_column("Valid Labels", style="cyan", no_wrap=True)
for label in sorted(valid_labels):
labels_table.add_row(label)
# Print everything in a nice panel
console.print("\n")
console.print(Panel(title, border_style="red"))
console.print(labels_table)
console.print("\n")
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.error(
f"Invalid labels for metric '{metric_name}': {invalid_labels}. Valid labels: {sorted(valid_labels)}"
)
def _pretty_print_invalid_metric_error(
self, invalid_metric_name: str, valid_metrics: tuple
) -> None:
"""Pretty print error message for invalid metric name using rich"""
try:
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text
console = Console()
# Create error panel title
title = Text(f"🚨🚨 Invalid Metric Name: '{invalid_metric_name}'\nPlease specify one of the allowed metrics below", style="bold red")
# Create valid metrics table
metrics_table = Table(
title="📊 Valid Metric Names",
show_header=True,
header_style="bold green",
title_justify="left",
border_style="green",
)
metrics_table.add_column("Available Metrics", style="cyan", no_wrap=True)
for metric in sorted(valid_metrics):
metrics_table.add_row(metric)
# Print everything in a nice panel
console.print("\n")
console.print(Panel(title, border_style="red"))
console.print(metrics_table)
console.print("\n")
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.error(
f"Invalid metric name: {invalid_metric_name}. Valid metrics: {sorted(valid_metrics)}"
)
#########################################################
# End of pretty print functions
#########################################################
def _valid_metric_name(self, metric_name: str):
"""
Raises ValueError if the metric name is invalid and pretty prints the error
"""
error = self._validate_single_metric_name(metric_name)
if error:
self._pretty_print_invalid_metric_error(
invalid_metric_name=error.metric_name,
valid_metrics=error.valid_metrics)
raise ValueError(error.message)
def _pretty_print_prometheus_config(
self, label_filters: Dict[str, List[str]]
@ -447,6 +707,7 @@ class PrometheusLogger(CustomLogger):
)
verbose_logger.info(f"Label filters: {label_filters}")
def _is_metric_enabled(self, metric_name: str) -> bool:
"""Check if a metric is enabled based on configuration"""
# If no specific configuration is provided, enable all metrics (default behavior)

View file

@ -34,6 +34,7 @@ class PromptManagementBase(ABC):
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> PromptManagementClient:
pass
@ -51,12 +52,14 @@ class PromptManagementBase(ABC):
client_messages: List[AllMessageValues],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> PromptManagementClient:
compiled_prompt_client = self._compile_prompt_helper(
prompt_id=prompt_id,
prompt_variables=prompt_variables,
dynamic_callback_params=dynamic_callback_params,
prompt_label=prompt_label,
prompt_version=prompt_version,
)
try:
@ -86,6 +89,7 @@ class PromptManagementBase(ABC):
prompt_variables: Optional[dict],
dynamic_callback_params: StandardCallbackDynamicParams,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
if prompt_id is None:
raise ValueError("prompt_id is required for Prompt Management Base class")
@ -100,6 +104,7 @@ class PromptManagementBase(ABC):
client_messages=messages,
dynamic_callback_params=dynamic_callback_params,
prompt_label=prompt_label,
prompt_version=prompt_version,
)
completed_messages = prompt_template["completed_messages"] or messages

View file

@ -2,7 +2,7 @@
s3 Bucket Logging Integration
async_log_success_event: Processes the event, stores it in memory for DEFAULT_S3_FLUSH_INTERVAL_SECONDS seconds or until DEFAULT_S3_BATCH_SIZE and then flushes to s3
async_log_failure_event: Processes the event, stores it in memory for DEFAULT_S3_FLUSH_INTERVAL_SECONDS seconds or until DEFAULT_S3_BATCH_SIZE and then flushes to s3
NOTE 1: S3 does not provide a BATCH PUT API endpoint, so we create tasks to upload each element individually
"""
@ -197,6 +197,24 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
return
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
await self._async_log_event_base(
kwargs=kwargs,
response_obj=response_obj,
start_time=start_time,
end_time=end_time,
)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
await self._async_log_event_base(
kwargs=kwargs,
response_obj=response_obj,
start_time=start_time,
end_time=end_time,
)
pass
async def _async_log_event_base(self, kwargs, response_obj, start_time, end_time):
try:
verbose_logger.debug(
f"s3 Logging - Enters logging function for model {kwargs}"
@ -224,6 +242,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
verbose_logger.exception(f"s3 Layer Error - {str(e)}")
pass
async def async_upload_data_to_s3(
self, batch_logging_element: s3BatchLoggingElement
):

275
litellm/integrations/sqs.py Normal file
View file

@ -0,0 +1,275 @@
"""SQS Logging Integration
This logger sends ``StandardLoggingPayload`` entries to an AWS SQS queue.
"""
from __future__ import annotations
import asyncio
from typing import List, Optional
import litellm
from litellm._logging import print_verbose, verbose_logger
from litellm.constants import (
DEFAULT_SQS_BATCH_SIZE,
DEFAULT_SQS_FLUSH_INTERVAL_SECONDS,
SQS_API_VERSION,
SQS_SEND_MESSAGE_ACTION,
)
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.custom_httpx.http_handler import (
get_async_httpx_client,
httpxSpecialProvider,
)
from litellm.types.utils import StandardLoggingPayload
from .custom_batch_logger import CustomBatchLogger
class SQSLogger(CustomBatchLogger, BaseAWSLLM):
"""Batching logger that writes logs to an AWS SQS queue."""
def __init__(
self,
sqs_queue_url: Optional[str] = None,
sqs_region_name: Optional[str] = None,
sqs_api_version: Optional[str] = None,
sqs_use_ssl: bool = True,
sqs_verify: Optional[bool] = None,
sqs_endpoint_url: Optional[str] = None,
sqs_aws_access_key_id: Optional[str] = None,
sqs_aws_secret_access_key: Optional[str] = None,
sqs_aws_session_token: Optional[str] = None,
sqs_aws_session_name: Optional[str] = None,
sqs_aws_profile_name: Optional[str] = None,
sqs_aws_role_name: Optional[str] = None,
sqs_aws_web_identity_token: Optional[str] = None,
sqs_aws_sts_endpoint: Optional[str] = None,
sqs_flush_interval: Optional[int] = DEFAULT_SQS_FLUSH_INTERVAL_SECONDS,
sqs_batch_size: Optional[int] = DEFAULT_SQS_BATCH_SIZE,
sqs_config=None,
**kwargs,
) -> None:
try:
verbose_logger.debug(
f"in init sqs logger - sqs_callback_params {litellm.aws_sqs_callback_params}"
)
self.async_httpx_client = get_async_httpx_client(
llm_provider=httpxSpecialProvider.LoggingCallback,
)
self._init_sqs_params(
sqs_queue_url=sqs_queue_url,
sqs_region_name=sqs_region_name,
sqs_api_version=sqs_api_version,
sqs_use_ssl=sqs_use_ssl,
sqs_verify=sqs_verify,
sqs_endpoint_url=sqs_endpoint_url,
sqs_aws_access_key_id=sqs_aws_access_key_id,
sqs_aws_secret_access_key=sqs_aws_secret_access_key,
sqs_aws_session_token=sqs_aws_session_token,
sqs_aws_session_name=sqs_aws_session_name,
sqs_aws_profile_name=sqs_aws_profile_name,
sqs_aws_role_name=sqs_aws_role_name,
sqs_aws_web_identity_token=sqs_aws_web_identity_token,
sqs_aws_sts_endpoint=sqs_aws_sts_endpoint,
sqs_config=sqs_config,
)
asyncio.create_task(self.periodic_flush())
self.flush_lock = asyncio.Lock()
verbose_logger.debug(
f"sqs flush interval: {sqs_flush_interval}, sqs batch size: {sqs_batch_size}"
)
CustomBatchLogger.__init__(
self,
flush_lock=self.flush_lock,
flush_interval=sqs_flush_interval,
batch_size=sqs_batch_size,
)
self.log_queue: List[StandardLoggingPayload] = []
BaseAWSLLM.__init__(self)
except Exception as e:
print_verbose(f"Got exception on init sqs client {str(e)}")
raise e
def _init_sqs_params(
self,
sqs_queue_url: Optional[str] = None,
sqs_region_name: Optional[str] = None,
sqs_api_version: Optional[str] = None,
sqs_use_ssl: bool = True,
sqs_verify: Optional[bool] = None,
sqs_endpoint_url: Optional[str] = None,
sqs_aws_access_key_id: Optional[str] = None,
sqs_aws_secret_access_key: Optional[str] = None,
sqs_aws_session_token: Optional[str] = None,
sqs_aws_session_name: Optional[str] = None,
sqs_aws_profile_name: Optional[str] = None,
sqs_aws_role_name: Optional[str] = None,
sqs_aws_web_identity_token: Optional[str] = None,
sqs_aws_sts_endpoint: Optional[str] = None,
sqs_config=None,
) -> None:
litellm.aws_sqs_callback_params = litellm.aws_sqs_callback_params or {}
# read in .env variables - example os.environ/AWS_BUCKET_NAME
for key, value in litellm.aws_sqs_callback_params.items():
if isinstance(value, str) and value.startswith("os.environ/"):
litellm.aws_sqs_callback_params[key] = litellm.get_secret(value)
self.sqs_queue_url = (
litellm.aws_sqs_callback_params.get("sqs_queue_url") or sqs_queue_url
)
self.sqs_region_name = (
litellm.aws_sqs_callback_params.get("sqs_region_name") or sqs_region_name
)
self.sqs_api_version = (
litellm.aws_sqs_callback_params.get("sqs_api_version") or sqs_api_version
)
self.sqs_use_ssl = (
litellm.aws_sqs_callback_params.get("sqs_use_ssl", True) or sqs_use_ssl
)
self.sqs_verify = litellm.aws_sqs_callback_params.get("sqs_verify") or sqs_verify
self.sqs_endpoint_url = (
litellm.aws_sqs_callback_params.get("sqs_endpoint_url") or sqs_endpoint_url
)
self.sqs_aws_access_key_id = (
litellm.aws_sqs_callback_params.get("sqs_aws_access_key_id")
or sqs_aws_access_key_id
)
self.sqs_aws_secret_access_key = (
litellm.aws_sqs_callback_params.get("sqs_aws_secret_access_key")
or sqs_aws_secret_access_key
)
self.sqs_aws_session_token = (
litellm.aws_sqs_callback_params.get("sqs_aws_session_token")
or sqs_aws_session_token
)
self.sqs_aws_session_name = (
litellm.aws_sqs_callback_params.get("sqs_aws_session_name") or sqs_aws_session_name
)
self.sqs_aws_profile_name = (
litellm.aws_sqs_callback_params.get("sqs_aws_profile_name") or sqs_aws_profile_name
)
self.sqs_aws_role_name = (
litellm.aws_sqs_callback_params.get("sqs_aws_role_name") or sqs_aws_role_name
)
self.sqs_aws_web_identity_token = (
litellm.aws_sqs_callback_params.get("sqs_aws_web_identity_token")
or sqs_aws_web_identity_token
)
self.sqs_aws_sts_endpoint = (
litellm.aws_sqs_callback_params.get("sqs_aws_sts_endpoint") or sqs_aws_sts_endpoint
)
self.sqs_config = litellm.aws_sqs_callback_params.get("sqs_config") or sqs_config
async def async_log_success_event(
self, kwargs, response_obj, start_time, end_time
) -> None:
try:
verbose_logger.debug(
"SQS Logging - Enters logging function for model %s", kwargs
)
standard_logging_payload = kwargs.get("standard_logging_object")
if standard_logging_payload is None:
raise ValueError("standard_logging_payload is None")
self.log_queue.append(standard_logging_payload)
verbose_logger.debug(
"sqs logging: queue length %s, batch size %s",
len(self.log_queue),
self.batch_size,
)
except Exception as e:
verbose_logger.exception(f"sqs Layer Error - {str(e)}")
async def async_send_batch(self) -> None:
verbose_logger.debug(
f"sqs logger - sending batch of {len(self.log_queue)}"
)
if not self.log_queue:
return
for payload in self.log_queue:
asyncio.create_task(self.async_send_message(payload))
async def async_send_message(self, payload: StandardLoggingPayload) -> None:
try:
from urllib.parse import quote
import requests
from botocore.auth import SigV4Auth
from botocore.awsrequest import AWSRequest
from litellm.litellm_core_utils.asyncify import asyncify
asyncified_get_credentials = asyncify(self.get_credentials)
credentials = await asyncified_get_credentials(
aws_access_key_id=self.sqs_aws_access_key_id,
aws_secret_access_key=self.sqs_aws_secret_access_key,
aws_session_token=self.sqs_aws_session_token,
aws_region_name=self.sqs_region_name,
aws_session_name=self.sqs_aws_session_name,
aws_profile_name=self.sqs_aws_profile_name,
aws_role_name=self.sqs_aws_role_name,
aws_web_identity_token=self.sqs_aws_web_identity_token,
aws_sts_endpoint=self.sqs_aws_sts_endpoint,
)
if self.sqs_queue_url is None:
raise ValueError("sqs_queue_url not set")
json_string = safe_dumps(payload)
body = (
f"Action={SQS_SEND_MESSAGE_ACTION}&Version={SQS_API_VERSION}&MessageBody="
+ quote(json_string, safe="")
)
headers = {
"Content-Type": "application/x-www-form-urlencoded",
}
req = requests.Request(
"POST", self.sqs_queue_url, data=body, headers=headers
)
prepped = req.prepare()
aws_request = AWSRequest(
method=prepped.method,
url=prepped.url,
data=prepped.body,
headers=prepped.headers,
)
SigV4Auth(credentials, "sqs", self.sqs_region_name).add_auth(
aws_request
)
signed_headers = dict(aws_request.headers.items())
response = await self.async_httpx_client.post(
self.sqs_queue_url,
data=body,
headers=signed_headers,
)
response.raise_for_status()
except Exception as e:
verbose_logger.exception(f"Error sending to SQS: {str(e)}")

View file

@ -77,6 +77,7 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM):
litellm_logging_obj: LiteLLMLoggingObj,
tools: Optional[List[Dict]] = None,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
"""
Retrieves the context from the Bedrock Knowledge Base and appends it to the messages.
@ -129,9 +130,9 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM):
)
)
litellm_logging_obj.model_call_details[
"vector_store_request_metadata"
] = vector_store_request_metadata
litellm_logging_obj.model_call_details["vector_store_request_metadata"] = (
vector_store_request_metadata
)
return model, messages, non_default_params
@ -143,9 +144,9 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM):
"""
Transform a BedrockKBResponse to a VectorStoreSearchResponse
"""
retrieval_results: Optional[
List[BedrockKBRetrievalResult]
] = bedrock_kb_response.get("retrievalResults", None)
retrieval_results: Optional[List[BedrockKBRetrievalResult]] = (
bedrock_kb_response.get("retrievalResults", None)
)
vector_store_search_response: VectorStoreSearchResponse = (
VectorStoreSearchResponse(search_query=query, data=[])
)

View file

@ -77,6 +77,7 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM):
litellm_logging_obj: LiteLLMLoggingObj,
tools: Optional[List[Dict]] = None,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
"""
Retrieves the context from the Bedrock Knowledge Base and appends it to the messages.
@ -129,9 +130,9 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM):
)
)
litellm_logging_obj.model_call_details[
"vector_store_request_metadata"
] = vector_store_request_metadata
litellm_logging_obj.model_call_details["vector_store_request_metadata"] = (
vector_store_request_metadata
)
return model, messages, non_default_params
@ -143,9 +144,9 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM):
"""
Transform a BedrockKBResponse to a VectorStoreSearchResponse
"""
retrieval_results: Optional[
List[BedrockKBRetrievalResult]
] = bedrock_kb_response.get("retrievalResults", None)
retrieval_results: Optional[List[BedrockKBRetrievalResult]] = (
bedrock_kb_response.get("retrievalResults", None)
)
vector_store_search_response: VectorStoreSearchResponse = (
VectorStoreSearchResponse(search_query=query, data=[])
)

View file

@ -32,6 +32,7 @@ from litellm.integrations.opentelemetry import OpenTelemetry
from litellm.integrations.opik.opik import OpikLogger
from litellm.integrations.prometheus import PrometheusLogger
from litellm.integrations.s3_v2 import S3Logger
from litellm.integrations.sqs import SQSLogger
from litellm.integrations.vector_store_integrations.bedrock_vector_store import (
BedrockVectorStore,
)
@ -73,6 +74,7 @@ class CustomLoggerRegistry:
"bedrock_vector_store": BedrockVectorStore,
"deepeval": DeepEvalLogger,
"s3_v2": S3Logger,
"aws_sqs": SQSLogger,
"dynamic_rate_limiter": _PROXY_DynamicRateLimitHandler,
}

View file

@ -28,23 +28,31 @@ class ExceptionCheckers:
"""
Helper class for checking various error conditions in exception strings.
"""
@staticmethod
def is_error_str_rate_limit(error_str: str) -> bool:
"""
Check if an error string indicates a rate limit error.
Args:
error_str: The error string to check
Returns:
True if the error indicates a rate limit, False otherwise
"""
if not isinstance(error_str, str):
return False
return "429" in error_str or "rate limit" in error_str.lower()
if "429" in error_str or "rate limit" in error_str.lower():
return True
#######################################
# Mistral API returns this error string
#########################################
if "service tier capacity exceeded" in error_str.lower():
return True
return False
@staticmethod
def is_error_str_context_window_exceeded(error_str: str) -> bool:
@ -289,6 +297,7 @@ def exception_type( # type: ignore # noqa: PLR0915
or custom_llm_provider == "text-completion-openai"
or custom_llm_provider == "custom_openai"
or custom_llm_provider in litellm.openai_compatible_providers
or custom_llm_provider == "mistral"
):
# custom_llm_provider is openai, make it OpenAI
message = get_error_message(error_obj=original_exception)

View file

@ -624,6 +624,14 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915
or "https://api.galadriel.com/v1"
) # type: ignore
dynamic_api_key = api_key or get_secret_str("GALADRIEL_API_KEY")
elif custom_llm_provider == "github_copilot":
(
api_base,
dynamic_api_key,
custom_llm_provider,
) = litellm.GithubCopilotConfig()._get_openai_compatible_provider_info(
model, api_base, api_key, custom_llm_provider
)
elif custom_llm_provider == "novita":
api_base = (
api_base

View file

@ -18,6 +18,7 @@ def initialize_standard_callback_dynamic_params(
_supported_callback_params = (
StandardCallbackDynamicParams.__annotations__.keys()
)
for param in _supported_callback_params:
if param in kwargs:
_param_value = kwargs.pop(param)

View file

@ -44,6 +44,8 @@ from litellm.caching.caching_handler import LLMCachingHandler
from litellm.constants import (
DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT,
DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT,
SENTRY_DENYLIST,
SENTRY_PII_DENYLIST,
)
from litellm.cost_calculator import (
RealtimeAPITokenUsageProcessor,
@ -56,6 +58,7 @@ from litellm.integrations.custom_guardrail import CustomGuardrail
from litellm.integrations.custom_logger import CustomLogger
from litellm.integrations.deepeval.deepeval import DeepEvalLogger
from litellm.integrations.mlflow import MlflowLogger
from litellm.integrations.sqs import SQSLogger
from litellm.integrations.vector_store_integrations.bedrock_vector_store import (
BedrockVectorStore,
)
@ -432,6 +435,7 @@ class Logging(LiteLLMLoggingBaseClass):
checks if langfuse_secret_key, gcs_bucket_name in kwargs and sets the corresponding attributes in StandardCallbackDynamicParams
"""
return _initialize_standard_callback_dynamic_params(kwargs)
def initialize_standard_built_in_tools_params(
@ -553,6 +557,7 @@ class Logging(LiteLLMLoggingBaseClass):
prompt_variables: Optional[dict],
prompt_management_logger: Optional[CustomLogger] = None,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
custom_logger = (
prompt_management_logger
@ -574,6 +579,7 @@ class Logging(LiteLLMLoggingBaseClass):
prompt_variables=prompt_variables,
dynamic_callback_params=self.standard_callback_dynamic_params,
prompt_label=prompt_label,
prompt_version=prompt_version,
)
self.messages = messages
return model, messages, non_default_params
@ -588,6 +594,7 @@ class Logging(LiteLLMLoggingBaseClass):
prompt_management_logger: Optional[CustomLogger] = None,
tools: Optional[List[Dict]] = None,
prompt_label: Optional[str] = None,
prompt_version: Optional[int] = None,
) -> Tuple[str, List[AllMessageValues], dict]:
custom_logger = (
prompt_management_logger
@ -611,6 +618,7 @@ class Logging(LiteLLMLoggingBaseClass):
litellm_logging_obj=self,
tools=tools,
prompt_label=prompt_label,
prompt_version=prompt_version,
)
self.messages = messages
return model, messages, non_default_params
@ -2903,31 +2911,37 @@ def _get_masked_values(
]
return {
k: (
(
v[: unmasked_length // 2]
+ "*" * number_of_asterisks
+ v[-unmasked_length // 2 :]
)
if (
isinstance(v, str)
and len(v) > unmasked_length
and number_of_asterisks is not None
# If ignore_sensitive_values is True, or if this key doesn't contain sensitive keywords, return original value
v
if ignore_sensitive_values
or not any(
sensitive_keyword in k.lower()
for sensitive_keyword in sensitive_keywords
)
else (
# Apply masking to sensitive keys
(
v[: unmasked_length // 2]
+ "*" * (len(v) - unmasked_length)
+ "*" * number_of_asterisks
+ v[-unmasked_length // 2 :]
)
if (isinstance(v, str) and len(v) > unmasked_length)
else "*****"
if (
isinstance(v, str)
and len(v) > unmasked_length
and number_of_asterisks is not None
)
else (
(
v[: unmasked_length // 2]
+ "*" * (len(v) - unmasked_length)
+ v[-unmasked_length // 2 :]
)
if (isinstance(v, str) and len(v) > unmasked_length)
else ("*****" if isinstance(v, str) else v)
)
)
)
for k, v in sensitive_object.items()
if not ignore_sensitive_values
or not any(
sensitive_keyword in k.lower() for sensitive_keyword in sensitive_keywords
)
}
@ -2948,6 +2962,8 @@ def set_callbacks(callback_list, function_id=None): # noqa: PLR0915
[sys.executable, "-m", "pip", "install", "sentry_sdk"]
)
import sentry_sdk
from sentry_sdk.scrubber import EventScrubber
sentry_sdk_instance = sentry_sdk
sentry_trace_rate = (
os.environ.get("SENTRY_API_TRACE_RATE")
@ -2965,6 +2981,10 @@ def set_callbacks(callback_list, function_id=None): # noqa: PLR0915
sample_rate=float(
sentry_sample_rate if sentry_sample_rate else 1.0
),
send_default_pii=False, # Prevent sending Personal Identifiable Information
event_scrubber=EventScrubber(
denylist=SENTRY_DENYLIST, pii_denylist=SENTRY_PII_DENYLIST
),
)
capture_exception = sentry_sdk_instance.capture_exception
add_breadcrumb = sentry_sdk_instance.add_breadcrumb
@ -3021,6 +3041,7 @@ def set_callbacks(callback_list, function_id=None): # noqa: PLR0915
s3Logger = S3Logger()
elif callback == "wandb":
from litellm.integrations.weights_biases import WeightsBiasesLogger
weightsBiasesLogger = WeightsBiasesLogger()
elif callback == "logfire":
logfireLogger = LogfireLogger()
@ -3075,6 +3096,7 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
return _openmeter_logger # type: ignore
elif logging_integration == "braintrust":
from litellm.integrations.braintrust_logging import BraintrustLogger
for callback in _in_memory_loggers:
if isinstance(callback, BraintrustLogger):
return callback # type: ignore
@ -3142,6 +3164,14 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
_s3_v2_logger = S3V2Logger()
_in_memory_loggers.append(_s3_v2_logger)
return _s3_v2_logger # type: ignore
elif logging_integration == "aws_sqs":
for callback in _in_memory_loggers:
if isinstance(callback, SQSLogger):
return callback # type: ignore
_aws_sqs_logger = SQSLogger()
_in_memory_loggers.append(_aws_sqs_logger)
return _aws_sqs_logger # type: ignore
elif logging_integration == "azure_storage":
for callback in _in_memory_loggers:
if isinstance(callback, AzureBlobStorageLogger):
@ -3433,6 +3463,7 @@ def get_custom_logger_compatible_class( # noqa: PLR0915
return callback
elif logging_integration == "braintrust":
from litellm.integrations.braintrust_logging import BraintrustLogger
for callback in _in_memory_loggers:
if isinstance(callback, BraintrustLogger):
return callback
@ -3476,6 +3507,13 @@ def get_custom_logger_compatible_class( # noqa: PLR0915
for callback in _in_memory_loggers:
if isinstance(callback, S3V2Logger):
return callback
elif logging_integration == "aws_sqs":
for callback in _in_memory_loggers:
if isinstance(callback, SQSLogger):
return callback
_aws_sqs_logger = SQSLogger()
_in_memory_loggers.append(_aws_sqs_logger)
return _aws_sqs_logger # type: ignore
elif logging_integration == "azure_storage":
for callback in _in_memory_loggers:
if isinstance(callback, AzureBlobStorageLogger):

View file

@ -114,8 +114,8 @@ def _get_token_base_cost(model_info: ModelInfo, usage: Usage) -> Tuple[float, fl
If input_tokens > threshold and `input_cost_per_token_above_[x]k_tokens` or `input_cost_per_token_above_[x]_tokens` is set,
then we use the corresponding threshold cost.
"""
prompt_base_cost = model_info["input_cost_per_token"]
completion_base_cost = model_info["output_cost_per_token"]
prompt_base_cost = cast(float, _get_cost_per_unit(model_info, "input_cost_per_token"))
completion_base_cost = cast(float, _get_cost_per_unit(model_info, "output_cost_per_token"))
## CHECK IF ABOVE THRESHOLD
threshold: Optional[float] = None
@ -128,17 +128,13 @@ def _get_token_base_cost(model_info: ModelInfo, usage: Usage) -> Tuple[float, fl
1000 if "k" in threshold_str else 1
)
if usage.prompt_tokens > threshold:
prompt_base_cost = cast(
float,
model_info.get(key, prompt_base_cost),
)
completion_base_cost = cast(
float,
model_info.get(
f"output_cost_per_token_above_{threshold_str}_tokens",
completion_base_cost,
),
)
prompt_base_cost = cast(float, _get_cost_per_unit(model_info, key, prompt_base_cost))
completion_base_cost = cast(float, _get_cost_per_unit(
model_info,
f"output_cost_per_token_above_{threshold_str}_tokens",
completion_base_cost,
))
break
except (IndexError, ValueError):
continue
@ -162,7 +158,7 @@ def calculate_cost_component(
Returns:
float: The calculated cost
"""
cost_per_unit = model_info.get(cost_key)
cost_per_unit = _get_cost_per_unit(model_info, cost_key)
if (
cost_per_unit is not None
and isinstance(cost_per_unit, float)
@ -173,6 +169,24 @@ def calculate_cost_component(
return 0.0
def _get_cost_per_unit(model_info: ModelInfo, cost_key: str, default_value: Optional[float] = 0.0) -> Optional[float]:
# Sometimes the cost per unit is a string (e.g.: If a value like "3e-7" was read from the config.yaml)
cost_per_unit = model_info.get(cost_key)
if isinstance(cost_per_unit, float):
return cost_per_unit
if isinstance(cost_per_unit, int):
return float(cost_per_unit)
if isinstance(cost_per_unit, str):
try:
return float(cost_per_unit)
except ValueError:
verbose_logger.exception(
f"litellm.litellm_core_utils.llm_cost_calc.utils.py::calculate_cost_per_component(): Exception occured - {cost_per_unit}\nDefaulting to 0.0"
)
return default_value
def generic_cost_per_token(
model: str, usage: Usage, custom_llm_provider: str
) -> Tuple[float, float]:
@ -316,13 +330,8 @@ def generic_cost_per_token(
## TEXT COST
completion_cost = float(text_tokens) * completion_base_cost
_output_cost_per_audio_token: Optional[float] = model_info.get(
"output_cost_per_audio_token"
)
_output_cost_per_reasoning_token: Optional[float] = model_info.get(
"output_cost_per_reasoning_token"
)
_output_cost_per_audio_token = _get_cost_per_unit(model_info, "output_cost_per_audio_token", None)
_output_cost_per_reasoning_token = _get_cost_per_unit(model_info, "output_cost_per_reasoning_token", None)
## AUDIO COST
if not is_text_tokens_total and audio_tokens is not None and audio_tokens > 0:

View file

@ -2631,7 +2631,7 @@ def _convert_to_bedrock_tool_call_invoke(
id = tool["id"]
name = tool["function"].get("name", "")
arguments = tool["function"].get("arguments", "")
arguments_dict = json.loads(arguments)
arguments_dict = json.loads(arguments) if arguments else {}
bedrock_tool = BedrockToolUseBlock(
input=arguments_dict, name=name, toolUseId=id
)

View file

@ -53,29 +53,45 @@ def perform_redaction(model_call_details: dict, result):
and "complete_streaming_response" in model_call_details
):
_streaming_response = model_call_details["complete_streaming_response"]
for choice in _streaming_response.choices:
if isinstance(choice, litellm.Choices):
choice.message.content = "redacted-by-litellm"
elif isinstance(choice, litellm.utils.StreamingChoices):
choice.delta.content = "redacted-by-litellm"
# Redact result
if result is not None and isinstance(result, litellm.ModelResponse):
_result = copy.deepcopy(result)
if hasattr(_result, "choices") and _result.choices is not None:
for choice in _result.choices:
if hasattr(_streaming_response, "choices"):
for choice in _streaming_response.choices:
if isinstance(choice, litellm.Choices):
choice.message.content = "redacted-by-litellm"
elif isinstance(choice, litellm.utils.StreamingChoices):
choice.delta.content = "redacted-by-litellm"
return _result
if result is not None and isinstance(result, litellm.EmbeddingResponse):
elif hasattr(_streaming_response, "output"):
# Handle ResponsesAPIResponse format
for output_item in _streaming_response.output:
if hasattr(output_item, "content") and isinstance(
output_item.content, list
):
for content_part in output_item.content:
if hasattr(content_part, "text"):
content_part.text = "redacted-by-litellm"
# Redact result
if result is not None:
_result = copy.deepcopy(result)
if hasattr(_result, "data") and _result.data is not None:
_result.data = []
if isinstance(_result, litellm.ModelResponse):
if hasattr(_result, "choices") and _result.choices is not None:
for choice in _result.choices:
if isinstance(choice, litellm.Choices):
choice.message.content = "redacted-by-litellm"
elif isinstance(choice, litellm.utils.StreamingChoices):
choice.delta.content = "redacted-by-litellm"
elif isinstance(_result, litellm.ResponsesAPIResponse):
if hasattr(_result, "output"):
for output_item in _result.output:
if hasattr(output_item, "content") and isinstance(output_item.content, list):
for content_part in output_item.content:
if hasattr(content_part, "text"):
content_part.text = "redacted-by-litellm"
elif isinstance(_result, litellm.EmbeddingResponse):
if hasattr(_result, "data") and _result.data is not None:
_result.data = []
else:
return {"text": "redacted-by-litellm"}
return _result
else:
return {"text": "redacted-by-litellm"}
def should_redact_message_logging(model_call_details: dict) -> bool:
@ -140,9 +156,9 @@ def _get_turn_off_message_logging_from_dynamic_params(
handles boolean and string values of `turn_off_message_logging`
"""
standard_callback_dynamic_params: Optional[
StandardCallbackDynamicParams
] = model_call_details.get("standard_callback_dynamic_params", None)
standard_callback_dynamic_params: Optional[StandardCallbackDynamicParams] = (
model_call_details.get("standard_callback_dynamic_params", None)
)
if standard_callback_dynamic_params:
_turn_off_message_logging = standard_callback_dynamic_params.get(
"turn_off_message_logging"

View file

@ -1,6 +1,6 @@
import base64
import time
from typing import Any, Dict, List, Optional, Union, cast
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union, cast
from litellm.types.llms.openai import (
ChatCompletionAssistantContentValue,
@ -16,11 +16,16 @@ from litellm.types.utils import (
FunctionCall,
ModelResponse,
ModelResponseStream,
PromptTokensDetails,
PromptTokensDetailsWrapper,
Usage,
)
from litellm.utils import print_verbose, token_counter
if TYPE_CHECKING:
from litellm.types.litellm_core_utils.streaming_chunk_builder_utils import (
UsagePerChunk,
)
class ChunkProcessor:
def __init__(self, chunks: List, messages: Optional[list] = None):
@ -107,9 +112,9 @@ class ChunkProcessor:
self, tool_call_chunks: List[Dict[str, Any]]
) -> List[ChatCompletionMessageToolCall]:
tool_calls_list: List[ChatCompletionMessageToolCall] = []
tool_call_map: Dict[
int, Dict[str, Any]
] = {} # Map to store tool calls by index
tool_call_map: Dict[int, Dict[str, Any]] = (
{}
) # Map to store tool calls by index
for chunk in tool_call_chunks:
choices = chunk["choices"]
@ -256,7 +261,7 @@ class ChunkProcessor:
cache_creation_input_tokens: Optional[int] = None
cache_read_input_tokens: Optional[int] = None
completion_tokens_details: Optional[CompletionTokensDetails] = None
prompt_tokens_details: Optional[PromptTokensDetails] = None
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
if "prompt_tokens" in usage_chunk:
prompt_tokens = usage_chunk.get("prompt_tokens", 0) or 0
@ -277,10 +282,12 @@ class ChunkProcessor:
completion_tokens_details = usage_chunk.completion_tokens_details
if hasattr(usage_chunk, "prompt_tokens_details"):
if isinstance(usage_chunk.prompt_tokens_details, dict):
prompt_tokens_details = PromptTokensDetails(
prompt_tokens_details = PromptTokensDetailsWrapper(
**usage_chunk.prompt_tokens_details
)
elif isinstance(usage_chunk.prompt_tokens_details, PromptTokensDetails):
elif isinstance(
usage_chunk.prompt_tokens_details, PromptTokensDetailsWrapper
):
prompt_tokens_details = usage_chunk.prompt_tokens_details
return {
@ -306,26 +313,24 @@ class ChunkProcessor:
return reasoning_tokens
def calculate_usage(
def _calculate_usage_per_chunk(
self,
chunks: List[Union[Dict[str, Any], ModelResponse]],
model: str,
completion_output: str,
messages: Optional[List] = None,
reasoning_tokens: Optional[int] = None,
) -> Usage:
"""
Calculate usage for the given chunks.
"""
returned_usage = Usage()
) -> "UsagePerChunk":
from litellm.types.litellm_core_utils.streaming_chunk_builder_utils import (
UsagePerChunk,
)
# # Update usage information if needed
prompt_tokens = 0
completion_tokens = 0
## anthropic prompt caching information ##
cache_creation_input_tokens: Optional[int] = None
cache_read_input_tokens: Optional[int] = None
web_search_requests: Optional[int] = None
completion_tokens_details: Optional[CompletionTokensDetails] = None
prompt_tokens_details: Optional[PromptTokensDetails] = None
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
for chunk in chunks:
usage_chunk: Optional[Usage] = None
if "usage" in chunk:
@ -366,7 +371,67 @@ class ChunkProcessor:
completion_tokens_details = usage_chunk_dict[
"completion_tokens_details"
]
if (
usage_chunk_dict["prompt_tokens_details"] is not None
and getattr(
usage_chunk_dict["prompt_tokens_details"],
"web_search_requests",
None,
)
is not None
):
web_search_requests = getattr(
usage_chunk_dict["prompt_tokens_details"],
"web_search_requests",
)
prompt_tokens_details = usage_chunk_dict["prompt_tokens_details"]
return UsagePerChunk(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
cache_creation_input_tokens=cache_creation_input_tokens,
cache_read_input_tokens=cache_read_input_tokens,
web_search_requests=web_search_requests,
completion_tokens_details=completion_tokens_details,
prompt_tokens_details=prompt_tokens_details,
)
def calculate_usage(
self,
chunks: List[Union[Dict[str, Any], ModelResponse]],
model: str,
completion_output: str,
messages: Optional[List] = None,
reasoning_tokens: Optional[int] = None,
) -> Usage:
"""
Calculate usage for the given chunks.
"""
returned_usage = Usage()
# # Update usage information if needed
calculated_usage_per_chunk = self._calculate_usage_per_chunk(chunks=chunks)
prompt_tokens = calculated_usage_per_chunk["prompt_tokens"]
completion_tokens = calculated_usage_per_chunk["completion_tokens"]
## anthropic prompt caching information ##
cache_creation_input_tokens: Optional[int] = calculated_usage_per_chunk[
"cache_creation_input_tokens"
]
cache_read_input_tokens: Optional[int] = calculated_usage_per_chunk[
"cache_read_input_tokens"
]
web_search_requests: Optional[int] = calculated_usage_per_chunk[
"web_search_requests"
]
completion_tokens_details: Optional[CompletionTokensDetails] = (
calculated_usage_per_chunk["completion_tokens_details"]
)
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = (
calculated_usage_per_chunk["prompt_tokens_details"]
)
try:
returned_usage.prompt_tokens = prompt_tokens or token_counter(
model=model, messages=messages
@ -415,6 +480,20 @@ class ChunkProcessor:
if prompt_tokens_details is not None:
returned_usage.prompt_tokens_details = prompt_tokens_details
if web_search_requests is not None:
if returned_usage.prompt_tokens_details is None:
returned_usage.prompt_tokens_details = PromptTokensDetailsWrapper(
web_search_requests=web_search_requests
)
else:
returned_usage.prompt_tokens_details.web_search_requests = (
web_search_requests
)
# Return a new usage object with the new values
returned_usage = Usage(**returned_usage.model_dump())
return returned_usage

View file

@ -620,7 +620,6 @@ class CustomStreamWrapper:
args = {
"model": _model,
"stream_options": self.stream_options,
**chunk_dict,
}
@ -758,6 +757,7 @@ class CustomStreamWrapper:
is_chunk_non_empty = self.is_chunk_non_empty(
completion_obj, model_response, response_obj
)
if (
is_chunk_non_empty
): # cannot set content of an OpenAI Object to be an empty string
@ -1203,6 +1203,9 @@ class CustomStreamWrapper:
if response_obj is None:
return
completion_obj["content"] = response_obj["text"]
self.intermittent_finish_reason = response_obj.get(
"finish_reason", None
)
if response_obj["is_finished"]:
if response_obj["finish_reason"] == "error":
raise Exception(
@ -1561,6 +1564,7 @@ class CustomStreamWrapper:
complete_streaming_response = litellm.stream_chunk_builder(
chunks=self.chunks, messages=self.messages
)
response = self.model_response_creator()
if complete_streaming_response is not None:
setattr(

View file

@ -515,9 +515,7 @@ class ModelResponseIterator:
usage_object=cast(dict, anthropic_usage_chunk), reasoning_content=None
)
def _content_block_delta_helper(
self, chunk: dict
) -> Tuple[
def _content_block_delta_helper(self, chunk: dict) -> Tuple[
str,
Optional[ChatCompletionToolCallChunk],
List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]],
@ -600,6 +598,19 @@ class ModelResponseIterator:
return thinking_blocks, provider_specific_fields
def get_content_block_start(self, chunk: dict) -> ContentBlockStart:
from litellm.types.llms.anthropic import (
ContentBlockStartText,
ContentBlockStartToolUse,
)
if chunk.get("content_block", {}).get("type") == "tool_use":
content_block_start = ContentBlockStartToolUse(**chunk) # type: ignore
else:
content_block_start = ContentBlockStartText(**chunk) # type: ignore
return content_block_start
def chunk_parser(self, chunk: dict) -> ModelResponseStream:
try:
type_chunk = chunk.get("type", "") or ""
@ -639,7 +650,8 @@ class ModelResponseIterator:
event: content_block_start
data: {"type":"content_block_start","index":1,"content_block":{"type":"tool_use","id":"toolu_01T1x1fJ34qAmk2tNTrN7Up6","name":"get_weather","input":{}}}
"""
content_block_start = ContentBlockStart(**chunk) # type: ignore
content_block_start = self.get_content_block_start(chunk=chunk)
self.content_blocks = [] # reset content blocks when new block starts
if content_block_start["content_block"]["type"] == "text":
text = content_block_start["content_block"]["text"]

View file

@ -709,7 +709,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
and isinstance(_litellm_metadata, dict)
and "user_id" in _litellm_metadata
and _litellm_metadata["user_id"] is not None
and not _valid_user_id(_litellm_metadata["user_id"])
and _valid_user_id(_litellm_metadata["user_id"])
):
optional_params["metadata"] = {"user_id": _litellm_metadata["user_id"]}

View file

@ -88,6 +88,9 @@ class LiteLLMMessagesToCompletionTransformationHandler:
if stream:
completion_kwargs["stream"] = stream
completion_kwargs["stream_options"] = {
"include_usage": True,
}
excluded_keys = {"anthropic_messages"}
extra_kwargs = extra_kwargs or {}
@ -100,6 +103,9 @@ class LiteLLMMessagesToCompletionTransformationHandler:
from litellm.types.utils import CallTypes
setattr(value, "call_type", CallTypes.completion.value)
setattr(
value, "stream_options", completion_kwargs.get("stream_options")
)
if (
key not in excluded_keys
and key not in completion_kwargs

View file

@ -3,12 +3,16 @@
import json
import traceback
import uuid
from typing import Any, AsyncIterator, Iterator, Optional
from collections import deque
from typing import TYPE_CHECKING, Any, AsyncIterator, Iterator, Literal, Optional
from litellm import verbose_logger
from litellm.types.llms.anthropic import UsageDelta
from litellm.types.utils import AdapterCompletionStreamWrapper
if TYPE_CHECKING:
from litellm.types.utils import ModelResponseStream
class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
"""
@ -17,6 +21,13 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
- finish_reason must map exactly to anthropic reason, else anthropic client won't be able to parse it.
"""
from litellm.types.llms.anthropic import (
ContentBlockContentBlockDict,
ContentBlockStart,
ContentBlockStartText,
TextBlock,
)
def __init__(self, completion_stream: Any, model: str):
super().__init__(completion_stream)
self.model = model
@ -24,8 +35,17 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
sent_first_chunk: bool = False
sent_content_block_start: bool = False
sent_content_block_finish: bool = False
current_content_block_type: Literal["text", "tool_use"] = "text"
sent_last_message: bool = False
holding_chunk: Optional[Any] = None
holding_stop_reason_chunk: Optional[Any] = None
current_content_block_index: int = 0
current_content_block_start: ContentBlockContentBlockDict = TextBlock(
type="text",
text="",
)
pending_new_content_block: bool = False
chunk_queue: deque = deque() # Queue for buffering multiple chunks
def __next__(self):
from .transformation import LiteLLMAnthropicMessagesAdapter
@ -50,17 +70,47 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
self.sent_content_block_start = True
return {
"type": "content_block_start",
"index": 0,
"index": self.current_content_block_index,
"content_block": {"type": "text", "text": ""},
}
# Handle pending new content block start
if self.pending_new_content_block:
self.pending_new_content_block = False
self.sent_content_block_finish = False # Reset for new block
return {
"type": "content_block_start",
"index": self.current_content_block_index,
"content_block": self.current_content_block_start,
}
for chunk in self.completion_stream:
if chunk == "None" or chunk is None:
raise Exception
should_start_new_block = self._should_start_new_content_block(chunk)
if should_start_new_block:
self._increment_content_block_index()
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
response=chunk
response=chunk,
current_content_block_index=self.current_content_block_index,
)
# Check if we need to start a new content block
# This is where you'd add your logic to detect when a new content block should start
# For example, if the chunk indicates a tool call or different content type
if should_start_new_block and not self.sent_content_block_finish:
# End current content block and prepare for new one
self.holding_chunk = processed_chunk
self.sent_content_block_finish = True
self.pending_new_content_block = True
return {
"type": "content_block_stop",
"index": max(self.current_content_block_index - 1, 0),
}
if (
processed_chunk["type"] == "message_delta"
and self.sent_content_block_finish is False
@ -69,7 +119,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
self.sent_content_block_finish = True
return {
"type": "content_block_stop",
"index": 0,
"index": self.current_content_block_index,
}
elif self.holding_chunk is not None:
return_chunk = self.holding_chunk
@ -96,64 +146,167 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
)
raise StopAsyncIteration
async def __anext__(self):
async def __anext__(self): # noqa: PLR0915
from .transformation import LiteLLMAnthropicMessagesAdapter
try:
# Always return queued chunks first
if self.chunk_queue:
return self.chunk_queue.popleft()
# Queue initial chunks if not sent yet
if self.sent_first_chunk is False:
self.sent_first_chunk = True
return {
"type": "message_start",
"message": {
"id": "msg_{}".format(uuid.uuid4()),
"type": "message",
"role": "assistant",
"content": [],
"model": self.model,
"stop_reason": None,
"stop_sequence": None,
"usage": UsageDelta(input_tokens=0, output_tokens=0),
},
}
self.chunk_queue.append(
{
"type": "message_start",
"message": {
"id": "msg_{}".format(uuid.uuid4()),
"type": "message",
"role": "assistant",
"content": [],
"model": self.model,
"stop_reason": None,
"stop_sequence": None,
"usage": UsageDelta(input_tokens=0, output_tokens=0),
},
}
)
return self.chunk_queue.popleft()
if self.sent_content_block_start is False:
self.sent_content_block_start = True
return {
"type": "content_block_start",
"index": 0,
"content_block": {"type": "text", "text": ""},
}
self.chunk_queue.append(
{
"type": "content_block_start",
"index": self.current_content_block_index,
"content_block": {"type": "text", "text": ""},
}
)
return self.chunk_queue.popleft()
async for chunk in self.completion_stream:
if chunk == "None" or chunk is None:
raise Exception
# Check if we need to start a new content block
should_start_new_block = self._should_start_new_content_block(chunk)
if should_start_new_block:
self._increment_content_block_index()
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
response=chunk
response=chunk,
current_content_block_index=self.current_content_block_index,
)
# Check if this is a usage chunk and we have a held stop_reason chunk
if (
self.holding_stop_reason_chunk is not None
and getattr(chunk, "usage", None) is not None
):
# Merge usage into the held stop_reason chunk
merged_chunk = self.holding_stop_reason_chunk.copy()
if "delta" not in merged_chunk:
merged_chunk["delta"] = {}
# Add usage to the held chunk
merged_chunk["usage"] = {
"input_tokens": chunk.usage.prompt_tokens or 0,
"output_tokens": chunk.usage.completion_tokens or 0,
}
# Queue the merged chunk and reset
self.chunk_queue.append(merged_chunk)
self.holding_stop_reason_chunk = None
return self.chunk_queue.popleft()
# Check if this processed chunk has a stop_reason - hold it for next chunk
if should_start_new_block and not self.sent_content_block_finish:
# Queue the sequence: content_block_stop -> content_block_start -> current_chunk
# 1. Stop current content block
self.chunk_queue.append(
{
"type": "content_block_stop",
"index": max(self.current_content_block_index - 1, 0),
}
)
# 2. Start new content block
self.chunk_queue.append(
{
"type": "content_block_start",
"index": self.current_content_block_index,
"content_block": self.current_content_block_start,
}
)
# 3. Queue the current chunk (don't lose it!)
self.chunk_queue.append(processed_chunk)
# Reset state for new block
self.sent_content_block_finish = False
# Return the first queued item
return self.chunk_queue.popleft()
if (
processed_chunk["type"] == "message_delta"
and self.sent_content_block_finish is False
):
self.holding_chunk = processed_chunk
# Queue both the content_block_stop and the holding chunk
self.chunk_queue.append(
{
"type": "content_block_stop",
"index": self.current_content_block_index,
}
)
self.sent_content_block_finish = True
return {
"type": "content_block_stop",
"index": 0,
}
if processed_chunk.get("delta", {}).get("stop_reason") is not None:
self.holding_stop_reason_chunk = processed_chunk
else:
self.chunk_queue.append(processed_chunk)
return self.chunk_queue.popleft()
elif self.holding_chunk is not None:
return_chunk = self.holding_chunk
self.holding_chunk = processed_chunk
return return_chunk
# Queue both chunks
self.chunk_queue.append(self.holding_chunk)
self.chunk_queue.append(processed_chunk)
self.holding_chunk = None
return self.chunk_queue.popleft()
else:
return processed_chunk
# Queue the current chunk
self.chunk_queue.append(processed_chunk)
return self.chunk_queue.popleft()
# Handle any remaining held chunks after stream ends
if self.holding_stop_reason_chunk is not None:
self.chunk_queue.append(self.holding_stop_reason_chunk)
self.holding_stop_reason_chunk = None
if self.holding_chunk is not None:
return_chunk = self.holding_chunk
self.chunk_queue.append(self.holding_chunk)
self.holding_chunk = None
return return_chunk
if self.sent_last_message is False:
if not self.sent_last_message:
self.sent_last_message = True
return {"type": "message_stop"}
self.chunk_queue.append({"type": "message_stop"})
# Return queued items if any
if self.chunk_queue:
return self.chunk_queue.popleft()
raise StopIteration
except StopIteration:
if self.sent_last_message is False:
# Handle any remaining queued chunks before stopping
if self.chunk_queue:
return self.chunk_queue.popleft()
# Handle any held stop_reason chunk
if self.holding_stop_reason_chunk is not None:
return self.holding_stop_reason_chunk
if not self.sent_last_message:
self.sent_last_message = True
return {"type": "message_stop"}
raise StopAsyncIteration
@ -187,3 +340,37 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
else:
# For non-dict chunks, forward the original value unchanged
yield chunk
def _increment_content_block_index(self):
self.current_content_block_index += 1
def _should_start_new_content_block(self, chunk: "ModelResponseStream") -> bool:
"""
Determine if we should start a new content block based on the processed chunk.
Override this method with your specific logic for detecting new content blocks.
Examples of when you might want to start a new content block:
- Switching from text to tool calls
- Different content types in the response
- Specific markers in the content
"""
from .transformation import LiteLLMAnthropicMessagesAdapter
# Example logic - customize based on your needs:
# If chunk indicates a tool call
if chunk.choices[0].finish_reason is not None:
return False
(
block_type,
content_block_start,
) = LiteLLMAnthropicMessagesAdapter()._translate_streaming_openai_chunk_to_anthropic_content_block(
choices=chunk.choices # type: ignore
)
if block_type != self.current_content_block_type:
self.current_content_block_type = block_type
self.current_content_block_start = content_block_start
return True
return False

View file

@ -1,5 +1,15 @@
import json
from typing import Any, AsyncIterator, List, Literal, Optional, Tuple, Union, cast
from typing import (
TYPE_CHECKING,
Any,
AsyncIterator,
List,
Literal,
Optional,
Tuple,
Union,
cast,
)
from openai.types.chat.chat_completion_chunk import Choice as OpenAIStreamingChoice
@ -45,6 +55,9 @@ from litellm.types.utils import Choices, ModelResponse, Usage
from .streaming_iterator import AnthropicStreamWrapper
if TYPE_CHECKING:
from litellm.types.llms.anthropic import ContentBlockContentBlockDict
class AnthropicAdapter:
def __init__(self) -> None:
@ -439,12 +452,40 @@ class LiteLLMAnthropicMessagesAdapter:
return translated_obj
def _translate_streaming_openai_chunk_to_anthropic_content_block(
self, choices: List[OpenAIStreamingChoice]
) -> Tuple[
Literal["text", "tool_use"],
"ContentBlockContentBlockDict",
]:
import uuid
from litellm.types.llms.anthropic import TextBlock, ToolUseBlock
for choice in choices:
if choice.delta.content is not None and len(choice.delta.content) > 0:
return "text", TextBlock(type="text", text="")
elif (
choice.delta.tool_calls is not None
and len(choice.delta.tool_calls) > 0
and choice.delta.tool_calls[0].function is not None
):
return "tool_use", ToolUseBlock(
type="tool_use",
id=choice.delta.tool_calls[0].id or str(uuid.uuid4()),
name=choice.delta.tool_calls[0].function.name or "",
input={},
)
return "text", TextBlock(type="text", text="")
def _translate_streaming_openai_chunk_to_anthropic(
self, choices: List[OpenAIStreamingChoice]
) -> Tuple[
Literal["text_delta", "input_json_delta"],
Union[ContentTextBlockDelta, ContentJsonBlockDelta],
]:
text: str = ""
partial_json: Optional[str] = None
for choice in choices:
@ -467,7 +508,7 @@ class LiteLLMAnthropicMessagesAdapter:
return "text_delta", ContentTextBlockDelta(type="text_delta", text=text)
def translate_streaming_openai_response_to_anthropic(
self, response: ModelResponse
self, response: ModelResponse, current_content_block_index: int
) -> Union[ContentBlockDelta, MessageBlockDelta]:
## base case - final chunk w/ finish reason
if response.choices[0].finish_reason is not None:
@ -503,6 +544,6 @@ class LiteLLMAnthropicMessagesAdapter:
)
return ContentBlockDelta(
type="content_block_delta",
index=response.choices[0].index,
index=current_content_block_index,
delta=content_block_delta,
)

View file

@ -17,6 +17,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from litellm.types.llms.anthropic_messages.anthropic_request import AnthropicMetadata
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
@ -91,6 +92,16 @@ async def anthropic_messages(
response = init_response
return response
def validate_anthropic_api_metadata(metadata: Optional[Dict] = None) -> Optional[Dict]:
"""
Validate Anthropic API metadata - This is done to ensure only allowed `metadata` fields are passed to Anthropic API
If there are any litellm specific metadata fields, use `litellm_metadata` key to pass them.
"""
if metadata is None:
return None
anthropic_metadata_obj = AnthropicMetadata(**metadata)
return anthropic_metadata_obj.model_dump(exclude_none=True)
def anthropic_messages_handler(
max_tokens: int,
@ -120,6 +131,7 @@ def anthropic_messages_handler(
Makes Anthropic `/v1/messages` API calls In the Anthropic API Spec
"""
from litellm.types.utils import LlmProviders
metadata = validate_anthropic_api_metadata(metadata)
local_vars = locals()
is_async = kwargs.pop("is_async", False)

View file

@ -60,12 +60,17 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
api_key: Optional[str] = None,
api_base: Optional[str] = None,
) -> Tuple[dict, Optional[str]]:
import os
if api_key is None:
api_key = os.getenv("ANTHROPIC_API_KEY")
if "x-api-key" not in headers and api_key:
headers["x-api-key"] = api_key
if "anthropic-version" not in headers:
headers["anthropic-version"] = DEFAULT_ANTHROPIC_API_VERSION
if "content-type" not in headers:
headers["content-type"] = "application/json"
return headers, api_base
def transform_anthropic_messages_request(

Some files were not shown because too many files have changed in this diff Show more