Merge branch 'main' into litellm_codex_anthropic_model_issue

This commit is contained in:
Sameer Kankute 2025-11-10 22:21:24 +05:30 • committed by GitHub
commit 2ba4f10bae
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
611 changed files with 31095 additions and 6183 deletions

View file

@ -532,7 +532,7 @@ jobs:
command: |
pwd
ls
python -m pytest -vv tests/router_unit_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5
python -m pytest -vv tests/router_unit_tests --cov=litellm --cov-report=xml -x -s --junitxml=test-results/junit.xml --durations=5
no_output_timeout: 120m
- run:
name: Rename the coverage files
@ -1164,7 +1164,7 @@ jobs:
command: |
pwd
ls
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
no_output_timeout: 120m
- run:
name: Rename the coverage files
@ -1396,7 +1396,7 @@ jobs:
command: |
pwd
ls
python -m pytest -vv tests/image_gen_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5
python -m pytest -vv tests/image_gen_tests --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5
no_output_timeout: 120m
- run:
name: Rename the coverage files

View file

@ -15,4 +15,5 @@ fastapi-sso==0.16.0
uvloop==0.21.0
mcp==1.10.1 # for MCP server
semantic_router==0.1.10 # for auto-routing with litellm
fastuuid==0.12.0
fastuuid==0.12.0
responses==0.25.7 # for proxy client tests

View file

@ -65,6 +65,10 @@ COPY --from=builder /wheels/ /wheels/
# Install the built wheel using pip; again using a wildcard if it's the only file
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
# Remove test files and keys from dependencies
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
find /usr/lib -type d -path "*/tornado/test" -delete
# Install semantic_router and aurelio-sdk using script
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh

View file

@ -0,0 +1,20 @@
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
key_management_system: "custom"
key_management_settings:
custom_secret_manager: my_secret_manager.InMemorySecretManager
store_virtual_keys: true
prefix_for_stored_virtual_keys: "litellm/"
access_mode: "read_and_write"
model_list:
- model_name: gpt-4
litellm_params:
model: openai/gpt-4
api_key: os.environ/OPENAI_API_KEY # Read from custom secret manager
- model_name: claude-3-5-sonnet
litellm_params:
model: anthropic/claude-3-5-sonnet-20241022
api_key: os.environ/ANTHROPIC_API_KEY # Read from custom secret manager

View file

@ -0,0 +1,79 @@
"""
Example custom secret manager for LiteLLM Proxy.
This is a simple in-memory secret manager for testing purposes.
In production, replace this with your actual secret management system.
"""
from typing import Optional, Union
import httpx
from litellm.integrations.custom_secret_manager import CustomSecretManager
class InMemorySecretManager(CustomSecretManager):
def __init__(self):
super().__init__(secret_manager_name="in_memory_secrets")
# Store your secrets in memory
print("INITIALIZING CUSTOM SECRET MANAGER IN MEMORY")
self.secrets = {}
print("CUSTOM SECRET MANAGER IN MEMORY INITIALIZED")
async def async_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""Read secret asynchronously"""
print("READING SECRET ASYNCHRONOUSLY")
print("SECRET NAME: %s", secret_name)
print("SECRET: %s", self.secrets.get(secret_name))
return self.secrets.get(secret_name)
def sync_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""Read secret synchronously"""
from litellm._logging import verbose_proxy_logger
verbose_proxy_logger.info(f"CUSTOM SECRET MANAGER: LOOKING FOR SECRET: {secret_name}")
value = self.secrets.get(secret_name)
verbose_proxy_logger.info(f"CUSTOM SECRET MANAGER: READ SECRET: {value}")
return value
async def async_write_secret(
self,
secret_name: str,
secret_value: str,
description: Optional[str] = None,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
tags: Optional[Union[dict, list]] = None,
) -> dict:
"""Write a secret to the in-memory store"""
self.secrets[secret_name] = secret_value
print("ALL SECRETS=%s", self.secrets)
return {
"status": "success",
"secret_name": secret_name,
"description": description,
}
async def async_delete_secret(
self,
secret_name: str,
recovery_window_in_days: Optional[int] = 7,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> dict:
"""Delete a secret from the in-memory store"""
if secret_name in self.secrets:
del self.secrets[secret_name]
return {"status": "deleted", "secret_name": secret_name}
return {"status": "not_found", "secret_name": secret_name}

View file

@ -57,6 +57,9 @@ USER root
# Install only runtime dependencies
RUN apt-get update && apt-get install -y --no-install-recommends \
libssl3 \
libatomic1 \
nodejs \
npm \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app

View file

@ -8,16 +8,36 @@ ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev
FROM $LITELLM_BUILD_IMAGE AS builder
WORKDIR /app
# Install build dependencies
# Install build dependencies including Node.js for UI build
USER root
RUN apk add --no-cache build-base bash \
RUN apk add --no-cache build-base bash nodejs npm \
&& pip install --no-cache-dir --upgrade pip build
# Copy project files
COPY . .
# Set LITELLM_NON_ROOT flag for build time
ENV LITELLM_NON_ROOT=true
# Build Admin UI
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
RUN mkdir -p /tmp/litellm_ui && \
cd ui/litellm-dashboard && \
if [ -f "../../enterprise/enterprise_ui/enterprise_colors.json" ]; then \
cp ../../enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
fi && \
npm install && \
npm run build && \
cp -r ./out/* /tmp/litellm_ui/ && \
cd /tmp/litellm_ui && \
for html_file in *.html; do \
if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
folder_name="${html_file%.html}" && \
mkdir -p "$folder_name" && \
mv "$html_file" "$folder_name/index.html"; \
fi; \
done && \
cd /app/ui/litellm-dashboard && \
rm -rf ./out
# Build package and wheel dependencies
RUN rm -rf dist/* && python -m build && \
@ -42,12 +62,17 @@ COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/dist/*.whl .
COPY --from=builder /wheels/ /wheels/
COPY --from=builder /tmp/litellm_ui /tmp/litellm_ui
# Install package from wheel and dependencies
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
&& rm -f *.whl \
&& rm -rf /wheels
# Remove test files and keys from dependencies
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
find /usr/lib -type d -path "*/tornado/test" -delete
# Install semantic_router and aurelio-sdk using script
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
@ -56,7 +81,6 @@ RUN pip uninstall jwt -y && \
pip uninstall PyJWT -y && \
pip install PyJWT==2.9.0 --no-cache-dir
# --- Prisma Handling for Non-Root User ---
# Set Prisma cache directories
ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
ENV NPM_CONFIG_CACHE=/.npm
@ -68,25 +92,20 @@ RUN pip install --no-cache-dir prisma && \
# Create directories and set permissions for non-root user
RUN mkdir -p /nonexistent /.npm && \
chown -R nobody:nogroup /app && \
chown -R nobody:nogroup /nonexistent /.npm && \
chown -R nobody:nogroup /app /tmp/litellm_ui /nonexistent /.npm && \
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
chown -R nobody:nogroup $PRISMA_PATH && \
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH
# --- OpenShift Compatibility: Apply Red Hat recommended pattern ---
# Get paths for directories that need write access at runtime
# OpenShift compatibility
RUN PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
# Set group ownership to 0 (root group) for OpenShift compatibility && \
chgrp -R 0 $PRISMA_PATH && \
chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui && \
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
# Mirror owner permissions to group (g=u) as recommended by Red Hat && \
chmod -R g=u $PRISMA_PATH && \
chmod -R g=u $PRISMA_PATH /tmp/litellm_ui && \
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
# Ensure directories are writable by group && \
chmod -R g+w $PRISMA_PATH && \
chmod -R g+w $PRISMA_PATH /tmp/litellm_ui && \
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true
# Switch to non-root user
@ -94,14 +113,14 @@ USER nobody
# Set HOME for prisma generate to have a writable directory
ENV HOME=/app
# Set LITELLM_NON_ROOT flag for runtime
ENV LITELLM_NON_ROOT=true
RUN prisma generate
# --- End of Prisma Handling ---
EXPOSE 4000/tcp
# Set entrypoint and command
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
# CMD ["--port", "4000", "--detailed_debug"]
CMD ["--port", "4000"]
CMD ["--port", "4000"]

View file

@ -122,6 +122,48 @@ class MyUser(HttpUser):
```
## LiteLLM vs Portkey Performance Comparison
**Test Configuration**: 4 CPUs, 8 GB RAM per instance | Load: 1k concurrent users, 500 ramp-up
### Multi-Instance (4×) Performance
| Metric | Portkey (no DB) | LiteLLM (with DB) |
| ------------------- | --------------- | ----------------- |
| **Total Requests** | 293,796 | 312,405 |
| **Failed Requests** | 0 | 0 |
| **Median Latency** | 100 ms | 100 ms |
| **p95 Latency** | 230 ms | 150 ms |
| **p99 Latency** | 500 ms | 240 ms |
| **Average Latency** | 123 ms | 111 ms |
| **Current RPS** | 1,170.9 | 1,170 |
### Technical Insights
**Portkey**
**Pros**
* Low memory footprint
* Stable latency with minimal spikes
**Cons**
* CPU utilization capped around ~40%, indicating underutilization of available compute resources
* Experienced three I/O timeout outages
**LiteLLM**
**Pros**
* Fully utilizes available CPU capacity
* Strong connection handling and low latency after initial warm-up spikes
**Cons**
* High memory usage during initialization and per request
## Logging Callbacks

View file

@ -15,16 +15,22 @@ Supported Providers:
- Google AI Studio (`gemini`)
- Vertex AI (`vertex_ai/`)
LiteLLM will standardize the `image` response in the assistant message for models that support image generation during chat completions.
LiteLLM will standardize the `images` response in the assistant message for models that support image generation during chat completions.
```python title="Example response from litellm"
"message": {
...
"content": "Here's the image you requested:",
"image": {
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
"detail": "auto"
}
"images": [
{
"image_url": {
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
"detail": "auto"
},
"index": 0,
"type": "image_url"
}
]
}
```
@ -47,7 +53,7 @@ response = completion(
)
print(response.choices[0].message.content) # Text response
print(response.choices[0].message.image) # Image data
print(response.choices[0].message.images) # List of image objects
```
</TabItem>
@ -103,10 +109,16 @@ curl http://0.0.0.0:4000/v1/chat/completions \
"message": {
"content": "Here's the image you requested:",
"role": "assistant",
"image": {
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
"detail": "auto"
}
"images": [
{
"image_url": {
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
"detail": "auto"
},
"index": 0,
"type": "image_url"
}
]
}
}
],
@ -141,8 +153,8 @@ response = completion(
)
for chunk in response:
if hasattr(chunk.choices[0].delta, "image") and chunk.choices[0].delta.image is not None:
print("Generated image:", chunk.choices[0].delta.image["url"])
if hasattr(chunk.choices[0].delta, "images") and chunk.choices[0].delta.images is not None:
print("Generated image:", chunk.choices[0].delta.images[0]["image_url"]["url"])
break
```
@ -175,7 +187,7 @@ data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"content":"Here's the image you requested:"},"finish_reason":null}]}
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"image":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...","detail":"auto"}},"finish_reason":null}]}
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"images":[{"image_url":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...","detail":"auto"},"index":0,"type":"image_url"}]},"finish_reason":null}]}
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}
@ -200,8 +212,8 @@ async def generate_image():
)
print(response.choices[0].message.content) # Text response
print(response.choices[0].message.image) # Image data
print(response.choices[0].message.images) # List of image objects
return response
# Run the async function
@ -215,18 +227,28 @@ asyncio.run(generate_image())
| Google AI Studio | `gemini/gemini-2.5-flash-image-preview` |
| Vertex AI | `vertex_ai/gemini-2.5-flash-image-preview` |
## Spec
## Spec
The `image` field in the response follows this structure:
The `images` field in the response follows this structure:
```python
"image": {
"url": "data:image/png;base64,<base64_encoded_image>",
"detail": "auto"
}
"images": [
{
"image_url": {
"url": "data:image/png;base64,<base64_encoded_image>",
"detail": "auto"
},
"index": 0,
"type": "image_url"
}
]
```
- `url` - str: Base64 encoded image data in data URI format
- `detail` - str: Image detail level (always "auto" for generated images)
- `images` - List[ImageURLListItem]: Array of generated images
- `image_url` - ImageURLObject: Container for image data
- `url` - str: Base64 encoded image data in data URI format
- `detail` - str: Image detail level (always "auto" for generated images)
- `index` - int: Index of the image in the response
- `type` - str: Type identifier (always "image_url")
The image is returned as a base64-encoded data URI that can be directly used in HTML `<img>` tags or saved to a file.
The images are returned as base64-encoded data URIs that can be directly used in HTML `<img>` tags or saved to files.

View file

@ -1,6 +1,3 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# /containers
Manage OpenAI code interpreter containers (sessions) for executing code in isolated environments.
@ -14,17 +11,15 @@ Manage OpenAI code interpreter containers (sessions) for executing code in isola
| Spend Management | ✅ Budget tracking and rate limiting |
| Supported Providers | `openai`|
## **Supported Providers**:
- [OpenAI](#quick-start)
## Quick Start
:::tip
Containers provide isolated execution environments for code interpreter sessions. You can create, list, retrieve, and delete containers.
### SDK, PROXY, and OpenAI Client
:::
<Tabs>
<TabItem value="sdk" label="SDK">
## **LiteLLM Python SDK Usage**
### Quick Start
**Create a Container**
@ -46,22 +41,33 @@ container = litellm.create_container(
print(f"Container ID: {container.id}")
print(f"Container Name: {container.name}")
### ASYNC USAGE ###
# container = await litellm.acreate_container(
# name="My Code Interpreter Container",
# custom_llm_provider="openai",
# expires_after={
# "anchor": "last_active_at",
# "minutes": 20
# }
# )
```
**List Containers**
### Async Usage
```python
from litellm import list_containers, alist_containers
from litellm import acreate_container
import os
os.environ["OPENAI_API_KEY"] = "sk-.."
container = await acreate_container(
name="My Code Interpreter Container",
custom_llm_provider="openai",
expires_after={
"anchor": "last_active_at",
"minutes": 20
}
)
print(f"Container ID: {container.id}")
print(f"Container Name: {container.name}")
```
### List Containers
```python
from litellm import list_containers
import os
os.environ["OPENAI_API_KEY"] = "sk-.."
@ -75,19 +81,28 @@ containers = list_containers(
print(f"Found {len(containers.data)} containers")
for container in containers.data:
print(f" - {container.id}: {container.name}")
### ASYNC USAGE ###
# containers = await alist_containers(
# custom_llm_provider="openai",
# limit=20,
# order="desc"
# )
```
**Retrieve a Container**
**Async Usage:**
```python
from litellm import retrieve_container, aretrieve_container
from litellm import alist_containers
containers = await alist_containers(
custom_llm_provider="openai",
limit=20,
order="desc"
)
print(f"Found {len(containers.data)} containers")
for container in containers.data:
print(f" - {container.id}: {container.name}")
```
### Retrieve a Container
```python
from litellm import retrieve_container
import os
os.environ["OPENAI_API_KEY"] = "sk-.."
@ -100,18 +115,27 @@ container = retrieve_container(
print(f"Container: {container.name}")
print(f"Status: {container.status}")
print(f"Created: {container.created_at}")
### ASYNC USAGE ###
# container = await aretrieve_container(
# container_id="cntr_123...",
# custom_llm_provider="openai"
# )
```
**Delete a Container**
**Async Usage:**
```python
from litellm import delete_container, adelete_container
from litellm import aretrieve_container
container = await aretrieve_container(
container_id="cntr_123...",
custom_llm_provider="openai"
)
print(f"Container: {container.name}")
print(f"Status: {container.status}")
print(f"Created: {container.created_at}")
```
### Delete a Container
```python
from litellm import delete_container
import os
os.environ["OPENAI_API_KEY"] = "sk-.."
@ -123,16 +147,30 @@ result = delete_container(
print(f"Deleted: {result.deleted}")
print(f"Container ID: {result.id}")
### ASYNC USAGE ###
# result = await adelete_container(
# container_id="cntr_123...",
# custom_llm_provider="openai"
# )
```
</TabItem>
<TabItem value="proxy" label="LiteLLM PROXY Server">
**Async Usage:**
```python
from litellm import adelete_container
result = await adelete_container(
container_id="cntr_123...",
custom_llm_provider="openai"
)
print(f"Deleted: {result.deleted}")
print(f"Container ID: {result.id}")
```
## **LiteLLM Proxy Usage**
LiteLLM provides OpenAI API compatible container endpoints for managing code interpreter sessions:
- `/v1/containers` - Create and list containers
- `/v1/containers/{container_id}` - Retrieve and delete containers
**Setup**
```bash
$ export OPENAI_API_KEY="sk-..."
@ -208,10 +246,13 @@ curl -X DELETE "http://localhost:4000/v1/containers/cntr_123..." \
-H "Authorization: Bearer sk-1234"
```
</TabItem>
<TabItem value="openai" label="OpenAI Python Client">
## **Using OpenAI Client with LiteLLM Proxy**
**Setup**
You can use the standard OpenAI Python client to interact with LiteLLM's container endpoints. This provides a familiar interface while leveraging LiteLLM's proxy features.
### Setup
First, configure your OpenAI client to point to your LiteLLM proxy:
```python
from openai import OpenAI
@ -222,7 +263,7 @@ client = OpenAI(
)
```
**Create a Container**
### Create a Container
```python
container = client.containers.create(
@ -239,7 +280,7 @@ print(f"Container Name: {container.name}")
print(f"Created at: {container.created_at}")
```
**List Containers**
### List Containers
```python
containers = client.containers.list(
@ -252,7 +293,7 @@ for container in containers.data:
print(f" - {container.id}: {container.name}")
```
**Retrieve a Container**
### Retrieve a Container
```python
container = client.containers.retrieve(
@ -265,7 +306,7 @@ print(f"Status: {container.status}")
print(f"Last active: {container.last_active_at}")
```
**Delete a Container**
### Delete a Container
```python
result = client.containers.delete(
@ -277,8 +318,62 @@ print(f"Deleted: {result.deleted}")
print(f"Container ID: {result.id}")
```
</TabItem>
</Tabs>
### Complete Workflow Example
Here's a complete example showing the full container management workflow:
```python
from openai import OpenAI
# Initialize client
client = OpenAI(
api_key="sk-1234",
base_url="http://localhost:4000"
)
# 1. Create a container
print("Creating container...")
container = client.containers.create(
name="My Code Interpreter Session",
expires_after={
"anchor": "last_active_at",
"minutes": 20
},
extra_body={"custom_llm_provider": "openai"}
)
container_id = container.id
print(f"Container created. ID: {container_id}")
# 2. List all containers
print("\nListing containers...")
containers = client.containers.list(
extra_body={"custom_llm_provider": "openai"}
)
for c in containers.data:
print(f" - {c.id}: {c.name} (Status: {c.status})")
# 3. Retrieve specific container
print(f"\nRetrieving container {container_id}...")
retrieved = client.containers.retrieve(
container_id=container_id,
extra_body={"custom_llm_provider": "openai"}
)
print(f"Container: {retrieved.name}")
print(f"Status: {retrieved.status}")
print(f"Last active: {retrieved.last_active_at}")
# 4. Delete container
print(f"\nDeleting container {container_id}...")
result = client.containers.delete(
container_id=container_id,
extra_body={"custom_llm_provider": "openai"}
)
print(f"Deleted: {result.deleted}")
```
## Container Parameters
@ -356,3 +451,15 @@ print(f"Container ID: {result.id}")
}
```
## **Supported Providers**
| Provider | Support Status | Notes |
|-------------|----------------|-------|
| OpenAI | ✅ Supported | Full support for all container operations |
:::info
Currently, only OpenAI supports container management for code interpreter sessions. Support for additional providers may be added in the future.
:::

View file

@ -112,6 +112,85 @@ except openai.APITimeoutError as e:
print(f"should_retry: {should_retry}")
```
## Advanced
### Accessing Provider-Specific Error Details
LiteLLM exceptions include a `provider_specific_fields` attribute that contains additional error information specific to each provider. This is particularly useful for Azure OpenAI, which provides detailed content filtering information.
#### Azure OpenAI - Content Policy Violation Inner Error Access
When Azure OpenAI returns content policy violations, you can access the detailed content filtering results through the `innererror` field:
```python
import litellm
from litellm.exceptions import ContentPolicyViolationError
try:
response = litellm.completion(
model="azure/gpt-4",
messages=[
{
"role": "user",
"content": "Some content that might violate policies"
}
]
)
except ContentPolicyViolationError as e:
# Access Azure-specific error details
if e.provider_specific_fields and "innererror" in e.provider_specific_fields:
innererror = e.provider_specific_fields["innererror"]
# Access content filter results
content_filter_result = innererror.get("content_filter_result", {})
print(f"Content filter code: {innererror.get('code')}")
print(f"Hate filtered: {content_filter_result.get('hate', {}).get('filtered')}")
print(f"Violence severity: {content_filter_result.get('violence', {}).get('severity')}")
print(f"Sexual content filtered: {content_filter_result.get('sexual', {}).get('filtered')}")
```
**Example Response Structure:**
When calling the LiteLLM proxy, content policy violations will return detailed filtering information:
```json
{
"error": {
"message": "litellm.ContentPolicyViolationError: AzureException - The response was filtered due to the prompt triggering Azure OpenAI's content management policy...",
"type": null,
"param": null,
"code": "400",
"provider_specific_fields": {
"innererror": {
"code": "ResponsibleAIPolicyViolation",
"content_filter_result": {
"hate": {
"filtered": true,
"severity": "high"
},
"jailbreak": {
"filtered": false,
"detected": false
},
"self_harm": {
"filtered": false,
"severity": "safe"
},
"sexual": {
"filtered": false,
"severity": "safe"
},
"violence": {
"filtered": true,
"severity": "medium"
}
}
}
}
}
}
## Details
To see how it's implemented - [check out the code](https://github.com/BerriAI/litellm/blob/a42c197e5a6de56ea576c73715e6c7c6b19fa249/litellm/utils.py#L1217)

View file

@ -107,6 +107,26 @@ For stdio MCP servers, select "Standard Input/Output (stdio)" as the transport t
style={{width: '80%', display: 'block', margin: '0'}}
/>
<br/>
<br/>
### Static Headers
Sometimes your MCP server needs specific headers on every request. Maybe it's an API key, maybe it's a custom header the server expects. Instead of configuring auth, you can just set them directly.
<Image
img={require('../img/static_headers.png')}
style={{width: '80%', display: 'block', margin: '0'}}
/>
These headers get sent with every request to the server. That's it.
**When to use this:**
- Your server needs custom headers that don't fit the standard auth patterns
- You want full control over exactly what headers are sent
- You're debugging and need to quickly add headers without changing auth configuration
</TabItem>
<TabItem value="config" label="config.yaml">
@ -175,6 +195,7 @@ mcp_servers:
| `authorization` | `Authorization: <auth_value>` |
- **Extra Headers**: Optional list of additional header names that should be forwarded from client to the MCP server
- **Static Headers**: Optional map of header key/value pairs to include every request to the MCP server.
- **Spec Version**: Optional MCP specification version (defaults to `2025-06-18`)
Examples for each auth type:
@ -217,28 +238,15 @@ mcp_servers:
auth_type: "bearer_token"
auth_value: "ghp_example_token"
extra_headers: ["custom_key", "x-custom-header"] # These headers will be forwarded from client
```
### Static Headers
Sometimes your MCP server needs specific headers on every request. Maybe it's an API key, maybe it's a custom header the server expects. Instead of configuring auth, you can just set them directly.
```yaml title="config.yaml" showLineNumbers
mcp_servers:
# Example with static headers
my_mcp_server:
url: "https://my-mcp-server.com/mcp"
static_headers:
static_headers: # These headers will be requested to the MCP server
X-API-Key: "abc123"
X-Custom-Header: "some-value"
```
These headers get sent with every request to the server. That's it.
**When to use this:**
- Your server needs custom headers that don't fit the standard auth patterns
- You want full control over exactly what headers are sent
- You're debugging and need to quickly add headers without changing auth configuration
### MCP Aliases
You can define aliases for your MCP servers in the `litellm_settings` section. This allows you to:

View file

@ -22,10 +22,19 @@ response = moderation(
For `/moderations` endpoint, there is **no need to specify `model` in the request or on the litellm config.yaml**
Start litellm proxy server
1. Setup config.yaml
```yaml
model_list:
- model_name: text-moderation-stable
litellm_params:
model: openai/omni-moderation-latest
```
2. Start litellm proxy server
```
litellm
litellm --config /path/to/config.yaml
```
@ -41,7 +50,7 @@ client = OpenAI(api_key="<proxy-api-key>", base_url="http://0.0.0.0:4000")
response = client.moderations.create(
input="hello from litellm",
model="text-moderation-stable" # optional, defaults to `omni-moderation-latest`
model="text-moderation-stable"
)
print(response)

View file

@ -56,12 +56,32 @@ litellm_settings:
**Step 2**: Set Required env variables for datadog
#### Direct API
Send logs directly to Datadog API:
```shell
DD_API_KEY="5f2d0f310***********" # your datadog API Key
DD_SITE="us5.datadoghq.com" # your datadog base url
DD_SOURCE="litellm_dev" # [OPTIONAL] your datadog source. use to differentiate dev vs. prod deployments
```
#### Via DataDog Agent
Send logs through a local DataDog agent (useful for containerized environments):
```shell
DD_AGENT_HOST="localhost" # hostname or IP of DataDog agent
DD_AGENT_PORT="10518" # [OPTIONAL] port of DataDog agent (default: 10518)
DD_API_KEY="5f2d0f310***********" # [OPTIONAL] your datadog API Key (agent handles auth)
DD_SOURCE="litellm_dev" # [OPTIONAL] your datadog source
```
When `DD_AGENT_HOST` is set, logs are sent to the agent instead of directly to DataDog API. This is useful for:
- Centralized log shipping in containerized environments
- Reducing direct API calls from multiple services
- Leveraging agent-side processing and filtering
**Step 3**: Start the proxy, make a test request
Start proxy
@ -169,8 +189,10 @@ LiteLLM supports customizing the following Datadog environment variables
| Environment Variable | Description | Default Value | Required |
|---------------------|-------------|---------------|----------|
| `DD_API_KEY` | Your Datadog API key for authentication | None | ✅ Yes |
| `DD_SITE` | Your Datadog site (e.g., "us5.datadoghq.com") | None | ✅ Yes |
| `DD_API_KEY` | Your Datadog API key for authentication (required for direct API, optional for agent) | None | Conditional* |
| `DD_SITE` | Your Datadog site (e.g., "us5.datadoghq.com") (required for direct API) | None | Conditional* |
| `DD_AGENT_HOST` | Hostname or IP of DataDog agent (e.g., "localhost"). When set, logs are sent to agent instead of direct API | None | ❌ No |
| `DD_AGENT_PORT` | Port of DataDog agent for log intake | "10518" | ❌ No |
| `DD_ENV` | Environment tag for your logs (e.g., "production", "staging") | "unknown" | ❌ No |
| `DD_SERVICE` | Service name for your logs | "litellm-server" | ❌ No |
| `DD_SOURCE` | Source name for your logs | "litellm" | ❌ No |
@ -178,3 +200,6 @@ LiteLLM supports customizing the following Datadog environment variables
| `HOSTNAME` | Hostname tag for your logs | "" | ❌ No |
| `POD_NAME` | Pod name tag (useful for Kubernetes deployments) | "unknown" | ❌ No |
\* **Required when using Direct API** (default): `DD_API_KEY` and `DD_SITE` are required
\* **Optional when using DataDog Agent**: Set `DD_AGENT_HOST` to use agent mode; `DD_API_KEY` and `DD_SITE` are not required

View file

@ -5,7 +5,7 @@
| Cost Tracking | ✅ |
| Logging | ✅ (Basic Logging not supported) |
| Load Balancing | ✅ |
| Supported Providers | `mistral`, `azure_ai` |
| Supported Providers | `mistral`, `azure_ai`, `vertex_ai` |
:::tip
@ -262,4 +262,5 @@ The response follows Mistral's OCR format with the following structure:
|-------------|--------------------|
| Mistral AI | [Usage](#quick-start) |
| Azure AI | [Usage](../docs/providers/azure_ocr) |
| Vertex AI | [Usage](../docs/providers/vertex_ocr) |

View file

@ -19,6 +19,9 @@ Simply replace `https://api.openai.com` with `LITELLM_PROXY_BASE_URL/openai`
## Usage Examples
Requirements:
Set `OPENAI_API_KEY` in your environment variables.
### Assistants API
#### Create OpenAI Client

View file

@ -953,7 +953,7 @@ except Exception as e:
s/o @[Shekhar Patnaik](https://www.linkedin.com/in/patnaikshekhar) for requesting this!
### Anthropic Hosted Tools (Computer, Text Editor, Web Search)
### Anthropic Hosted Tools (Computer, Text Editor, Web Search, Memory)
<Tabs>
@ -1183,6 +1183,72 @@ curl http://0.0.0.0:4000/v1/chat/completions \
</Tabs>
</TabItem>
<TabItem value="memory" label="Memory">
:::info
The Anthropic Memory tool is currently in beta.
:::
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
tools = [{
"type": "memory_20250818",
"name": "memory"
}]
model = "claude-sonnet-4-5-20250929"
messages = [{"role": "user", "content": "Please remember that my favorite color is blue."}]
response = completion(
model=model,
messages=messages,
tools=tools,
)
print(response)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
1. Setup config.yaml
```yaml
model_list:
- model_name: claude-memory-model
litellm_params:
model: anthropic/claude-sonnet-4-5-20250929
api_key: os.environ/ANTHROPIC_API_KEY
```
2. Start proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
```bash
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer $LITELLM_KEY" \
-d '{
"model": "claude-memory-model",
"messages": [{"role": "user", "content": "Please remember that my favorite color is blue."}],
"tools": [{"type": "memory_20250818", "name": "memory"}]
}'
```
</TabItem>
</Tabs>
</TabItem>
</Tabs>

View file

@ -25,7 +25,6 @@ LiteLLM supports Azure OpenAI's video generation models including Sora with full
import os
os.environ["AZURE_OPENAI_API_KEY"] = "your-azure-api-key"
os.environ["AZURE_OPENAI_API_BASE"] = "https://your-resource.openai.azure.com/"
os.environ["AZURE_OPENAI_API_VERSION"] = "2024-02-15-preview"
```
### Basic Usage
@ -37,7 +36,6 @@ import time
os.environ["AZURE_OPENAI_API_KEY"] = "your-azure-api-key"
os.environ["AZURE_OPENAI_API_BASE"] = "https://your-resource.openai.azure.com/"
os.environ["AZURE_OPENAI_API_VERSION"] = "2024-02-15-preview"
# Generate video
response = video_generation(
@ -53,8 +51,7 @@ print(f"Initial Status: {response.status}")
# Check status until video is ready
while True:
status_response = video_status(
video_id=response.id,
custom_llm_provider="azure"
video_id=response.id
)
print(f"Current Status: {status_response.status}")
@ -69,8 +66,7 @@ while True:
# Download video content when ready
video_bytes = video_content(
video_id=response.id,
custom_llm_provider="azure"
video_id=response.id
)
# Save to file
@ -87,7 +83,6 @@ Here's how to call Azure video generation models with the LiteLLM Proxy Server
```bash
export AZURE_OPENAI_API_KEY="your-azure-api-key"
export AZURE_OPENAI_API_BASE="https://your-resource.openai.azure.com/"
export AZURE_OPENAI_API_VERSION="2024-02-15-preview"
```
### 2. Start the proxy
@ -102,7 +97,6 @@ model_list:
model: azure/sora-2
api_key: os.environ/AZURE_OPENAI_API_KEY
api_base: os.environ/AZURE_OPENAI_API_BASE
api_version: "2024-02-15-preview"
```
</TabItem>
@ -211,8 +205,7 @@ general_settings:
```python
# Download video content
video_bytes = video_content(
video_id="video_1234567890",
model="azure/sora-2"
video_id="video_1234567890"
)
# Save to file
@ -243,8 +236,7 @@ def generate_and_download_video(prompt):
# Step 3: Download video
video_bytes = litellm.video_content(
video_id=video_id,
custom_llm_provider="azure"
video_id=video_id
)
# Step 4: Save to file
@ -264,9 +256,9 @@ video_file = generate_and_download_video(
```python
# Video editing with reference image
response = litellm.video_remix(
video_id="video_456",
prompt="Make the cat jump higher",
input_reference=open("path/to/image.jpg", "rb"), # Reference image as file object
custom_llm_provider="azure"
seconds="8"
)

View file

@ -0,0 +1,408 @@
# Azure Document Intelligence OCR
## Overview
| Property | Details |
|-------|-------|
| Description | Azure Document Intelligence (formerly Form Recognizer) provides advanced document analysis capabilities including text extraction, layout analysis, and structure recognition |
| Provider Route on LiteLLM | `azure_ai/doc-intelligence/` |
| Supported Operations | `/ocr` |
| Link to Provider Doc | [Azure Document Intelligence ↗](https://learn.microsoft.com/en-us/azure/ai-services/document-intelligence/)
Extract text and analyze document structure using Azure Document Intelligence's powerful prebuilt models.
## Quick Start
### **LiteLLM SDK**
```python showLineNumbers title="SDK Usage"
import litellm
import os
# Set environment variables
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
# OCR with PDF URL
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
# Access extracted text
for page in response.pages:
print(f"Page {page.index}:")
print(page.markdown)
```
### **LiteLLM PROXY**
```yaml showLineNumbers title="proxy_config.yaml"
model_list:
- model_name: azure-doc-intel
litellm_params:
model: azure_ai/doc-intelligence/prebuilt-layout
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
model_info:
mode: ocr
```
**Start Proxy**
```bash
litellm --config proxy_config.yaml
```
**Call OCR via Proxy**
```bash showLineNumbers title="cURL Request"
curl -X POST http://localhost:4000/ocr \
-H "Content-Type: application/json" \
-H "Authorization: Bearer your-api-key" \
-d '{
"model": "azure-doc-intel",
"document": {
"type": "document_url",
"document_url": "https://arxiv.org/pdf/2201.04234"
}
}'
```
## How It Works
Azure Document Intelligence uses an asynchronous API pattern. LiteLLM AI Gateway handles the request/response transformation and polling automatically.
### Complete Flow Diagram
```mermaid
sequenceDiagram
participant Client
box rgb(200, 220, 255) LiteLLM AI Gateway
participant LiteLLM
end
participant Azure as Azure Document Intelligence
Client->>LiteLLM: POST /ocr (Mistral format)
Note over LiteLLM: Transform to Azure format
LiteLLM->>Azure: POST :analyze
Azure-->>LiteLLM: 202 Accepted + polling URL
Note over LiteLLM: Automatic Polling
loop Every 2-10 seconds
LiteLLM->>Azure: GET polling URL
Azure-->>LiteLLM: Status: running
end
LiteLLM->>Azure: GET polling URL
Azure-->>LiteLLM: Status: succeeded + results
Note over LiteLLM: Transform to Mistral format
LiteLLM-->>Client: OCR Response (Mistral format)
```
### What LiteLLM Does For You
When you call `litellm.ocr()` via SDK or `/ocr` via Proxy:
1. **Request Transformation**: Converts Mistral OCR format → Azure Document Intelligence format
2. **Submits Document**: Sends transformed request to Azure DI API
3. **Handles 202 Response**: Captures the `Operation-Location` URL from response headers
4. **Automatic Polling**:
- Polls the operation URL at intervals specified by `retry-after` header (default: 2 seconds)
- Continues until status is `succeeded` or `failed`
- Respects Azure's rate limiting via `retry-after` headers
5. **Response Transformation**: Converts Azure DI format → Mistral OCR format
6. **Returns Result**: Sends unified Mistral format response to client
**Polling Configuration:**
- Default timeout: 120 seconds
- Configurable via `AZURE_OPERATION_POLLING_TIMEOUT` environment variable
- Uses sync (`time.sleep()`) or async (`await asyncio.sleep()`) based on call type
:::info
**Typical processing time**: 2-10 seconds depending on document size and complexity
:::
## Supported Models
Azure Document Intelligence offers several prebuilt models optimized for different use cases:
### prebuilt-layout (Recommended)
Best for general document OCR with structure preservation.
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
<Tabs>
<TabItem value="sdk" label="SDK">
```python showLineNumbers title="Layout Model - SDK"
import litellm
import os
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
```
</TabItem>
<TabItem value="proxy" label="Proxy Config">
```yaml showLineNumbers title="proxy_config.yaml"
model_list:
- model_name: azure-layout
litellm_params:
model: azure_ai/doc-intelligence/prebuilt-layout
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
model_info:
mode: ocr
```
**Usage:**
```bash
curl -X POST http://localhost:4000/ocr \
-H "Authorization: Bearer your-api-key" \
-d '{"model": "azure-layout", "document": {"type": "document_url", "document_url": "https://example.com/doc.pdf"}}'
```
</TabItem>
</Tabs>
**Features:**
- Text extraction with markdown formatting
- Table detection and extraction
- Document structure analysis
- Paragraph and section recognition
**Pricing:** $10 per 1,000 pages
### prebuilt-read
Optimized for reading text from documents - fastest and most cost-effective.
<Tabs>
<TabItem value="sdk" label="SDK">
```python showLineNumbers title="Read Model - SDK"
import litellm
import os
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-read",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
```
</TabItem>
<TabItem value="proxy" label="Proxy Config">
```yaml showLineNumbers title="proxy_config.yaml"
model_list:
- model_name: azure-read
litellm_params:
model: azure_ai/doc-intelligence/prebuilt-read
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
model_info:
mode: ocr
```
**Usage:**
```bash
curl -X POST http://localhost:4000/ocr \
-H "Authorization: Bearer your-api-key" \
-d '{"model": "azure-read", "document": {"type": "document_url", "document_url": "https://example.com/doc.pdf"}}'
```
</TabItem>
</Tabs>
**Features:**
- Fast text extraction
- Optimized for reading-heavy documents
- Basic structure recognition
**Pricing:** $1.50 per 1,000 pages
### prebuilt-document
General-purpose document analysis with key-value pairs.
<Tabs>
<TabItem value="sdk" label="SDK">
```python showLineNumbers title="Document Model - SDK"
import litellm
import os
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-document",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
```
</TabItem>
<TabItem value="proxy" label="Proxy Config">
```yaml showLineNumbers title="proxy_config.yaml"
model_list:
- model_name: azure-document
litellm_params:
model: azure_ai/doc-intelligence/prebuilt-document
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
model_info:
mode: ocr
```
**Usage:**
```bash
curl -X POST http://localhost:4000/ocr \
-H "Authorization: Bearer your-api-key" \
-d '{"model": "azure-document", "document": {"type": "document_url", "document_url": "https://example.com/doc.pdf"}}'
```
</TabItem>
</Tabs>
**Pricing:** $10 per 1,000 pages
## Document Types
Azure Document Intelligence supports various document formats.
### PDF Documents
```python showLineNumbers title="PDF OCR"
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
```
### Image Documents
```python showLineNumbers title="Image OCR"
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={
"type": "image_url",
"image_url": "https://example.com/image.png"
}
)
```
**Supported image formats:** JPEG, PNG, BMP, TIFF
### Base64 Encoded Documents
```python showLineNumbers title="Base64 PDF"
import base64
# Read and encode PDF
with open("document.pdf", "rb") as f:
pdf_base64 = base64.b64encode(f.read()).decode()
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={
"type": "document_url",
"document_url": f"data:application/pdf;base64,{pdf_base64}"
}
)
```
## Response Format
```python showLineNumbers title="Response Structure"
# Response has the following structure
response.pages # List of pages with extracted text
response.model # Model used
response.object # "ocr"
response.usage_info # Token usage information
# Access page content
for page in response.pages:
print(f"Page {page.index}:")
print(page.markdown)
# Page dimensions (in pixels)
if page.dimensions:
print(f"Width: {page.dimensions.width}px")
print(f"Height: {page.dimensions.height}px")
```
## Async Support
```python showLineNumbers title="Async Usage"
import litellm
import asyncio
async def process_document():
response = await litellm.aocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
return response
# Run async function
response = asyncio.run(process_document())
```
## Cost Tracking
LiteLLM automatically tracks costs for Azure Document Intelligence OCR:
| Model | Cost per 1,000 Pages |
|-------|---------------------|
| prebuilt-read | $1.50 |
| prebuilt-layout | $10.00 |
| prebuilt-document | $10.00 |
```python showLineNumbers title="View Cost"
response = litellm.ocr(
model="azure_ai/doc-intelligence/prebuilt-layout",
document={"type": "document_url", "document_url": "https://..."}
)
# Access cost information
print(f"Cost: ${response._hidden_params.get('response_cost', 0)}")
```
## Additional Resources
- [Azure Document Intelligence Documentation](https://learn.microsoft.com/en-us/azure/ai-services/document-intelligence/)
- [Pricing Details](https://azure.microsoft.com/en-us/pricing/details/ai-document-intelligence/)
- [Supported File Formats](https://learn.microsoft.com/en-us/azure/ai-services/document-intelligence/concept-model-overview)
- [LiteLLM OCR Documentation](https://docs.litellm.ai/docs/ocr)

View file

@ -1,4 +1,4 @@
# Azure AI OCR
# Azure AI OCR (Mistral)
## Overview

View file

@ -0,0 +1,246 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Bedrock AgentCore
Call Bedrock AgentCore in the OpenAI Request/Response format.
| Property | Details |
|----------|---------|
| Description | Amazon Bedrock AgentCore provides direct access to hosted agent runtimes for executing agentic workflows with foundation models. |
| Provider Route on LiteLLM | `bedrock/agentcore/{AGENT_RUNTIME_ARN}` |
| Provider Doc | [AWS Bedrock AgentCore ↗](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_agentcore_InvokeAgentRuntime.html) |
## Quick Start
### Model Format to LiteLLM
To call a bedrock agent runtime through LiteLLM, use the following model format.
Here the `model=bedrock/agentcore/` tells LiteLLM to call the bedrock `InvokeAgentRuntime` API.
```shell showLineNumbers title="Model Format to LiteLLM"
bedrock/agentcore/{AGENT_RUNTIME_ARN}
```
**Example:**
- `bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime`
You can find the Agent Runtime ARN in your AWS Bedrock console under AgentCore.
### LiteLLM Python SDK
```python showLineNumbers title="Basic AgentCore Completion"
import litellm
# Make a completion request to your AgentCore runtime
response = litellm.completion(
model="bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime",
messages=[
{
"role": "user",
"content": "Explain machine learning in simple terms"
}
],
)
print(response.choices[0].message.content)
print(f"Usage: {response.usage}")
```
```python showLineNumbers title="Streaming AgentCore Responses"
import litellm
# Stream responses from your AgentCore runtime
response = litellm.completion(
model="bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime",
messages=[
{
"role": "user",
"content": "What are the key principles of software architecture?"
}
],
stream=True,
)
for chunk in response:
if chunk.choices[0].delta.content:
print(chunk.choices[0].delta.content, end="")
```
### LiteLLM Proxy
#### 1. Configure your model in config.yaml
<Tabs>
<TabItem value="config-yaml" label="config.yaml">
```yaml showLineNumbers title="LiteLLM Proxy Configuration"
model_list:
- model_name: agentcore-runtime-1
litellm_params:
model: bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: us-west-2
- model_name: agentcore-runtime-2
litellm_params:
model: bedrock/agentcore/arn:aws:bedrock-agentcore:us-east-1:987654321098:runtime/production-runtime
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: us-east-1
```
</TabItem>
</Tabs>
#### 2. Start the LiteLLM Proxy
```bash showLineNumbers title="Start LiteLLM Proxy"
litellm --config config.yaml
```
#### 3. Make requests to your AgentCore runtimes
<Tabs>
<TabItem value="curl" label="Curl">
```bash showLineNumbers title="Basic AgentCore Request"
curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer $LITELLM_API_KEY" \
-d '{
"model": "agentcore-runtime-1",
"messages": [
{
"role": "user",
"content": "Summarize the main benefits of cloud computing"
}
]
}'
```
```bash showLineNumbers title="Streaming AgentCore Request"
curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer $LITELLM_API_KEY" \
-d '{
"model": "agentcore-runtime-2",
"messages": [
{
"role": "user",
"content": "Explain the differences between SQL and NoSQL databases"
}
],
"stream": true
}'
```
</TabItem>
<TabItem value="openai-sdk" label="OpenAI Python SDK">
```python showLineNumbers title="Using OpenAI SDK with LiteLLM Proxy"
from openai import OpenAI
# Initialize client with your LiteLLM proxy URL
client = OpenAI(
base_url="http://localhost:4000",
api_key="your-litellm-api-key"
)
# Make a completion request to your AgentCore runtime
response = client.chat.completions.create(
model="agentcore-runtime-1",
messages=[
{
"role": "user",
"content": "What are best practices for API design?"
}
]
)
print(response.choices[0].message.content)
```
```python showLineNumbers title="Streaming with OpenAI SDK"
from openai import OpenAI
client = OpenAI(
base_url="http://localhost:4000",
api_key="your-litellm-api-key"
)
# Stream AgentCore responses
stream = client.chat.completions.create(
model="agentcore-runtime-2",
messages=[
{
"role": "user",
"content": "Describe the microservices architecture pattern"
}
],
stream=True
)
for chunk in stream:
if chunk.choices[0].delta.content is not None:
print(chunk.choices[0].delta.content, end="")
```
</TabItem>
</Tabs>
## Provider-specific Parameters
AgentCore supports additional parameters that can be passed to customize the runtime invocation.
<Tabs>
<TabItem value="sdk" label="SDK">
```python showLineNumbers title="Using AgentCore-specific parameters"
from litellm import completion
response = litellm.completion(
model="bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime",
messages=[
{
"role": "user",
"content": "Analyze this data and provide insights",
}
],
qualifier="production", # PROVIDER-SPECIFIC: Runtime qualifier/version
runtimeSessionId="session-abc-123", # PROVIDER-SPECIFIC: Custom session ID
)
```
</TabItem>
<TabItem value="proxy" label="Proxy">
```yaml showLineNumbers title="LiteLLM Proxy Configuration with Parameters"
model_list:
- model_name: agentcore-runtime-prod
litellm_params:
model: bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: us-west-2
qualifier: production
```
</TabItem>
</Tabs>
### Available Parameters
| Parameter | Type | Description |
|-----------|------|-------------|
| `qualifier` | string | Optional runtime qualifier/version to invoke a specific version of the agent runtime |
| `runtimeSessionId` | string | Optional custom session ID (must be 33+ characters). If not provided, LiteLLM generates one automatically |
## Further Reading
- [AWS Bedrock AgentCore Documentation](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_agentcore_InvokeAgentRuntime.html)
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)

View file

@ -1,69 +0,0 @@
# Custom LLM API-Endpoints
LiteLLM supports Custom deploy api endpoints
LiteLLM Expects the following input and output for custom LLM API endpoints
### Model Details
For calls to your custom API base ensure:
* Set `api_base="your-api-base"`
* Add `custom/` as a prefix to the `model` param. If your API expects `meta-llama/Llama-2-13b-hf` set `model=custom/meta-llama/Llama-2-13b-hf`
| Model Name | Function Call |
|------------------|--------------------------------------------|
| meta-llama/Llama-2-13b-hf | `response = completion(model="custom/meta-llama/Llama-2-13b-hf", messages=messages, api_base="https://your-custom-inference-endpoint")` |
| meta-llama/Llama-2-13b-hf | `response = completion(model="custom/meta-llama/Llama-2-13b-hf", messages=messages, api_base="https://api.autoai.dev/inference")` |
### Example Call to Custom LLM API using LiteLLM
```python
from litellm import completion
response = completion(
model="custom/meta-llama/Llama-2-13b-hf",
messages= [{"content": "what is custom llama?", "role": "user"}],
temperature=0.2,
max_tokens=10,
api_base="https://api.autoai.dev/inference",
request_timeout=300,
)
print("got response\n", response)
```
#### Setting your Custom API endpoint
Inputs to your custom LLM api bases should follow this format:
```python
resp = requests.post(
your-api_base,
json={
'model': 'meta-llama/Llama-2-13b-hf', # model name
'params': {
'prompt': ["The capital of France is P"],
'max_tokens': 32,
'temperature': 0.7,
'top_p': 1.0,
'top_k': 40,
}
}
)
```
Outputs from your custom LLM api bases should follow this format:
```python
{
'data': [
{
'prompt': 'The capital of France is P',
'output': [
'The capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France'
],
'params': {
'temperature': 0.7,
'top_k': 40,
'top_p': 1
}
}
],
'message': 'ok'
}
```

View file

@ -204,7 +204,7 @@ from litellm import completion
import os
os.environ["FIREWORKS_AI_API_KEY"] = "YOUR_API_KEY"
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.us-virginia-1.direct.fireworks.ai/v1"
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.api.fireworks.ai/v1"
completion = litellm.completion(
model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct",
@ -343,7 +343,7 @@ from litellm import transcription
import os
os.environ["FIREWORKS_AI_API_KEY"] = "YOUR_API_KEY"
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.us-virginia-1.direct.fireworks.ai/v1"
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.api.fireworks.ai/v1"
response = transcription(
model="fireworks_ai/whisper-v3",
@ -363,7 +363,7 @@ model_list:
- model_name: whisper-v3
litellm_params:
model: fireworks_ai/whisper-v3
api_base: https://audio-prod.us-virginia-1.direct.fireworks.ai/v1
api_base: https://audio-prod.api.fireworks.ai/v1
api_key: os.environ/FIREWORKS_API_KEY
model_info:
mode: audio_transcription

View file

@ -10,7 +10,7 @@ import TabItem from '@theme/TabItem';
| Provider Route on LiteLLM | `gemini/` |
| Provider Doc | [Google AI Studio ↗](https://aistudio.google.com/) |
| API Endpoint for Provider | https://generativelanguage.googleapis.com |
| Supported OpenAI Endpoints | `/chat/completions`, [`/embeddings`](../embedding/supported_embedding#gemini-ai-embedding-models), `/completions` |
| Supported OpenAI Endpoints | `/chat/completions`, [`/embeddings`](../embedding/supported_embedding#gemini-ai-embedding-models), `/completions`, [`/videos`](./gemini/videos.md) |
| Pass-through Endpoint | [Supported](../pass_through/google_ai_studio.md) |
<br />

View file

@ -0,0 +1,409 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Gemini Video Generation (Veo)
LiteLLM supports Google's Veo video generation models through a unified API interface.
| Property | Details |
|-------|-------|
| Description | Google's Veo AI video generation models |
| Provider Route on LiteLLM | `gemini/` |
| Supported Models | `veo-3.0-generate-preview`, `veo-3.1-generate-preview` |
| Cost Tracking | ✅ Duration-based pricing |
| Logging Support | ✅ Full request/response logging |
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
| Spend Management | ✅ Budget tracking and rate limiting |
| Link to Provider Doc | [Google Veo Documentation ↗](https://ai.google.dev/gemini-api/docs/video) |
## Quick Start
### Required API Keys
```python
import os
os.environ["GEMINI_API_KEY"] = "your-google-api-key"
# OR
os.environ["GOOGLE_API_KEY"] = "your-google-api-key"
```
### Basic Usage
```python
from litellm import video_generation, video_status, video_content
import os
import time
os.environ["GEMINI_API_KEY"] = "your-google-api-key"
# Step 1: Generate video
response = video_generation(
model="gemini/veo-3.0-generate-preview",
prompt="A cat playing with a ball of yarn in a sunny garden"
)
print(f"Video ID: {response.id}")
print(f"Initial Status: {response.status}") # "processing"
# Step 2: Poll for completion
while True:
status_response = video_status(
video_id=response.id
)
print(f"Current Status: {status_response.status}")
if status_response.status == "completed":
break
elif status_response.status == "failed":
print("Video generation failed")
break
time.sleep(10) # Wait 10 seconds before checking again
# Step 3: Download video content
video_bytes = video_content(
video_id=response.id
)
# Save to file
with open("generated_video.mp4", "wb") as f:
f.write(video_bytes)
print("Video downloaded successfully!")
```
## Supported Models
| Model Name | Description | Max Duration | Status |
|------------|-------------|--------------|--------|
| veo-3.0-generate-preview | Veo 3.0 video generation | 8 seconds | Preview |
| veo-3.1-generate-preview | Veo 3.1 video generation | 8 seconds | Preview |
## Video Generation Parameters
LiteLLM automatically maps OpenAI-style parameters to Veo's format:
| OpenAI Parameter | Veo Parameter | Description | Example |
|------------------|---------------|-------------|---------|
| `prompt` | `prompt` | Text description of the video | "A cat playing" |
| `size` | `aspectRatio` | Video dimensions → aspect ratio | "1280x720" → "16:9" |
| `seconds` | `durationSeconds` | Duration in seconds | "8" → 8 |
| `input_reference` | `image` | Reference image to animate | File object or path |
| `model` | `model` | Model to use | "gemini/veo-3.0-generate-preview" |
### Size to Aspect Ratio Mapping
LiteLLM automatically converts size dimensions to Veo's aspect ratio format:
- `"1280x720"`, `"1920x1080"` → `"16:9"` (landscape)
- `"720x1280"`, `"1080x1920"` → `"9:16"` (portrait)
### Supported Veo Parameters
Based on Veo's API:
- **prompt** (required): Text description with optional audio cues
- **aspectRatio**: `"16:9"` (default) or `"9:16"`
- **resolution**: `"720p"` (default) or `"1080p"` (Veo 3.1 only, 16:9 aspect ratio only)
- **durationSeconds**: Video length (max 8 seconds for most models)
- **image**: Reference image for animation
- **negativePrompt**: What to exclude from the video (Veo 3.1)
- **referenceImages**: Style and content references (Veo 3.1 only)
## Complete Workflow Example
```python
import litellm
import time
def generate_and_download_veo_video(
prompt: str,
output_file: str = "video.mp4",
size: str = "1280x720",
seconds: str = "8"
):
"""
Complete workflow for Veo video generation.
Args:
prompt: Text description of the video
output_file: Where to save the video
size: Video dimensions (e.g., "1280x720" for 16:9)
seconds: Duration in seconds
Returns:
bool: True if successful
"""
print(f"🎬 Generating video: {prompt}")
# Step 1: Initiate generation
response = litellm.video_generation(
model="gemini/veo-3.0-generate-preview",
prompt=prompt,
size=size, # Maps to aspectRatio
seconds=seconds # Maps to durationSeconds
)
video_id = response.id
print(f"✓ Video generation started (ID: {video_id})")
# Step 2: Wait for completion
max_wait_time = 600 # 10 minutes
start_time = time.time()
while time.time() - start_time < max_wait_time:
status_response = litellm.video_status(video_id=video_id)
if status_response.status == "completed":
print("✓ Video generation completed!")
break
elif status_response.status == "failed":
print("✗ Video generation failed")
return False
print(f"⏳ Status: {status_response.status}")
time.sleep(10)
else:
print("✗ Timeout waiting for video generation")
return False
# Step 3: Download video
print("⬇️ Downloading video...")
video_bytes = litellm.video_content(video_id=video_id)
with open(output_file, "wb") as f:
f.write(video_bytes)
print(f"✓ Video saved to {output_file}")
return True
# Use it
generate_and_download_veo_video(
prompt="A serene lake at sunset with mountains in the background",
output_file="sunset_lake.mp4"
)
```
## Async Usage
```python
from litellm import avideo_generation, avideo_status, avideo_content
import asyncio
async def async_video_workflow():
# Generate video
response = await avideo_generation(
model="gemini/veo-3.0-generate-preview",
prompt="A cat playing with a ball of yarn"
)
# Poll for completion
while True:
status = await avideo_status(video_id=response.id)
if status.status == "completed":
break
await asyncio.sleep(10)
# Download content
video_bytes = await avideo_content(video_id=response.id)
with open("video.mp4", "wb") as f:
f.write(video_bytes)
# Run it
asyncio.run(async_video_workflow())
```
## LiteLLM Proxy Usage
### Configuration
Add Veo models to your `config.yaml`:
```yaml
model_list:
- model_name: veo-3
litellm_params:
model: gemini/veo-3.0-generate-preview
api_key: os.environ/GEMINI_API_KEY
```
Start the proxy:
```bash
litellm --config config.yaml
# Server running on http://0.0.0.0:4000
```
### Making Requests
<Tabs>
<TabItem value="curl" label="Curl">
```bash
# Step 1: Generate video
curl --location 'http://0.0.0.0:4000/v1/videos' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "veo-3",
"prompt": "A cat playing with a ball of yarn in a sunny garden"
}'
# Response: {"id": "gemini::operations/generate_12345::...", "status": "processing", ...}
# Step 2: Check status
curl --location 'http://localhost:4000/v1/videos/{video_id}' \
--header 'x-litellm-api-key: sk-1234'
# Step 3: Download video (when status is "completed")
curl --location 'http://localhost:4000/v1/videos/{video_id}/content' \
--header 'x-litellm-api-key: sk-1234' \
--output video.mp4
```
</TabItem>
<TabItem value="python" label="Python SDK">
```python
import litellm
litellm.api_base = "http://0.0.0.0:4000"
litellm.api_key = "sk-1234"
# Generate video
response = litellm.video_generation(
model="veo-3",
prompt="A cat playing with a ball of yarn in a sunny garden"
)
# Check status
import time
while True:
status = litellm.video_status(video_id=response.id)
if status.status == "completed":
break
time.sleep(10)
# Download video
video_bytes = litellm.video_content(video_id=response.id)
with open("video.mp4", "wb") as f:
f.write(video_bytes)
```
</TabItem>
</Tabs>
## Cost Tracking
LiteLLM automatically tracks costs for Veo video generation:
```python
response = litellm.video_generation(
model="gemini/veo-3.0-generate-preview",
prompt="A beautiful sunset"
)
# Cost is calculated based on video duration
# Veo pricing: ~$0.10 per second (estimated)
# Default video duration: ~5 seconds
# Estimated cost: ~$0.50
```
## Differences from OpenAI Video API
| Feature | OpenAI (Sora) | Gemini (Veo) |
|---------|---------------|--------------|
| Reference Images | ✅ Supported | ❌ Not supported |
| Size Control | ✅ Supported | ❌ Not supported |
| Duration Control | ✅ Supported | ❌ Not supported |
| Video Remix/Edit | ✅ Supported | ❌ Not supported |
| Video List | ✅ Supported | ❌ Not supported |
| Prompt-based Generation | ✅ Supported | ✅ Supported |
| Async Operations | ✅ Supported | ✅ Supported |
## Error Handling
```python
from litellm import video_generation, video_status, video_content
from litellm.exceptions import APIError, Timeout
try:
response = video_generation(
model="gemini/veo-3.0-generate-preview",
prompt="A beautiful landscape"
)
# Poll with timeout
max_attempts = 60 # 10 minutes (60 * 10s)
for attempt in range(max_attempts):
status = video_status(video_id=response.id)
if status.status == "completed":
video_bytes = video_content(video_id=response.id)
with open("video.mp4", "wb") as f:
f.write(video_bytes)
break
elif status.status == "failed":
raise APIError("Video generation failed")
time.sleep(10)
else:
raise Timeout("Video generation timed out")
except APIError as e:
print(f"API Error: {e}")
except Timeout as e:
print(f"Timeout: {e}")
except Exception as e:
print(f"Unexpected error: {e}")
```
## Best Practices
1. **Always poll for completion**: Veo video generation is asynchronous and can take several minutes
2. **Set reasonable timeouts**: Allow at least 5-10 minutes for video generation
3. **Handle failures gracefully**: Check for `failed` status and implement retry logic
4. **Use descriptive prompts**: More detailed prompts generally produce better results
5. **Store video IDs**: Save the operation ID/video ID to resume polling if your application restarts
## Troubleshooting
### Video generation times out
```python
# Increase polling timeout
max_wait_time = 900 # 15 minutes instead of 10
```
### Video not found when downloading
```python
# Make sure video is completed before downloading
status = video_status(video_id=video_id)
if status.status != "completed":
print("Video not ready yet!")
```
### API key errors
```python
# Verify your API key is set
import os
print(os.environ.get("GEMINI_API_KEY"))
# Or pass it explicitly
response = video_generation(
model="gemini/veo-3.0-generate-preview",
prompt="...",
api_key="your-api-key-here"
)
```
## See Also
- [OpenAI Video Generation](../openai/videos.md)
- [Azure Video Generation](../azure/videos.md)
- [Vertex AI Video Generation](../vertex_ai/videos.md)
- [Video Generation API Reference](/docs/videos)
- [Veo Pass-through Endpoints](/docs/pass_through/google_ai_studio#example-4-video-generation-with-veo)

View file

@ -36,7 +36,6 @@ print(f"Status: {response.status}")
# Download video content when ready
video_bytes = video_content(
video_id=response.id,
model="sora-2"
)
# Save to file
@ -44,6 +43,113 @@ with open("generated_video.mp4", "wb") as f:
f.write(video_bytes)
```
## **LiteLLM Proxy Usage**
LiteLLM provides OpenAI API compatible video endpoints for complete video generation workflow:
- `/videos/generations` - Generate new videos
- `/videos/remix` - Edit existing videos with reference images
- `/videos/status` - Check video generation status
- `/videos/retrieval` - Download completed videos
**Setup**
Add this to your litellm proxy config.yaml
```yaml
model_list:
- model_name: sora-2
litellm_params:
model: openai/sora-2
api_key: os.environ/OPENAI_API_KEY
```
Start litellm
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
Test video generation request
```bash
curl --location 'http://localhost:4000/v1/videos' \
--header 'Content-Type: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--data '{
"model": "sora-2",
"prompt": "A beautiful sunset over the ocean"
}'
```
Test video status request
```bash
# Using custom-llm-provider header
curl --location 'http://localhost:4000/v1/videos/video_id' \
--header 'Accept: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: openai'
```
Test video retrieval request
```bash
# Using custom-llm-provider header
curl --location 'http://localhost:4000/v1/videos/video_id/content' \
--header 'Accept: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: openai' \
--output video.mp4
# Or using query parameter
curl --location 'http://localhost:4000/v1/videos/video_id/content?custom_llm_provider=openai' \
--header 'Accept: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--output video.mp4
```
Test video remix request
```bash
# Using custom_llm_provider in request body
curl --location --request POST 'http://localhost:4000/v1/videos/video_id/remix' \
--header 'Accept: application/json' \
--header 'Content-Type: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--data '{
"prompt": "New remix instructions",
"custom_llm_provider": "openai"
}'
# Or using custom-llm-provider header
curl --location --request POST 'http://localhost:4000/v1/videos/video_id/remix' \
--header 'Accept: application/json' \
--header 'Content-Type: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: openai' \
--data '{
"prompt": "New remix instructions"
}'
```
Test OpenAI video generation request
```bash
curl http://localhost:4000/v1/videos \
-H "Authorization: Bearer sk-1234" \
-H "Content-Type: application/json" \
-d '{
"model": "sora-2",
"prompt": "A cat playing with a ball of yarn in a sunny garden",
"seconds": "8",
"size": "720x1280"
}'
```
## Supported Models
| Model Name | Description | Max Duration | Supported Sizes |
@ -64,8 +170,7 @@ with open("generated_video.mp4", "wb") as f:
```python
# Download video content
video_bytes = video_content(
video_id="video_1234567890",
custom_llm_provider="openai" # Or use model="sora-2"
video_id="video_1234567890"
)
# Save to file
@ -96,8 +201,7 @@ def generate_and_download_video(prompt):
# Step 3: Download video
video_bytes = litellm.video_content(
video_id=video_id,
custom_llm_provider="openai"
video_id=video_id
)
# Step 4: Save to file
@ -112,6 +216,7 @@ video_file = generate_and_download_video(
)
```
## Video Editing with Reference Images
```python
@ -133,8 +238,7 @@ from litellm.exceptions import BadRequestError, AuthenticationError
try:
response = video_generation(
prompt="A cat playing with a ball of yarn",
model="sora-2"
prompt="A cat playing with a ball of yarn"
)
except AuthenticationError as e:
print(f"Authentication failed: {e}")

View file

@ -0,0 +1,268 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Vertex AI Video Generation (Veo)
LiteLLM supports Vertex AI's Veo video generation models using the unified OpenAI video API surface.
| Property | Details |
|-------|-------|
| Description | Google Cloud Vertex AI Veo video generation models |
| Provider Route on LiteLLM | `vertex_ai/` |
| Supported Models | `veo-2.0-generate-001`, `veo-3.0-generate-preview`, `veo-3.0-fast-generate-preview`, `veo-3.1-generate-preview`, `veo-3.1-fast-generate-preview` |
| Cost Tracking | ✅ Duration-based pricing |
| Logging Support | ✅ Full request/response logging |
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
| Spend Management | ✅ Budget tracking and rate limiting |
| Link to Provider Doc | [Vertex AI Veo Documentation ↗](https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation) |
## Quick Start
### Required Environment Setup
```python
import json
import os
os.environ["VERTEXAI_PROJECT"] = "your-gcp-project-id"
os.environ["VERTEXAI_LOCATION"] = "us-central1"
# Option 1: Point to a service account file
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = "/path/to/service_account.json"
# Option 2: Store the service account JSON directly
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
os.environ["VERTEXAI_CREDENTIALS"] = f.read()
```
### Basic Usage
```python
from litellm import video_generation, video_status, video_content
import json
import os
import time
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
vertex_credentials = f.read()
response = video_generation(
model="vertex_ai/veo-3.0-generate-preview",
prompt="A cat playing with a ball of yarn in a sunny garden",
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
seconds="8",
size="1280x720",
)
print(f"Video ID: {response.id}")
print(f"Initial Status: {response.status}")
# Poll for completion
while True:
status = video_status(
video_id=response.id,
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
)
print(f"Current Status: {status.status}")
if status.status == "completed":
break
if status.status == "failed":
raise RuntimeError("Video generation failed")
time.sleep(10)
# Download the rendered video
video_bytes = video_content(
video_id=response.id,
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
)
with open("generated_video.mp4", "wb") as f:
f.write(video_bytes)
```
## Supported Models
| Model Name | Description | Max Duration | Status |
|------------|-------------|--------------|--------|
| veo-2.0-generate-001 | Veo 2.0 video generation | 5 seconds | GA |
| veo-3.0-generate-preview | Veo 3.0 high quality | 8 seconds | Preview |
| veo-3.0-fast-generate-preview | Veo 3.0 fast generation | 8 seconds | Preview |
| veo-3.1-generate-preview | Veo 3.1 high quality | 10 seconds | Preview |
| veo-3.1-fast-generate-preview | Veo 3.1 fast | 10 seconds | Preview |
## Video Generation Parameters
LiteLLM converts OpenAI-style parameters to Veo's API shape automatically:
| OpenAI Parameter | Vertex AI Parameter | Description | Example |
|------------------|---------------------|-------------|---------|
| `prompt` | `instances[].prompt` | Text description of the video | "A cat playing" |
| `size` | `parameters.aspectRatio` | Converted to `16:9` or `9:16` | "1280x720" → `16:9` |
| `seconds` | `parameters.durationSeconds` | Clip length in seconds | "8" → `8` |
| `input_reference` | `instances[].image` | Reference image for animation | `open("image.jpg", "rb")` |
| Provider-specific params | `extra_body` | Forwarded to Vertex API | `{"negativePrompt": "blurry"}` |
### Size to Aspect Ratio Mapping
- `1280x720`, `1920x1080` → `16:9`
- `720x1280`, `1080x1920` → `9:16`
- Unknown sizes default to `16:9`
## Async Usage
```python
from litellm import avideo_generation, avideo_status, avideo_content
import asyncio
import json
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
vertex_credentials = f.read()
async def workflow():
response = await avideo_generation(
model="vertex_ai/veo-3.1-generate-preview",
prompt="Slow motion water droplets splashing into a pool",
seconds="10",
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
)
while True:
status = await avideo_status(
video_id=response.id,
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
)
if status.status == "completed":
break
if status.status == "failed":
raise RuntimeError("Video generation failed")
await asyncio.sleep(10)
video_bytes = await avideo_content(
video_id=response.id,
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
)
with open("veo_water.mp4", "wb") as f:
f.write(video_bytes)
asyncio.run(workflow())
```
## LiteLLM Proxy Usage
Add Veo models to your `config.yaml`:
```yaml
model_list:
- model_name: veo-3
litellm_params:
model: vertex_ai/veo-3.0-generate-preview
vertex_project: os.environ/VERTEXAI_PROJECT
vertex_location: os.environ/VERTEXAI_LOCATION
vertex_credentials: os.environ/VERTEXAI_CREDENTIALS
```
Start the proxy and make requests:
<Tabs>
<TabItem value="curl" label="Curl">
```bash
# Step 1: Generate video
curl --location 'http://0.0.0.0:4000/videos' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "veo-3",
"prompt": "Aerial shot over a futuristic city at sunrise",
"seconds": "8"
}'
# Step 2: Poll status
curl --location 'http://localhost:4000/v1/videos/{video_id}' \
--header 'x-litellm-api-key: sk-1234'
# Step 3: Download video
curl --location 'http://localhost:4000/v1/videos/{video_id}/content' \
--header 'x-litellm-api-key: sk-1234' \
--output video.mp4
```
</TabItem>
<TabItem value="python" label="Python SDK">
```python
import litellm
litellm.api_base = "http://0.0.0.0:4000"
litellm.api_key = "sk-1234"
response = litellm.video_generation(
model="veo-3",
prompt="Aerial shot over a futuristic city at sunrise",
)
status = litellm.video_status(video_id=response.id)
while status.status not in ["completed", "failed"]:
status = litellm.video_status(video_id=response.id)
if status.status == "completed":
content = litellm.video_content(video_id=response.id)
with open("veo_city.mp4", "wb") as f:
f.write(content)
```
</TabItem>
</Tabs>
## Cost Tracking
LiteLLM records the duration returned by Veo so you can apply duration-based pricing.
```python
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
vertex_credentials = f.read()
response = video_generation(
model="vertex_ai/veo-2.0-generate-001",
prompt="Flowers blooming in fast forward",
seconds="5",
vertex_project="your-gcp-project-id",
vertex_location="us-central1",
vertex_credentials=vertex_credentials,
)
print(response.usage) # {"duration_seconds": 5.0}
```
## Troubleshooting
- **`vertex_project is required`**: set `VERTEXAI_PROJECT` env var or pass `vertex_project` in the request.
- **`Permission denied`**: ensure the service account has the `Vertex AI User` role and the correct region enabled.
- **Video stuck in `processing`**: Veo operations are long-running. Continue polling every 10–15 seconds up to ~10 minutes.
## See Also
- [OpenAI Video Generation](../openai/videos.md)
- [Azure Video Generation](../azure/videos.md)
- [Gemini Video Generation](../gemini/videos.md)
- [Video Generation API Reference](/docs/videos)

View file

@ -0,0 +1,237 @@
# Vertex AI OCR
## Overview
| Property | Details |
|-------|-------|
| Description | Vertex AI OCR provides document intelligence capabilities powered by Mistral, enabling text extraction from PDFs and images |
| Provider Route on LiteLLM | `vertex_ai/` |
| Supported Operations | `/ocr` |
| Link to Provider Doc | [Vertex AI ↗](https://cloud.google.com/vertex-ai)
Extract text from documents and images using Vertex AI's OCR models, powered by Mistral.
## Quick Start
### **LiteLLM SDK**
```python showLineNumbers title="SDK Usage"
import litellm
import os
# Set environment variables
os.environ["VERTEXAI_PROJECT"] = "your-project-id"
os.environ["VERTEXAI_LOCATION"] = "us-central1"
# OCR with PDF URL
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
}
)
# Access extracted text
for page in response.pages:
print(page.text)
```
### **LiteLLM PROXY**
```yaml showLineNumbers title="proxy_config.yaml"
model_list:
- model_name: vertex-ocr
litellm_params:
model: vertex_ai/mistral-ocr-2505
vertex_project: os.environ/VERTEXAI_PROJECT
vertex_location: os.environ/VERTEXAI_LOCATION
vertex_credentials: path/to/service-account.json # Optional
model_info:
mode: ocr
```
**Start Proxy**
```bash
litellm --config proxy_config.yaml
```
**Call OCR via Proxy**
```bash showLineNumbers title="cURL Request"
curl -X POST http://localhost:4000/ocr \
-H "Content-Type: application/json" \
-H "Authorization: Bearer your-api-key" \
-d '{
"model": "vertex-ocr",
"document": {
"type": "document_url",
"document_url": "https://arxiv.org/pdf/2201.04234"
}
}'
```
## Authentication
Vertex AI OCR supports multiple authentication methods:
### Service Account JSON
```python showLineNumbers title="Service Account Auth"
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={"type": "document_url", "document_url": "https://..."},
vertex_project="your-project-id",
vertex_location="us-central1",
vertex_credentials="path/to/service-account.json"
)
```
### Application Default Credentials
```python showLineNumbers title="Default Credentials"
# Relies on GOOGLE_APPLICATION_CREDENTIALS environment variable
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={"type": "document_url", "document_url": "https://..."},
vertex_project="your-project-id",
vertex_location="us-central1"
)
```
## Document Types
Vertex AI OCR supports both PDFs and images.
### PDF Documents
```python showLineNumbers title="PDF OCR"
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
},
vertex_project="your-project-id",
vertex_location="us-central1"
)
```
### Image Documents
```python showLineNumbers title="Image OCR"
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={
"type": "image_url",
"image_url": "https://example.com/image.png"
},
vertex_project="your-project-id",
vertex_location="us-central1"
)
```
### Base64 Encoded Documents
```python showLineNumbers title="Base64 PDF"
import base64
# Read and encode PDF
with open("document.pdf", "rb") as f:
pdf_base64 = base64.b64encode(f.read()).decode()
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={
"type": "document_url",
"document_url": f"data:application/pdf;base64,{pdf_base64}"
},
vertex_project="your-project-id",
vertex_location="us-central1"
)
```
## Supported Parameters
```python showLineNumbers title="All Parameters"
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={ # Required: Document to process
"type": "document_url",
"document_url": "https://..."
},
vertex_project="your-project-id", # Required: GCP project ID
vertex_location="us-central1", # Optional: Defaults to us-central1
vertex_credentials="path/to/key.json", # Optional: Service account key
include_image_base64=True, # Optional: Include base64 images
pages=[0, 1, 2], # Optional: Specific pages to process
image_limit=10 # Optional: Limit number of images
)
```
## Response Format
```python showLineNumbers title="Response Structure"
# Response has the following structure
response.pages # List of pages with extracted text
response.model # Model used
response.object # "ocr"
response.usage_info # Token usage information
# Access page content
for page in response.pages:
print(f"Page {page.page_number}:")
print(page.text)
```
## Async Support
```python showLineNumbers title="Async Usage"
import litellm
response = await litellm.aocr(
model="vertex_ai/mistral-ocr-2505",
document={
"type": "document_url",
"document_url": "https://example.com/document.pdf"
},
vertex_project="your-project-id",
vertex_location="us-central1"
)
```
## Cost Tracking
LiteLLM automatically tracks costs for Vertex AI OCR:
- **Cost per page**: $0.0005 (based on $1.50 per 1,000 pages)
```python showLineNumbers title="View Cost"
response = litellm.ocr(
model="vertex_ai/mistral-ocr-2505",
document={"type": "document_url", "document_url": "https://..."},
vertex_project="your-project-id"
)
# Access cost information
print(f"Cost: ${response._hidden_params.get('response_cost', 0)}")
```
## Important Notes
:::info URL Conversion
Vertex AI OCR endpoints don't have internet access. LiteLLM automatically converts public URLs to base64 data URIs before sending requests to Vertex AI.
:::
:::tip Regional Availability
Mistral OCR is available in multiple regions. Specify `vertex_location` to use a region closer to your data:
- `us-central1` (default)
- `europe-west1`
- `asia-southeast1`
:::
## Supported Models
- `mistral-ocr-2505` - Latest Mistral OCR model on Vertex AI
Use the Vertex AI provider prefix: `vertex_ai/<model-name>`

View file

@ -399,6 +399,8 @@ router_settings:
| AZURE_COMPUTER_USE_INPUT_COST_PER_1K_TOKENS | Input cost per 1K tokens for Azure Computer Use service
| AZURE_COMPUTER_USE_OUTPUT_COST_PER_1K_TOKENS | Output cost per 1K tokens for Azure Computer Use service
| AZURE_DEFAULT_RESPONSES_API_VERSION | Version of the Azure Default Responses API being used. Default is "preview"
| AZURE_DOCUMENT_INTELLIGENCE_API_VERSION | API version for Azure Document Intelligence service
| AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI | Default DPI (dots per inch) setting for Azure Document Intelligence service
| AZURE_TENANT_ID | Tenant ID for Azure Active Directory
| AZURE_USERNAME | Username for Azure services, use in conjunction with AZURE_PASSWORD for azure ad token with basic username/password workflow
| AZURE_PASSWORD | Password for Azure services, use in conjunction with AZURE_USERNAME for azure ad token with basic username/password workflow
@ -429,6 +431,12 @@ router_settings:
| CLOUDZERO_MAX_FETCHED_DATA_RECORDS | Maximum number of data records to fetch from CloudZero
| CLOUDZERO_TIMEZONE | Timezone for date handling (default: UTC)
| CONFIG_FILE_PATH | File path for configuration file
| CYBERARK_ACCOUNT | CyberArk account name for secret management
| CYBERARK_API_BASE | Base URL for CyberArk API
| CYBERARK_API_KEY | API key for CyberArk secret management service
| CYBERARK_CLIENT_CERT | Path to client certificate for CyberArk authentication
| CYBERARK_CLIENT_KEY | Path to client key for CyberArk authentication
| CYBERARK_USERNAME | Username for CyberArk authentication
| CONFIDENT_API_KEY | API key for DeepEval integration
| CUSTOM_TIKTOKEN_CACHE_DIR | Custom directory for Tiktoken cache
| CONFIDENT_API_KEY | API key for Confident AI (Deepeval) Logging service
@ -452,6 +460,8 @@ router_settings:
| DD_BASE_URL | Base URL for Datadog integration
| DATADOG_BASE_URL | (Alternative to DD_BASE_URL) Base URL for Datadog integration
| _DATADOG_BASE_URL | (Alternative to DD_BASE_URL) Base URL for Datadog integration
| DD_AGENT_HOST | Hostname or IP of DataDog agent (e.g., "localhost"). When set, logs are sent to agent instead of direct API
| DD_AGENT_PORT | Port of DataDog agent for log intake. Default is 10518
| DD_API_KEY | API key for Datadog integration
| DD_SITE | Site URL for Datadog (e.g., datadoghq.com)
| DD_SOURCE | Source identifier for Datadog logs
@ -470,6 +480,7 @@ router_settings:
| DEFAULT_FAILURE_THRESHOLD_PERCENT | Threshold percentage of failures to cool down a deployment. Default is 0.5 (50%)
| DEFAULT_FLUSH_INTERVAL_SECONDS | Default interval in seconds for flushing operations. Default is 5
| DEFAULT_HEALTH_CHECK_INTERVAL | Default interval in seconds for health checks. Default is 300 (5 minutes)
| DEFAULT_HEALTH_CHECK_PROMPT | Default prompt used during health checks for non-image models. Default is "test from litellm"
| DEFAULT_IMAGE_HEIGHT | Default height for images. Default is 300
| DEFAULT_IMAGE_TOKEN_COUNT | Default token count for images. Default is 250
| DEFAULT_IMAGE_WIDTH | Default width for images. Default is 300
@ -496,6 +507,7 @@ router_settings:
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH | Default minimal reasoning effort thinking budget for Gemini 2.5 Flash. Default is 512
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE | Default minimal reasoning effort thinking budget for Gemini 2.5 Flash Lite. Default is 512
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO | Default minimal reasoning effort thinking budget for Gemini 2.5 Pro. Default is 512
| DEFAULT_REDIS_MAJOR_VERSION | Default Redis major version to assume when version cannot be determined. Default is 7
| DEFAULT_REDIS_SYNC_INTERVAL | Default Redis synchronization interval in seconds. Default is 1
| DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND | Default price per second for Replicate GPU. Default is 0.001400
| DEFAULT_REPLICATE_POLLING_DELAY_SECONDS | Default delay in seconds for Replicate polling. Default is 1
@ -507,6 +519,7 @@ router_settings:
| DEFAULT_SLACK_ALERTING_THRESHOLD | Default threshold for Slack alerting. Default is 300
| DEFAULT_SOFT_BUDGET | Default soft budget for LiteLLM proxy keys. Default is 50.0
| DEFAULT_TRIM_RATIO | Default ratio of tokens to trim from prompt end. Default is 0.75
| DEFAULT_GOOGLE_VIDEO_DURATION_SECONDS | Default duration for video generation in seconds in google. Default is 8
| DIRECT_URL | Direct URL for service endpoint
| DISABLE_ADMIN_UI | Toggle to disable the admin UI
| DISABLE_AIOHTTP_TRANSPORT | Flag to disable aiohttp transport. When this is set to True, litellm will use httpx instead of aiohttp. **Default is False**
@ -581,9 +594,14 @@ router_settings:
| HEROKU_API_KEY | API key for Heroku services
| HF_API_BASE | Base URL for Hugging Face API
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_APPROLE_MOUNT_PATH | Mount path for AppRole authentication in [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault). Default is "approle"
| HCP_VAULT_APPROLE_ROLE_ID | Role ID for AppRole authentication in [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_APPROLE_SECRET_ID | Secret ID for AppRole authentication in [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_CLIENT_KEY | Path to client key for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_MOUNT_NAME | Mount name for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_NAMESPACE | Namespace for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_PATH_PREFIX | Path prefix for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_TOKEN | Token for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_CERT_ROLE | Role for [Hashicorp Vault Secret Manager Auth](../secret.md#hashicorp-vault)
| HELICONE_API_KEY | API key for Helicone service
@ -650,6 +668,7 @@ router_settings:
| LITELLM_OTEL_INTEGRATION_ENABLE_METRICS | Optionally enable emantic metrics for OTEL
| LITELLM_MASTER_KEY | Master key for proxy authentication
| LITELLM_MODE | Operating mode for LiteLLM (e.g., production, development)
| LITELLM_NON_ROOT | Flag to run LiteLLM in non-root mode for enhanced security in Docker containers
| LITELLM_RATE_LIMIT_WINDOW_SIZE | Rate limit window size for LiteLLM. Default is 60
| LITELLM_SALT_KEY | Salt key for encryption in LiteLLM
| LITELLM_SSL_CIPHERS | SSL/TLS cipher configuration for faster handshakes. Controls cipher suite preferences for OpenSSL connections.

View file

@ -9,7 +9,7 @@ Track spend for keys, users, and teams across 100+ LLMs.
LiteLLM automatically tracks spend for all known models. See our [model cost map](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json)
:::tip Keep Pricing Data Updated
[Sync model pricing data from GitHub](../sync_models_github.md) to ensure accurate cost tracking.
[Sync model pricing data from GitHub](./sync_models_github.md) to ensure accurate cost tracking.
:::
### How to Track Spend with LiteLLM

View file

@ -18,7 +18,7 @@ Send LiteLLM Proxy users emails for specific events.
| Category | Details |
|----------|---------|
| Supported Events | • User added as a user on LiteLLM Proxy<br/>• Proxy API Key created for user |
| Supported Events | • User added as a user on LiteLLM Proxy<br/>• Proxy API Key created for user<br/>• Proxy API Key rotated for user |
| Supported Email Integrations | • Resend API<br/>• SMTP |
## Usage
@ -123,6 +123,35 @@ On the Create Key Modal, Select Advanced Settings > Set Send Email to True.
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
/>
### 3. Proxy API Key Rotated for User
This email is sent when you rotate an API key for a user on LiteLLM Proxy.
<Image
img={require('../../img/email_regen2.png')}
style={{maxHeight: '600px', width: 'auto', display: 'block', margin: '0 0 2rem 0'}}
/>
**How to trigger this event**
On the LiteLLM Proxy UI, go to Virtual Keys > Click on a key > Click "Regenerate Key"
:::info
Ensure there is a `user_id` attached to the key. This would have been set when creating the key.
:::
<Image
img={require('../../img/email_regen.png')}
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
/>
After regenerating the key, the user will receive an email notification with:
- Security-focused messaging about the rotation
- The new API key (or a placeholder if `EMAIL_INCLUDE_API_KEY=false`)
- Instructions to update their applications
- Security best practices
## Email Customization
@ -141,6 +170,8 @@ LiteLLM allows you to customize various aspects of your email notifications. Bel
| Email Signature | `EMAIL_SIGNATURE` | string (HTML) | Standard LiteLLM footer | `"<p>Best regards,<br/>Your Team</p><p><a href='https://your-company.com'>Visit us</a></p>"` | HTML-formatted footer for all emails |
| Invitation Subject | `EMAIL_SUBJECT_INVITATION` | string | "LiteLLM: New User Invitation" | `"Welcome to Your Company!"` | Subject line for invitation emails |
| Key Creation Subject | `EMAIL_SUBJECT_KEY_CREATED` | string | "LiteLLM: API Key Created" | `"Your New API Key is Ready"` | Subject line for key creation emails |
| Key Rotation Subject | `EMAIL_SUBJECT_KEY_ROTATED` | string | "LiteLLM: API Key Rotated" | `"Your API Key Has Been Rotated"` | Subject line for key rotation emails |
| Include API Key | `EMAIL_INCLUDE_API_KEY` | boolean | true | `"false"` | Whether to include the actual API key in emails (set to false for enhanced security) |
| Proxy Base URL | `PROXY_BASE_URL` | string | http://0.0.0.0:4000 | `"https://proxy.your-company.com"` | Base URL for the LiteLLM Proxy (used in email links) |
@ -181,11 +212,44 @@ EMAIL_SIGNATURE="<p>Best regards,<br/>Your Company Team</p><p><a href='https://y
# Email Subject Lines
EMAIL_SUBJECT_INVITATION="Welcome to Your Company!" # Subject for invitation emails
EMAIL_SUBJECT_KEY_CREATED="Your API Key is Ready" # Subject for key creation emails
EMAIL_SUBJECT_KEY_ROTATED="Your API Key Has Been Rotated" # Subject for key rotation emails
# Security Settings
EMAIL_INCLUDE_API_KEY="false" # Set to false to hide API keys in emails (default: true)
# Proxy Configuration
PROXY_BASE_URL="https://proxy.your-company.com" # Base URL for the LiteLLM Proxy (used in email links)
```
## Security: Hiding API Keys in Emails
For enhanced security, you can configure LiteLLM to **not** include actual API keys in email notifications. This is useful when:
- You want to reduce the risk of key exposure via email interception
- Your security policy requires keys to only be retrieved from the secure dashboard
- You're concerned about email forwarding or storage security
When disabled, emails will show: `[Key hidden for security - retrieve from dashboard]` instead of the actual API key.
**Configuration:**
```bash
# Hide API keys in emails (enhanced security)
EMAIL_INCLUDE_API_KEY="false"
# Include API keys in emails (default behavior)
EMAIL_INCLUDE_API_KEY="true" # or omit this variable
```
**Behavior:**
| Setting | Key Created Email | Key Rotated Email |
|---------|------------------|-------------------|
| `true` (default) | Shows actual `sk-xxxxx` key | Shows actual `sk-xxxxx` key |
| `false` | Shows placeholder message | Shows placeholder message |
Users can always retrieve their keys from the LiteLLM Proxy dashboard.
## HTML Support in Email Signature
The `EMAIL_SIGNATURE` environment variable supports HTML formatting, allowing you to create rich, branded email footers. You can include:

View file

@ -0,0 +1,455 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
import Image from '@theme/IdealImage';
# LiteLLM Content Filter
**Built-in guardrail** for detecting and filtering sensitive information using regex patterns and keyword matching. No external dependencies required.
## Overview
| Property | Details |
|----------|---------|
| Description | On-device guardrail for detecting and filtering sensitive information using regex patterns and keyword matching. Built into LiteLLM with no external dependencies. |
| Guardrail Name | `litellm_content_filter` |
| Detection Methods | Prebuilt regex patterns, custom regex, keyword matching |
| Actions | `BLOCK` (reject request), `MASK` (redact content) |
| Supported Modes | `pre_call`, `post_call`, `during_call` (streaming) |
| Performance | Fast - runs locally, no external API calls |
## Quick Start
## LiteLLM UI
### Step 1: Select LiteLLM Content Filter
Click "Add New Guardrail" and select "LiteLLM Content Filter" as your guardrail provider.
<Image img={require('../../../img/create_guard.gif')} alt="Select LiteLLM Content Filter" />
### Step 2: Configure Pattern Detection
Select the prebuilt entities you want to block or mask. In this example, we select "Email" to detect and block email addresses.
If you need to block a custom entity, you can add a custom regex pattern by clicking "Add custom regex".
<Image img={require('../../../img/add_Guard2.gif')} alt="Select prebuilt entities or add custom regex" />
### Step 3: Add Blocked Keywords
Enter specific keywords you want to block. This is useful if you have policies to block certain words or phrases.
<Image img={require('../../../img/create_guard3.gif')} alt="Add blocked keywords" />
### Step 4: Test Your Guardrail
After creating the guardrail, navigate to "Test Playground" to test it. Select the guardrail you just created.
Test examples:
- **Blocked keyword test**: Entering "hi blue" will trigger the block since we set "blue" as a blocked keyword
- **Pattern detection test**: Entering "Hi ishaan@berri.ai" will trigger the email pattern detector
<Image img={require('../../../img/add_guard5.gif')} alt="Test guardrail in playground" />
## LiteLLM Config.yaml Setup
### Step 1: Define Guardrails in config.yaml
```yaml showLineNumbers title="config.yaml"
model_list:
- model_name: gpt-3.5-turbo
litellm_params:
model: openai/gpt-3.5-turbo
api_key: os.environ/OPENAI_API_KEY
guardrails:
- guardrail_name: "content-filter-pre"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
# Prebuilt patterns for common PII
patterns:
- pattern_type: "prebuilt"
pattern_name: "us_ssn"
action: "BLOCK"
- pattern_type: "prebuilt"
pattern_name: "email"
action: "MASK"
# Custom blocked keywords
blocked_words:
- keyword: "confidential"
action: "BLOCK"
description: "Sensitive internal information"
```
### Step 2: Start LiteLLM Gateway
```shell
litellm --config config.yaml
```
### Step 3: Test Request
<Tabs>
<TabItem label="SSN Blocked" value="ssn-blocked">
```shell
curl -i http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"messages": [
{"role": "user", "content": "My SSN is 123-45-6789"}
],
"guardrails": ["content-filter-pre"]
}'
```
**Response: HTTP 400 Error**
```json
{
"error": {
"message": {
"error": "Content blocked: us_ssn pattern detected",
"pattern": "us_ssn"
},
"code": "400"
}
}
```
</TabItem>
<TabItem label="Email Masked" value="email-masked">
```shell
curl -i http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"messages": [
{"role": "user", "content": "Contact me at john@example.com"}
],
"guardrails": ["content-filter-pre"]
}'
```
The request is sent to the LLM with the email masked:
```
Contact me at [EMAIL_REDACTED]
```
</TabItem>
</Tabs>
## Configuration
### Supported Modes
- **`pre_call`** - Run before LLM call, filters input messages
- **`post_call`** - Run after LLM call, filters output responses
- **`during_call`** - Run during streaming, filters each chunk in real-time
### Actions
- **`BLOCK`** - Reject the request with HTTP 400 error
- **`MASK`** - Replace sensitive content with redaction tags (e.g., `[EMAIL_REDACTED]`)
## Prebuilt Patterns
### Available Patterns
| Pattern Name | Description | Example |
|-------------|-------------|---------|
| `us_ssn` | US Social Security Numbers | `123-45-6789` |
| `email` | Email addresses | `user@example.com` |
| `phone` | Phone numbers | `+1-555-123-4567` |
| `visa` | Visa credit cards | `4532-1234-5678-9010` |
| `mastercard` | Mastercard credit cards | `5425-2334-3010-9903` |
| `amex` | American Express cards | `3782-822463-10005` |
| `aws_access_key` | AWS access keys | `AKIAIOSFODNN7EXAMPLE` |
| `aws_secret_key` | AWS secret keys | `wJalrXUtnFEMI/K7MDENG/bPxRfi...` |
| `github_token` | GitHub tokens | `ghp_16C7e42F292c6912E7710c838347Ae178B4a` |
### Using Prebuilt Patterns
```yaml showLineNumbers title="config.yaml"
guardrails:
- guardrail_name: "pii-filter"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
patterns:
- pattern_type: "prebuilt"
pattern_name: "us_ssn"
action: "BLOCK"
- pattern_type: "prebuilt"
pattern_name: "email"
action: "MASK"
- pattern_type: "prebuilt"
pattern_name: "aws_access_key"
action: "BLOCK"
```
## Custom Regex Patterns
Define your own regex patterns for domain-specific sensitive data:
```yaml showLineNumbers title="config.yaml"
guardrails:
- guardrail_name: "custom-patterns"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
patterns:
# Custom employee ID format
- pattern_type: "regex"
pattern: '\b[A-Z]{3}-\d{4}\b'
name: "employee_id"
action: "MASK"
# Custom project code format
- pattern_type: "regex"
pattern: 'PROJECT-\d{6}'
name: "project_code"
action: "BLOCK"
```
## Keyword Filtering
Block or mask specific keywords:
```yaml showLineNumbers title="config.yaml"
guardrails:
- guardrail_name: "keyword-filter"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
blocked_words:
- keyword: "confidential"
action: "BLOCK"
description: "Internal confidential information"
- keyword: "proprietary"
action: "MASK"
description: "Proprietary company data"
- keyword: "secret_project"
action: "BLOCK"
```
### Loading Keywords from File
For large keyword lists, use a YAML file:
```yaml showLineNumbers title="config.yaml"
guardrails:
- guardrail_name: "keyword-file-filter"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
blocked_words_file: "/path/to/sensitive_keywords.yaml"
```
```yaml showLineNumbers title="sensitive_keywords.yaml"
blocked_words:
- keyword: "project_apollo"
action: "BLOCK"
description: "Confidential project codename"
- keyword: "internal_api"
action: "MASK"
description: "Internal API references"
- keyword: "customer_database"
action: "BLOCK"
description: "Protected database name"
```
## Streaming Support
Content filter works with streaming responses by checking each chunk:
```yaml showLineNumbers title="config.yaml"
guardrails:
- guardrail_name: "streaming-filter"
litellm_params:
guardrail: litellm_content_filter
mode: "during_call" # Check each streaming chunk
patterns:
- pattern_type: "prebuilt"
pattern_name: "email"
action: "MASK"
```
```python
import openai
client = openai.OpenAI(
api_key="sk-1234",
base_url="http://localhost:4000"
)
response = client.chat.completions.create(
model="gpt-3.5-turbo",
messages=[{"role": "user", "content": "Tell me about yourself"}],
stream=True,
extra_body={"guardrails": ["streaming-filter"]}
)
for chunk in response:
print(chunk.choices[0].delta.content)
# Emails automatically masked in real-time
```
## Customizing Redaction Tags
When using the `MASK` action, sensitive content is replaced with redaction tags. You can customize how these tags appear.
### Default Behavior
**Patterns:** Each pattern type gets its own tag based on the pattern name
```
Input: "My email is john@example.com and SSN is 123-45-6789"
Output: "My email is [EMAIL_REDACTED] and SSN is [US_SSN_REDACTED]"
```
**Keywords:** All keywords use the same generic tag
```
Input: "This is confidential and proprietary information"
Output: "This is [KEYWORD_REDACTED] and [KEYWORD_REDACTED] information"
```
### Customizing Tags
Use `pattern_redaction_format` and `keyword_redaction_tag` to change the redaction format:
```yaml showLineNumbers title="config.yaml"
guardrails:
- guardrail_name: "custom-redaction"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
pattern_redaction_format: "***{pattern_name}***" # Use {pattern_name} placeholder
keyword_redaction_tag: "***REDACTED***"
patterns:
- pattern_type: "prebuilt"
pattern_name: "email"
action: "MASK"
- pattern_type: "prebuilt"
pattern_name: "us_ssn"
action: "MASK"
blocked_words:
- keyword: "confidential"
action: "MASK"
```
**Output:**
```
Input: "Email john@example.com, SSN 123-45-6789, confidential data"
Output: "Email ***EMAIL***, SSN ***US_SSN***, ***REDACTED*** data"
```
**Key Points:**
- `pattern_redaction_format` must include `{pattern_name}` placeholder
- Pattern names are automatically uppercased (e.g., `email` → `EMAIL`)
- `keyword_redaction_tag` is a fixed string (no placeholders)
## Use Cases
### 1. PII Protection
Block or mask personally identifiable information before sending to LLMs:
```yaml
patterns:
- pattern_type: "prebuilt"
pattern_name: "us_ssn"
action: "BLOCK"
- pattern_type: "prebuilt"
pattern_name: "email"
action: "MASK"
```
### 2. Credential Detection
Prevent API keys and secrets from being exposed:
```yaml
patterns:
- pattern_type: "prebuilt"
pattern_name: "aws_access_key"
action: "BLOCK"
- pattern_type: "prebuilt"
pattern_name: "github_token"
action: "BLOCK"
```
### 3. Sensitive Internal Data Protection
Block or mask references to confidential internal projects, codenames, or proprietary information:
```yaml
blocked_words:
- keyword: "project_titan"
action: "BLOCK"
description: "Confidential project codename"
- keyword: "internal_api"
action: "MASK"
description: "Internal system references"
```
For large lists of sensitive terms, use a file:
```yaml
blocked_words_file: "/path/to/sensitive_terms.yaml"
```
### 4. Compliance
Ensure regulatory compliance by filtering sensitive data types:
```yaml
patterns:
- pattern_type: "prebuilt"
pattern_name: "visa"
action: "BLOCK"
- pattern_type: "prebuilt"
pattern_name: "us_ssn"
action: "BLOCK"
```
## Troubleshooting
### Pattern Not Matching
**Issue:** Regex pattern isn't detecting expected content
**Solution:** Test your regex pattern:
```python
import re
pattern = r'\b[A-Z]{3}-\d{4}\b'
test_text = "Employee ID: ABC-1234"
print(re.search(pattern, test_text)) # Should match
```
### Multiple Pattern Matches
**Issue:** Text contains multiple sensitive patterns
**Solution:** First matching pattern/keyword is processed. Order patterns by priority:
```yaml
patterns:
# Most critical first
- pattern_type: "prebuilt"
pattern_name: "us_ssn"
action: "BLOCK"
# Less critical
- pattern_type: "prebuilt"
pattern_name: "email"
action: "MASK"
```

View file

@ -4,12 +4,12 @@ import TabItem from '@theme/TabItem';
# PANW Prisma AIRS
LiteLLM supports PANW Prisma AIRS (AI Runtime Security) guardrails via the [Prisma AIRS Scan API](https://pan.dev/prisma-airs/api/airuntimesecurity/scan-sync-request/). This integration provides **Security-as-Code** for AI applications using Palo Alto Networks' AI security platform.
LiteLLM supports PANW Prisma AIRS (AI Runtime Security) guardrails via the [Prisma AIRS Scan API](https://pan.dev/prisma-airs/api/airuntimesecurity/airuntimesecurityapi//). This integration provides **Security-as-Code** for AI applications using Palo Alto Networks' AI security platform.
## Features
- ✅ **Real-time prompt injection detection**
- ✅ **Malicious content filtering**
- ✅ **Malicious URL detection**
- ✅ **Data loss prevention (DLP)**
- ✅ **Sensitive content masking** - Automatically mask PII, credit cards, SSNs instead of blocking
- ✅ **Comprehensive threat detection** for AI models and datasets
@ -17,6 +17,7 @@ LiteLLM supports PANW Prisma AIRS (AI Runtime Security) guardrails via the [Pris
- ✅ **Synchronous scanning** with immediate response
- ✅ **Configurable security profiles**
- ✅ **Streaming support** - Real-time masking for streaming responses
- ✅ **Multi-turn conversation tracking** - Automatic session grouping in Prisma AIRS SCM logs
- ✅ **Fail-closed security** - Blocks requests if PANW API is unavailable (maximum security)
## Quick Start
@ -237,6 +238,74 @@ You can override guardrail settings on a per-request basis using the `metadata`
- **Note:** If your API key is not linked to a profile, you must provide `profile_name` or `profile_id`
:::
## Multi-Turn Conversation Tracking
PANW Prisma AIRS automatically tracks multi-turn conversations using LiteLLM's `litellm_trace_id`. This enables you to:
- **Group related requests** - All requests in a conversation share the same AI Session ID in Prisma AIRS SCM logs
- **Track conversation context** - See the full history of prompts and responses for a user session
- **Analyze attack patterns** - Identify sophisticated multi-turn attacks across conversation history
### How It Works
LiteLLM automatically generates a unique `litellm_trace_id` for each conversation session. The PANW guardrail uses this as the PANW transaction ID (which maps to "AI Session ID" in Strata Cloud Manager):
```
Conversation Session: litellm_trace_id = "abc-123-def-456"
Turn 1 (User): "What's the capital of France?"
→ Scan ID: scan_001 | Prisma AIRS AI Session ID: abc-123-def-456
Turn 2 (Assistant): "Paris is the capital of France."
→ Scan ID: scan_002 | Prisma AIRS AI Session ID: abc-123-def-456
Turn 3 (User): "What's the population?"
→ Scan ID: scan_003 | Prisma AIRS AI Session ID: abc-123-def-456
Turn 4 (Assistant): "Paris has approximately 2.1 million residents."
→ Scan ID: scan_004 | Prisma AIRS AI Session ID: abc-123-def-456
```
All scans appear under the same AI Session ID in Prisma AIRS logs, making it easy to:
- Review complete conversation history (all 4 turns grouped together)
- Identify patterns across multiple turns
- Correlate security events within a session
- Track the flow of user prompts and AI responses
### Session Tracking
LiteLLM automatically generates a unique `litellm_trace_id` for each request, which the PANW guardrail uses as the AI Session ID in Strata Cloud Manager. All prompt and response scans for a request are automatically grouped under the same session.
#### Custom Session IDs (Per-App Tracking)
You can provide your own `litellm_trace_id` to track sessions on a per-app or per-conversation basis:
```bash
curl -X POST http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"messages": [{"role": "user", "content": "capital of France"}],
"litellm_trace_id": "my-app-session-123", # Custom AI Session ID
"metadata": {
"profile_name": "dev-allow-all-profile", # Override security profile
"user_ip": "192.168.1.1", # Track user IP
"app_name": "eng" # Custom app identifier
},
"guardrails": ["panw-prisma-airs-pre-guard", "panw-prisma-airs-post-guard"]
}'
```
**Result in PANW SCM:**
- AI Session ID: `my-app-session-123`
- All prompt and response scans will be grouped under this custom session ID
- Perfect for tracking multi-turn conversations or per-application sessions
:::tip Viewing Sessions in Prisma AIRS SCM Logs
In Strata Cloud Manager, navigate to **AI Runtime > Sessions** to view all AI Session IDs and their associated scans. Click on a session to see the complete conversation history with security analysis.
:::
## Environment Variables
```bash

View file

@ -0,0 +1,46 @@
import Image from '@theme/IdealImage';
# Guardrail Testing Playground
Test and compare multiple guardrails in real-time with an interactive playground interface.
<Image img={require('../../../img/guardrail_playground.png')} alt="Guardrail Test Playground" />
## How to Use the Guardrail Testing Playground
The Guardrail Testing Playground allows you to quickly test and compare the behavior of different guardrails with sample inputs.
### Steps to Test Guardrails
1. **Navigate to the Guardrails Section**
- Open the LiteLLM Admin UI
- Go to the **Guardrails** section
2. **Open Test Playground**
- Click on the **Test Playground** tab at the top of the page
3. **Select Guardrails to Test**
- Check the guardrails you want to compare
- You can select multiple guardrails to see how they each respond to the same input
4. **Enter Your Input**
- Type or paste your test input in the text area
- This could be a prompt, message, or any text you want to validate against the guardrails
5. **Run the Test**
- Click the **Test guardrails** button (or press Enter)
6. **View Results**
- See the output from each selected guardrail
- Compare how different guardrails handle the same input
- Results will show whether the input passed or was blocked by each guardrail
## Use Cases
This is ideal for **Security Teams** & **LiteLLM Admins** evaluating guardrail solutions.
This brings the following benefits for LiteLLM users:
- **Compare guardrail responses**: test the same prompt across multiple providers (Lakera, Noma AI, Bedrock Guardrails, etc.) simultaneously.
- **Validate configurations**: verify your guardrails catch the threats you care about before production deployment.

View file

@ -106,6 +106,13 @@ model_list:
mode: image_generation # 👈 ADD THIS
```
#### Custom Health Check Prompt
By default, health checks use the prompt `"test from litellm"`. You can customize this prompt globally by setting an environment variable, or per-model via config:
```bash
DEFAULT_HEALTH_CHECK_PROMPT="this is a test prompt"
```
### Text Completion Models

View file

@ -1366,14 +1366,12 @@ Your logs should be available on the specified s3 Bucket
### Team Alias Prefix in Object Key
**This is a preview feature**
You can add the team alias to the object key by setting the `team_alias` in the `config.yaml` file. This will prefix the object key with the team alias.
You can add the team alias to the object key by setting the `team_alias` in the `config.yaml` file.
This will prefix the object key with the team alias.
```yaml
litellm_settings:
callbacks: ["s3_v2"]
enable_preview_features: true
s3_callback_params:
s3_bucket_name: logs-bucket-litellm
s3_region_name: us-west-2
@ -1386,6 +1384,28 @@ litellm_settings:
On s3 bucket, you will see the object key as `my-test-path/my-team-alias/...`
### Key Alias Prefix in Object Key
You can add the user api key alias to the s3 object key by enabling s3_use_key_prefix.
```yaml
litellm_settings:
callbacks: ["s3_v2"]
s3_callback_params:
s3_bucket_name: logs-bucket-litellm
s3_region_name: us-west-2
s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
s3_path: my-test-path
s3_endpoint_url: https://s3.amazonaws.com
s3_use_key_prefix: true
```
On s3 bucket, you will see the object key as `my-test-path/my-key-alias/...`
if both team alias and key alias are enabled then the path becomes
`my-test-path/my-team-alias/my-key-alias/...`
## AWS SQS
@ -1432,9 +1452,13 @@ litellm_settings:
# AWS Region for your SQS queue (e.g., us-east-1, eu-central-1, etc.)
# --- Logging Controls ---
sqs_strip_base64_files: true
sqs_strip_base64_files: false
# If true, LiteLLM will remove or redact base64-encoded binary data (e.g., PDFs, images, audio)
# from logged messages to avoid large payloads. SQS has a 1 MB payload size limit.
s3_use_team_prefix: false
# If true, Litellm will add the team alias prefix to s3 path
s3_use_key_prefix: false
# If true, Litellm will add the key alias prefix to s3 path
```

View file

@ -20,7 +20,7 @@ model_list:
Retrieve detailed information about each model listed in the `/model/info` endpoint, including descriptions from the `config.yaml` file, and additional model info (e.g. max tokens, cost per input token, etc.) pulled from the model_info you set and the [litellm model cost map](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json). Sensitive details like API keys are excluded for security purposes.
:::tip Sync Model Data
Keep your model pricing data up to date by [syncing models from GitHub](../sync_models_github.md).
Keep your model pricing data up to date by [syncing models from GitHub](sync_models_github.md).
:::
<Tabs

View file

@ -1,7 +1,7 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# /responses [Beta]
# /responses
LiteLLM provides a BETA endpoint in the spec of [OpenAI's `/responses` API](https://platform.openai.com/docs/api-reference/responses)

View file

@ -0,0 +1,137 @@
# Firecrawl Search
**Get API Key:** [https://firecrawl.dev](https://firecrawl.dev)
## LiteLLM Python SDK
```python showLineNumbers title="Firecrawl Search"
import os
from litellm import search
os.environ["FIRECRAWL_API_KEY"] = "fc-..."
response = search(
query="latest AI developments",
search_provider="firecrawl",
max_results=5
)
```
## LiteLLM AI Gateway
### 1. Setup config.yaml
```yaml showLineNumbers title="config.yaml"
model_list:
- model_name: gpt-4
litellm_params:
model: gpt-4
api_key: os.environ/OPENAI_API_KEY
search_tools:
- search_tool_name: firecrawl-search
litellm_params:
search_provider: firecrawl
api_key: os.environ/FIRECRAWL_API_KEY
```
### 2. Start the proxy
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
### 3. Test the search endpoint
```bash showLineNumbers title="Test Request"
curl http://0.0.0.0:4000/v1/search/firecrawl-search \
-H "Authorization: Bearer sk-1234" \
-H "Content-Type: application/json" \
-d '{
"query": "latest AI developments",
"max_results": 5
}'
```
## Provider-specific Parameters
```python showLineNumbers title="Firecrawl Search with Provider-specific Parameters"
import os
from litellm import search
os.environ["FIRECRAWL_API_KEY"] = "fc-..."
response = search(
query="machine learning research",
search_provider="firecrawl",
max_results=10,
country="US",
# Firecrawl-specific parameters
sources=["web", "news"], # Search multiple sources
categories=[{"type": "github"}, {"type": "research"}], # Filter by categories
tbs="qdr:m", # Time-based search (past month)
location="San Francisco,California,United States", # Geo-targeting
ignoreInvalidURLs=True, # Exclude invalid URLs
scrapeOptions={ # Scraping options for results
"formats": ["markdown"],
"onlyMainContent": True,
"removeBase64Images": True
}
)
```
## Features
Firecrawl combines web search with powerful scraping capabilities:
### Multiple Sources
Search across different sources simultaneously:
- `web` - Web search results (default)
- `images` - Image search results
- `news` - News search results with dates
### Category Filtering
Filter results by specific categories:
- `github` - Search within GitHub repositories, code, issues, and documentation
- `research` - Search academic and research websites (arXiv, Nature, IEEE, PubMed, etc.)
- `pdf` - Search for PDFs
### Time-Based Search
Use the `tbs` parameter to filter by time periods:
- `qdr:h` - Past hour
- `qdr:d` - Past day
- `qdr:w` - Past week
- `qdr:m` - Past month
- `qdr:y` - Past year
### Content Scraping
Firecrawl automatically scrapes full page content for search results when `scrapeOptions` is specified. By default, LiteLLM requests markdown format with main content only.
### Geo-Targeting
Combine `location` and `country` parameters for geo-targeted results:
```python
response = search(
query="restaurants",
search_provider="firecrawl",
country="DE",
location="Berlin,Germany"
)
```
## Supported Query Operators
Firecrawl supports advanced search operators:
| Operator | Functionality | Example |
| ----------- | --------------------------------------------------------- | ------------------------------- |
| "" | Non-fuzzy matches a string of text | "Firecrawl" |
| \- | Excludes certain keywords | \-bad, \-site:example.com |
| site: | Only returns results from a specified website | site:firecrawl.dev |
| inurl: | Only returns results that include a word in the URL | inurl:firecrawl |
| allinurl: | Only returns results that include multiple words in URL | allinurl:git firecrawl |
| intitle: | Only returns results with a word in the title | intitle:Firecrawl |
| allintitle: | Only returns results with multiple words in the title | allintitle:firecrawl playground |
| related: | Only returns results related to a specific domain | related:firecrawl.dev |

View file

@ -2,7 +2,7 @@
| Feature | Supported |
|---------|-----------|
| Supported Providers | `perplexity`, `tavily`, `parallel_ai`, `exa_ai`, `google_pse`, `dataforseo` |
| Supported Providers | `perplexity`, `tavily`, `parallel_ai`, `exa_ai`, `google_pse`, `dataforseo`, `firecrawl`, `searxng` |
| Cost Tracking | ✅ |
| Logging | ✅ |
| Load Balancing | ❌ |
@ -205,7 +205,7 @@ See the [official Perplexity Search documentation](https://docs.perplexity.ai/ap
| Parameter | Type | Required | Description |
|-----------|------|----------|-------------|
| `query` | string or array | Yes | Search query. Can be a single string or array of strings |
| `search_provider` | string | Yes (SDK) | The search provider to use: `"perplexity"`, `"tavily"`, `"parallel_ai"`, `"exa_ai"`, or `"google_pse"` |
| `search_provider` | string | Yes (SDK) | The search provider to use: `"perplexity"`, `"tavily"`, `"parallel_ai"`, `"exa_ai"`, `"google_pse"`, `"dataforseo"`, `"firecrawl"`, or `"searxng"` |
| `search_tool_name` | string | Yes (Proxy) | Name of the search tool configured in `config.yaml` |
| `max_results` | integer | No | Maximum number of results to return (1-20). Default: 10 |
| `search_domain_filter` | array | No | List of domains to filter results (max 20 domains) |
@ -267,6 +267,8 @@ The response follows Perplexity's search format with the following structure:
| Parallel AI | `PARALLEL_AI_API_KEY` | `parallel_ai` |
| Google PSE | `GOOGLE_PSE_API_KEY`, `GOOGLE_PSE_ENGINE_ID` | `google_pse` |
| DataForSEO | `DATAFORSEO_LOGIN`, `DATAFORSEO_PASSWORD` | `dataforseo` |
| Firecrawl | `FIRECRAWL_API_KEY` | `firecrawl` |
| SearXNG | `SEARXNG_API_BASE` (required) | `searxng` |
See the individual provider documentation for detailed setup instructions and provider-specific parameters.

View file

@ -0,0 +1,318 @@
# SearXNG Search
**Open Source:** [https://github.com/searxng/searxng](https://github.com/searxng/searxng)
**Public Instances:** [https://searx.space/](https://searx.space/)
## Overview
SearXNG is a free, open-source metasearch engine that aggregates results from multiple search engines while protecting user privacy. It can be self-hosted or used via public instances.
**Note:** SearXNG returns a fixed number of results per page (~20 by default) and does not support limiting results via the API. The `max_results` parameter is not directly supported by SearXNG.
## LiteLLM Python SDK
```python showLineNumbers title="SearXNG Search"
import os
from litellm import search
# Set your SearXNG instance URL (REQUIRED)
os.environ["SEARXNG_API_BASE"] = "https://serxng-deployment-production.up.railway.app"
response = search(
query="latest AI developments",
search_provider="searxng",
max_results=10
)
```
## LiteLLM AI Gateway
### 1. Setup config.yaml
```yaml showLineNumbers title="config.yaml"
model_list:
- model_name: gpt-4
litellm_params:
model: gpt-4
api_key: os.environ/OPENAI_API_KEY
search_tools:
- search_tool_name: searxng-search
litellm_params:
search_provider: searxng
api_base: https://serxng-deployment-production.up.railway.app
```
### 2. Start the proxy
```bash
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:4000
```
### 3. Test the search endpoint
```bash showLineNumbers title="Test Request"
curl http://0.0.0.0:4000/v1/search/searxng-search \
-H "Authorization: Bearer sk-1234" \
-H "Content-Type: application/json" \
-d '{
"query": "latest AI developments",
"max_results": 10
}'
```
## Provider-specific Parameters
```python showLineNumbers title="SearXNG Search with Provider-specific Parameters"
import os
from litellm import search
# REQUIRED: Set your SearXNG instance URL
os.environ["SEARXNG_API_BASE"] = "https://serxng-deployment-production.up.railway.app"
response = search(
query="machine learning research",
search_provider="searxng",
max_results=10,
# SearXNG-specific parameters
categories="general,science", # Comma-separated categories
engines="google,duckduckgo,bing", # Comma-separated engines
language="en", # Language code
pageno=1, # Page number
time_range="month" # Time filter: day, month, year
)
```
## Features
SearXNG provides powerful metasearch capabilities:
### Multiple Search Engines
Aggregate results from multiple search engines simultaneously:
- Google, DuckDuckGo, Bing, Brave
- Wikipedia, Startpage
- And many more
### Categories
Search within specific categories:
- `general` - General web search
- `science` - Scientific articles and papers
- `images` - Image search
- `news` - News articles
- `videos` - Video content
- `music` - Music and audio
- `files` - File search
- `it` - IT and technology
- `map` - Maps and location
### Time-Based Filtering
Filter results by time range:
- `day` - Past day
- `month` - Past month
- `year` - Past year
### Privacy-Focused
- No user tracking
- No cookies required
- No profiling
- No ads
### Language Support
Support for 60+ languages with the `language` parameter.
## Self-Hosting
SearXNG can be self-hosted for complete control.
### Quick Deploy
Use our pre-configured deployment repository for easy setup:
**[Fork and Deploy: github.com/BerriAI/serxng-deployment](https://github.com/BerriAI/serxng-deployment)**
This repository includes:
- Docker and Docker Compose setup
- JSON API format pre-configured
- Ready to deploy
### Manual Installation
See the [official SearXNG installation instructions](https://docs.searxng.org/admin/installation.html) for detailed setup.
**Important:** When you install SearXNG, the only active output format by default is the HTML format. You need to activate the JSON format to use the API.
Add the following to your `settings.yml` file:
```yaml
search:
formats:
- html
- json
```
Then restart SearXNG:
```bash
# Using Docker
docker run -d -p 8080:8080 \
-v $(pwd)/settings.yml:/etc/searxng/settings.yml:ro \
-e SEARXNG_BASE_URL=http://localhost:8080 \
searxng/searxng
# Then configure LiteLLM to use your instance
export SEARXNG_API_BASE=http://localhost:8080
```
## Configuration
### Setting API Base URL (Required)
You **must** specify a SearXNG instance URL either via environment variable or in the search call:
```python
# Option 1: Environment variable (Recommended)
import os
os.environ["SEARXNG_API_BASE"] = "https://your-instance.com"
response = search(
query="AI developments",
search_provider="searxng"
)
# Option 2: Pass directly in search call
response = search(
query="AI developments",
search_provider="searxng",
api_base="https://your-instance.com"
)
```
**Note:** There is no default instance URL. You must choose either a [public instance](https://searx.space/) or self-host your own.
### Optional Authentication
Some SearXNG instances may require authentication:
```python
import os
# Set API key if required
os.environ["SEARXNG_API_KEY"] = "your-api-key"
response = search(
query="AI developments",
search_provider="searxng"
)
```
## Cost
SearXNG is completely free:
- **Open source** - No licensing costs
- **Self-hosted** - Only infrastructure costs
- **Public instances** - Usually free, check instance policies
## Advanced Usage
### Custom Engine Selection
```python
response = search(
query="Python tutorials",
search_provider="searxng",
engines="stackoverflow,github,reddit", # Only search these engines
categories="it"
)
```
### Multi-Category Search
```python
response = search(
query="climate change",
search_provider="searxng",
categories="general,science,news", # Search multiple categories
time_range="month"
)
```
### Pagination
```python
# Get page 1
page1 = search(
query="AI research",
search_provider="searxng",
pageno=1
)
# Get page 2
page2 = search(
query="AI research",
search_provider="searxng",
pageno=2
)
```
## Response Format
SearXNG returns results in the standard LiteLLM search format:
```json
{
"object": "search",
"results": [
{
"title": "Example Result",
"url": "https://example.com",
"snippet": "This is the content snippet from the search result...",
"date": "2024-01-15",
"last_updated": null
}
]
}
```
## Troubleshooting
### Test Your Instance First
If LiteLLM with searxng search provider is not working, test your SearXNG instance directly with curl:
```bash
# Test if JSON API is working
curl -s "https://your-searxng-instance.com/search?q=test&format=json" | head -50
# Example with specific instance
curl -s "https://serxng-deployment-production.up.railway.app/search?q=test&format=json" | head -50
```
**Expected response**: JSON with search results
**If you get HTML**: JSON format is not enabled in the instance's `settings.yml`
### No Results
If you get no results:
1. **Try different engines**: Specify `engines` parameter
2. **Broaden categories**: Use multiple categories
3. **Adjust language**: Set appropriate `language` parameter
### JSON Format Not Enabled
If you get HTML instead of JSON:
1. **Test with curl**: Use the curl command above to verify JSON output
2. **Self-host your own instance**: Use [our deployment repo](https://github.com/BerriAI/serxng-deployment) with JSON pre-configured
3. **Check instance configuration**: Not all public instances have JSON enabled
4. **Enable JSON manually**: Add to `settings.yml`:
```yaml
search:
formats:
- html
- json
```

View file

@ -1,8 +1,4 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
import Image from '@theme/IdealImage';
# Secret Manager
# Secret Managers Overview
:::info
@ -14,355 +10,19 @@ import Image from '@theme/IdealImage';
:::
LiteLLM supports **reading secrets (eg. `OPENAI_API_KEY`)** and **writing secrets (eg. Virtual Keys)** from Azure Key Vault, Google Secret Manager, Hashicorp Vault, and AWS Secret Manager.
LiteLLM supports **reading secrets (eg. `OPENAI_API_KEY`)** and **writing secrets (eg. Virtual Keys)** from Azure Key Vault, Google Secret Manager, Hashicorp Vault, CyberArk Conjur, and AWS Secret Manager.
## Supported Secret Managers
- AWS Key Management Service
- AWS Secret Manager
- [Azure Key Vault](#azure-key-vault)
- [Google Secret Manager](#google-secret-manager)
- Google Key Management Service
- [Hashicorp Vault](#hashicorp-vault)
## AWS Secret Manager
Store your proxy keys in AWS Secret Manager.
| Feature | Support | Description |
|---------|----------|-------------|
| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` |
| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` |
#### Proxy Usage
1. Save AWS Credentials in your environment
```bash
os.environ["AWS_ACCESS_KEY_ID"] = "" # Access key
os.environ["AWS_SECRET_ACCESS_KEY"] = "" # Secret access key
os.environ["AWS_REGION_NAME"] = "" # us-east-1, us-east-2, us-west-1, us-west-2
```
2. Enable AWS Secret Manager in config.
<Tabs>
<TabItem value="read_only" label="Read Keys from AWS Secret Manager">
```yaml
general_settings:
master_key: os.environ/litellm_master_key
key_management_system: "aws_secret_manager" # 👈 KEY CHANGE
key_management_settings:
hosted_keys: ["litellm_master_key"] # 👈 Specify which env keys you stored on AWS
```
</TabItem>
<TabItem value="write_only" label="Write Virtual Keys to AWS Secret Manager">
This will only store virtual keys in AWS Secret Manager. No keys will be read from AWS Secret Manager.
```yaml
general_settings:
key_management_system: "aws_secret_manager" # 👈 KEY CHANGE
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "write_only" # Literal["read_only", "write_only", "read_and_write"]
```
</TabItem>
<TabItem value="read_and_write" label="Read + Write Keys with AWS Secret Manager">
```yaml
general_settings:
master_key: os.environ/litellm_master_key
key_management_system: "aws_secret_manager" # 👈 KEY CHANGE
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"]
hosted_keys: ["litellm_master_key"] # OPTIONAL. Specify which env keys you stored on AWS
```
</TabItem>
</Tabs>
3. Run proxy
```bash
litellm --config /path/to/config.yaml
```
#### Using K/V pairs in 1 AWS Secret
You can read multiple keys from a single AWS Secret using the `primary_secret_name` parameter:
```yaml
general_settings:
key_management_system: "aws_secret_manager"
key_management_settings:
hosted_keys: [
"OPENAI_API_KEY_MODEL_1",
"OPENAI_API_KEY_MODEL_2",
]
primary_secret_name: "litellm_secrets" # 👈 Read multiple keys from one JSON secret
```
The `primary_secret_name` allows you to read multiple keys from a single AWS Secret as a JSON object. For example, the "litellm_secrets" would contain:
```json
{
"OPENAI_API_KEY_MODEL_1": "sk-key1...",
"OPENAI_API_KEY_MODEL_2": "sk-key2..."
}
```
This reduces the number of AWS Secrets you need to manage.
## Hashicorp Vault
| Feature | Support | Description |
|---------|----------|-------------|
| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` |
| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` |
Read secrets from [Hashicorp Vault](https://developer.hashicorp.com/vault/docs/secrets/kv/kv-v2)
**Step 1.** Add Hashicorp Vault details in your environment
LiteLLM supports two methods of authentication:
1. TLS cert authentication - `HCP_VAULT_CLIENT_CERT` and `HCP_VAULT_CLIENT_KEY`
2. Token authentication - `HCP_VAULT_TOKEN`
```bash
HCP_VAULT_ADDR="https://test-cluster-public-vault-0f98180c.e98296b2.z1.hashicorp.cloud:8200"
HCP_VAULT_NAMESPACE="admin"
# Authentication via TLS cert
HCP_VAULT_CLIENT_CERT="path/to/client.pem"
HCP_VAULT_CLIENT_KEY="path/to/client.key"
# OR - Authentication via token
HCP_VAULT_TOKEN="hvs.CAESIG52gL6ljBSdmq*****"
# OPTIONAL
HCP_VAULT_REFRESH_INTERVAL="86400" # defaults to 86400, frequency of cache refresh for Hashicorp Vault
```
**Step 2.** Add to proxy config.yaml
```yaml
general_settings:
key_management_system: "hashicorp_vault"
# [OPTIONAL SETTINGS]
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"]
```
**Step 3.** Start + test proxy
```
$ litellm --config /path/to/config.yaml
```
[Quick Test Proxy](./proxy/user_keys)
#### How it works
**Reading Secrets**
LiteLLM reads secrets from Hashicorp Vault's KV v2 engine using the following URL format:
```
{VAULT_ADDR}/v1/{NAMESPACE}/secret/data/{SECRET_NAME}
```
For example, if you have:
- `HCP_VAULT_ADDR="https://vault.example.com:8200"`
- `HCP_VAULT_NAMESPACE="admin"`
- Secret name: `AZURE_API_KEY`
LiteLLM will look up:
```
https://vault.example.com:8200/v1/admin/secret/data/AZURE_API_KEY
```
#### Expected Secret Format
LiteLLM expects all secrets to be stored as a JSON object with a `key` field containing the secret value.
For example, for `AZURE_API_KEY`, the secret should be stored as:
```json
{
"key": "sk-1234"
}
```
<Image img={require('../img/hcorp.png')} />
**Writing Secrets**
When a Virtual Key is Created / Deleted on LiteLLM, LiteLLM will automatically create / delete the secret in Hashicorp Vault.
- Create Virtual Key on LiteLLM either through the LiteLLM Admin UI or API
<Image img={require('../img/hcorp_create_virtual_key.png')} />
- Check Hashicorp Vault for secret
LiteLLM stores secret under the `prefix_for_stored_virtual_keys` path (default: `litellm/`)
<Image img={require('../img/hcorp_virtual_key.png')} />
## Azure Key Vault
#### Usage with LiteLLM Proxy Server
1. Install Proxy dependencies
```bash
pip install 'litellm[proxy]' 'litellm[extra_proxy]'
```
2. Save Azure details in your environment
```bash
export["AZURE_CLIENT_ID"]="your-azure-app-client-id"
export["AZURE_CLIENT_SECRET"]="your-azure-app-client-secret"
export["AZURE_TENANT_ID"]="your-azure-tenant-id"
export["AZURE_KEY_VAULT_URI"]="your-azure-key-vault-uri"
```
3. Add to proxy config.yaml
```yaml
model_list:
- model_name: "my-azure-models" # model alias
litellm_params:
model: "azure/<your-deployment-name>"
api_key: "os.environ/AZURE-API-KEY" # reads from key vault - get_secret("AZURE_API_KEY")
api_base: "os.environ/AZURE-API-BASE" # reads from key vault - get_secret("AZURE_API_BASE")
general_settings:
key_management_system: "azure_key_vault"
```
You can now test this by starting your proxy:
```bash
litellm --config /path/to/config.yaml
```
[Quick Test Proxy](./proxy/quick_start#using-litellm-proxy---curl-request-openai-package-langchain-langchain-js)
## Google Secret Manager
Support for [Google Secret Manager](https://cloud.google.com/security/products/secret-manager)
1. Save Google Secret Manager details in your environment
```shell
GOOGLE_SECRET_MANAGER_PROJECT_ID="your-project-id-on-gcp" # example: adroit-crow-413218
```
Optional Params
```shell
export GOOGLE_SECRET_MANAGER_REFRESH_INTERVAL = "" # (int) defaults to 86400
export GOOGLE_SECRET_MANAGER_ALWAYS_READ_SECRET_MANAGER = "" # (str) set to "true" if you want to always read from google secret manager without using in memory caching. NOT RECOMMENDED in PROD
```
2. Add to proxy config.yaml
```yaml
model_list:
- model_name: fake-openai-endpoint
litellm_params:
model: openai/fake
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_key: os.environ/OPENAI_API_KEY # this will be read from Google Secret Manager
general_settings:
key_management_system: "google_secret_manager"
```
You can now test this by starting your proxy:
```bash
litellm --config /path/to/config.yaml
```
[Quick Test Proxy](./proxy/quick_start#using-litellm-proxy---curl-request-openai-package-langchain-langchain-js)
## Google Key Management Service
Use encrypted keys from Google KMS on the proxy
Step 1. Add keys to env
```
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/credentials.json"
export GOOGLE_KMS_RESOURCE_NAME="projects/*/locations/*/keyRings/*/cryptoKeys/*"
export PROXY_DATABASE_URL_ENCRYPTED=b'\n$\x00D\xac\xb4/\x8e\xc...'
```
Step 2: Update Config
```yaml
general_settings:
key_management_system: "google_kms"
database_url: "os.environ/PROXY_DATABASE_URL_ENCRYPTED"
master_key: sk-1234
```
Step 3: Start + test proxy
```
$ litellm --config /path/to/config.yaml
```
And in another terminal
```
$ litellm --test
```
[Quick Test Proxy](./proxy/user_keys)
<!--
## .env Files
If no secret manager client is specified, Litellm automatically uses the `.env` file to manage sensitive data. -->
## AWS Key Management V1
:::tip
[BETA] AWS Key Management v2 is on the enterprise tier. Go [here for docs](./proxy/enterprise.md#beta-aws-key-manager---key-decryption)
:::
Use AWS KMS to storing a hashed copy of your Proxy Master Key in the environment.
```bash
export LITELLM_MASTER_KEY="djZ9xjVaZ..." # 👈 ENCRYPTED KEY
export AWS_REGION_NAME="us-west-2"
```
```yaml
general_settings:
key_management_system: "aws_kms"
key_management_settings:
hosted_keys: ["LITELLM_MASTER_KEY"] # 👈 WHICH KEYS ARE STORED ON KMS
```
[**See Decryption Code**](https://github.com/BerriAI/litellm/blob/a2da2a8f168d45648b61279d4795d647d94f90c9/litellm/utils.py#L10182)
## **All Secret Manager Settings**
- [AWS Key Management Service](./secret_managers/aws_kms)
- [AWS Secret Manager](./secret_managers/aws_secret_manager)
- [Azure Key Vault](./secret_managers/azure_key_vault)
- [CyberArk Conjur](./secret_managers/cyberark)
- [Google Secret Manager](./secret_managers/google_secret_manager)
- [Google Key Management Service](./secret_managers/google_kms)
- [Hashicorp Vault](./secret_managers/hashicorp_vault)
## All Secret Manager Settings
All settings related to secret management

View file

@ -0,0 +1,34 @@
# AWS Key Management V1
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
:::tip
[BETA] AWS Key Management v2 is on the enterprise tier. Go [here for docs](../proxy/enterprise.md#beta-aws-key-manager---key-decryption)
:::
Use AWS KMS to storing a hashed copy of your Proxy Master Key in the environment.
```bash
export LITELLM_MASTER_KEY="djZ9xjVaZ..." # 👈 ENCRYPTED KEY
export AWS_REGION_NAME="us-west-2"
```
```yaml
general_settings:
key_management_system: "aws_kms"
key_management_settings:
hosted_keys: ["LITELLM_MASTER_KEY"] # 👈 WHICH KEYS ARE STORED ON KMS
```
[**See Decryption Code**](https://github.com/BerriAI/litellm/blob/a2da2a8f168d45648b61279d4795d647d94f90c9/litellm/utils.py#L10182)

View file

@ -0,0 +1,112 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# AWS Secret Manager
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
Store your proxy keys in AWS Secret Manager.
| Feature | Support | Description |
|---------|----------|-------------|
| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` |
| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` |
## Proxy Usage
1. Save AWS Credentials in your environment
```bash
os.environ["AWS_ACCESS_KEY_ID"] = "" # Access key
os.environ["AWS_SECRET_ACCESS_KEY"] = "" # Secret access key
os.environ["AWS_REGION_NAME"] = "" # us-east-1, us-east-2, us-west-1, us-west-2
```
2. Enable AWS Secret Manager in config.
<Tabs>
<TabItem value="read_only" label="Read Keys from AWS Secret Manager">
```yaml
general_settings:
master_key: os.environ/litellm_master_key
key_management_system: "aws_secret_manager" # 👈 KEY CHANGE
key_management_settings:
hosted_keys: ["litellm_master_key"] # 👈 Specify which env keys you stored on AWS
```
</TabItem>
<TabItem value="write_only" label="Write Virtual Keys to AWS Secret Manager">
This will only store virtual keys in AWS Secret Manager. No keys will be read from AWS Secret Manager.
```yaml
general_settings:
key_management_system: "aws_secret_manager" # 👈 KEY CHANGE
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "write_only" # Literal["read_only", "write_only", "read_and_write"]
description: "litellm virtual key" # OPTIONAL, if set will set this as the description for all virtual keys
tags: # OPTIONAL, if set will set this as the tags for all virtual keys
Environment: "Prod"
Owner: "AI Platform team"
```
</TabItem>
<TabItem value="read_and_write" label="Read + Write Keys with AWS Secret Manager">
```yaml
general_settings:
master_key: os.environ/litellm_master_key
key_management_system: "aws_secret_manager" # 👈 KEY CHANGE
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"]
hosted_keys: ["litellm_master_key"] # OPTIONAL. Specify which env keys you stored on AWS
```
</TabItem>
</Tabs>
3. Run proxy
```bash
litellm --config /path/to/config.yaml
```
## Using K/V pairs in 1 AWS Secret
You can read multiple keys from a single AWS Secret using the `primary_secret_name` parameter:
```yaml
general_settings:
key_management_system: "aws_secret_manager"
key_management_settings:
hosted_keys: [
"OPENAI_API_KEY_MODEL_1",
"OPENAI_API_KEY_MODEL_2",
]
primary_secret_name: "litellm_secrets" # 👈 Read multiple keys from one JSON secret
```
The `primary_secret_name` allows you to read multiple keys from a single AWS Secret as a JSON object. For example, the "litellm_secrets" would contain:
```json
{
"OPENAI_API_KEY_MODEL_1": "sk-key1...",
"OPENAI_API_KEY_MODEL_2": "sk-key2..."
}
```
This reduces the number of AWS Secrets you need to manage.

View file

@ -0,0 +1,47 @@
# Azure Key Vault
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
## Usage with LiteLLM Proxy Server
1. Install Proxy dependencies
```bash
pip install 'litellm[proxy]' 'litellm[extra_proxy]'
```
2. Save Azure details in your environment
```bash
export["AZURE_CLIENT_ID"]="your-azure-app-client-id"
export["AZURE_CLIENT_SECRET"]="your-azure-app-client-secret"
export["AZURE_TENANT_ID"]="your-azure-tenant-id"
export["AZURE_KEY_VAULT_URI"]="your-azure-key-vault-uri"
```
3. Add to proxy config.yaml
```yaml
model_list:
- model_name: "my-azure-models" # model alias
litellm_params:
model: "azure/<your-deployment-name>"
api_key: "os.environ/AZURE-API-KEY" # reads from key vault - get_secret("AZURE_API_KEY")
api_base: "os.environ/AZURE-API-BASE" # reads from key vault - get_secret("AZURE_API_BASE")
general_settings:
key_management_system: "azure_key_vault"
```
You can now test this by starting your proxy:
```bash
litellm --config /path/to/config.yaml
```
[Quick Test Proxy](../proxy/quick_start#using-litellm-proxy---curl-request-openai-package-langchain-langchain-js)

View file

@ -0,0 +1,252 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Custom Secret Manager
Integrate your custom secret management system with LiteLLM.
## Quick Start
### 1. Create Your Secret Manager Class
Create a new file `my_secret_manager.py` with an in-memory secret store:
```python showLineNumbers title="my_secret_manager.py"
from typing import Optional, Union
import httpx
from litellm.integrations.custom_secret_manager import CustomSecretManager
class InMemorySecretManager(CustomSecretManager):
def __init__(self):
super().__init__(secret_manager_name="in_memory_secrets")
# Store your secrets in memory
self.secrets = {
"OPENAI_API_KEY": "sk-...",
"ANTHROPIC_API_KEY": "sk-ant-...",
}
async def async_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""Read secret asynchronously"""
return self.secrets.get(secret_name)
def sync_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""Read secret synchronously"""
return self.secrets.get(secret_name)
```
### 2. Configure Proxy
Reference your custom secret manager in `config.yaml`:
```yaml showLineNumbers title="config.yaml"
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
key_management_system: custom # 👈 KEY CHANGE
key_management_settings:
custom_secret_manager: my_secret_manager.InMemorySecretManager # 👈 KEY CHANGE
model_list:
- model_name: gpt-4
litellm_params:
model: openai/gpt-4
api_key: os.environ/OPENAI_API_KEY # Read from custom secret manager
```
### 3. Start LiteLLM Proxy
<Tabs>
<TabItem value="docker" label="Docker">
Mount your custom secret manager file on the container:
```bash showLineNumbers
docker run -d \
-p 4000:4000 \
-e LITELLM_MASTER_KEY=$LITELLM_MASTER_KEY \
--name litellm-proxy \
-v $(pwd)/config.yaml:/app/config.yaml \
-v $(pwd)/my_secret_manager.py:/app/my_secret_manager.py \
ghcr.io/berriai/litellm:main-latest \
--config /app/config.yaml \
--port 4000 \
--detailed_debug
```
</TabItem>
<TabItem value="pip" label="Python Package">
```bash
litellm --config config.yaml --detailed_debug
```
</TabItem>
</Tabs>
## Configuration Options
Customize secret manager behavior in your `config.yaml`:
<Tabs>
<TabItem value="read_only" label="Read Keys Only">
```yaml showLineNumbers title="config.yaml"
general_settings:
key_management_system: custom
key_management_settings:
custom_secret_manager: my_secret_manager.InMemorySecretManager
hosted_keys: ["OPENAI_API_KEY", "ANTHROPIC_API_KEY"] # Only check these keys
```
</TabItem>
<TabItem value="write_only" label="Store Virtual Keys">
Store LiteLLM proxy virtual keys in your secret manager:
```yaml showLineNumbers title="config.yaml"
general_settings:
key_management_system: custom
key_management_settings:
custom_secret_manager: my_secret_manager.InMemorySecretManager
access_mode: "write_only"
store_virtual_keys: true
prefix_for_stored_virtual_keys: "litellm/"
description: "LiteLLM virtual key"
tags:
Environment: "Production"
Team: "AI"
```
</TabItem>
<TabItem value="read_and_write" label="Read + Write">
```yaml showLineNumbers title="config.yaml"
general_settings:
key_management_system: custom
key_management_settings:
custom_secret_manager: my_secret_manager.InMemorySecretManager
access_mode: "read_and_write"
hosted_keys: ["OPENAI_API_KEY"]
store_virtual_keys: true
prefix_for_stored_virtual_keys: "litellm/"
```
</TabItem>
</Tabs>
### Available Settings
| Setting | Description | Default |
|---------|-------------|---------|
| `custom_secret_manager` | Path to your custom secret manager class | Required |
| `access_mode` | `"read_only"`, `"write_only"`, or `"read_and_write"` | `"read_only"` |
| `hosted_keys` | List of specific keys to check in secret manager | All keys |
| `store_virtual_keys` | Store LiteLLM virtual keys in secret manager | `false` |
| `prefix_for_stored_virtual_keys` | Prefix for stored virtual keys | `"litellm/"` |
| `description` | Description for stored secrets | `None` |
| `tags` | Tags to apply to stored secrets | `None` |
## Required Methods
Your custom secret manager **must** implement these two methods:
### `async_read_secret()`
```python showLineNumbers
async def async_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""
Read a secret asynchronously.
Returns:
Secret value if found, None otherwise
"""
pass
```
### `sync_read_secret()`
```python showLineNumbers
def sync_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""
Read a secret synchronously.
Returns:
Secret value if found, None otherwise
"""
pass
```
## Optional Methods
Implement these for additional functionality:
### `async_write_secret()`
```python showLineNumbers
async def async_write_secret(
self,
secret_name: str,
secret_value: str,
description: Optional[str] = None,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
tags: Optional[Union[dict, list]] = None,
) -> dict:
"""Write a secret to your secret manager"""
pass
```
### `async_delete_secret()`
```python showLineNumbers
async def async_delete_secret(
self,
secret_name: str,
recovery_window_in_days: Optional[int] = 7,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> dict:
"""Delete a secret from your secret manager"""
pass
```
## Use Cases
✅ Proprietary vault systems
✅ Custom authentication (mTLS, OAuth)
✅ Organization-specific security policies
✅ Legacy secret storage systems
✅ Multi-region secret replication
✅ Secret versioning and rotation
✅ Compliance requirements (HIPAA, SOC2)
## Example
See [cookbook/litellm_proxy_server/secret_manager/my_secret_manager.py](https://github.com/BerriAI/litellm/blob/main/cookbook/litellm_proxy_server/secret_manager/my_secret_manager.py) for a complete working example with:
- In-memory secret manager implementation
- Integration with LiteLLM Proxy
- Read, write, and delete operations

View file

@ -0,0 +1,179 @@
# CyberArk Conjur
import Image from '@theme/IdealImage';
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
| Feature | Support | Description |
|---------|----------|-------------|
| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` |
| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` |
| Deleting Secrets | ❌ | Secrets must be removed via policy updates |
Read and write secrets from [CyberArk Conjur](https://www.cyberark.com/products/secrets-management/) (self-hosted secrets manager)
**Step 1.** Add CyberArk Conjur details in your environment
LiteLLM supports two methods of authentication:
1. API key authentication - `CYBERARK_API_KEY` (recommended)
2. Certificate authentication - `CYBERARK_CLIENT_CERT` and `CYBERARK_CLIENT_KEY`
```bash title="Environment Variables" showLineNumbers
CYBERARK_API_BASE="http://your-conjur-instance:8080"
CYBERARK_ACCOUNT="default"
CYBERARK_USERNAME="admin"
# Authentication via API key (recommended)
CYBERARK_API_KEY="your-api-key-here"
# OR - Authentication via certificate
CYBERARK_CLIENT_CERT="path/to/client.pem"
CYBERARK_CLIENT_KEY="path/to/client.key"
# OPTIONAL
CYBERARK_REFRESH_INTERVAL="300" # defaults to 300 seconds (5 minutes), frequency of token refresh
```
**Step 2.** Add to proxy config.yaml
```yaml title="Proxy Config" showLineNumbers
general_settings:
key_management_system: "cyberark"
# [OPTIONAL SETTINGS]
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"]
```
**Step 3.** Start + test proxy
```bash title="Start Proxy" showLineNumbers
$ litellm --config /path/to/config.yaml
```
[Quick Test Proxy](../proxy/user_keys)
## Writing Virtual Keys to CyberArk
When you create a virtual key in the LiteLLM UI, it automatically gets stored in CyberArk Conjur.
**Step 1:** Create a virtual key in the LiteLLM Admin UI
In this example, we create a key named `litellm-cyber-ark-secret-key`:
<Image img={require('../../img/cyberark1.png')} alt="Creating virtual key in LiteLLM UI" />
**Step 2:** Verify the secret exists in CyberArk
You can verify the virtual key was stored in CyberArk by querying the secrets API:
```bash title="Verify Secret in CyberArk" showLineNumbers
TOKEN=$(curl -s -X POST http://0.0.0.0:8080/authn/default/admin/authenticate \
-d "your-api-key" | base64 | tr -d '\n')
curl -H "Authorization: Token token=\"$TOKEN\"" \
"http://0.0.0.0:8080/resources/default/variable" | jq .
```
The response shows `litellm-cyber-ark-secret-key` exists in CyberArk:
<Image img={require('../../img/cyberark2.png')} alt="Virtual key stored in CyberArk API" />
The virtual key is stored with the full path: `default:variable:litellm/litellm-cyber-ark-secret-key`
## How it works
**Authentication**
CyberArk Conjur uses a two-step authentication process:
1. LiteLLM authenticates with your API key to get a session token
2. The session token (base64-encoded) is used for subsequent API requests
3. Tokens expire after ~8 minutes, so LiteLLM caches and refreshes them automatically
**Reading Secrets**
LiteLLM reads secrets from CyberArk Conjur using the following URL format:
```
{CYBERARK_API_BASE}/secrets/{ACCOUNT}/variable/{SECRET_NAME}
```
For example, if you have:
- `CYBERARK_API_BASE="http://conjur.example.com:8080"`
- `CYBERARK_ACCOUNT="default"`
- Secret name: `AZURE_API_KEY`
LiteLLM will look up:
```
http://conjur.example.com:8080/secrets/default/variable/AZURE_API_KEY
```
**Writing Secrets**
When a Virtual Key is created on LiteLLM, the following happens automatically:
1. LiteLLM creates a policy entry to define the variable in Conjur (if it doesn't exist)
2. LiteLLM sets the secret value via the Conjur API
LiteLLM stores secrets under the `prefix_for_stored_virtual_keys` path (default: `litellm/`)
For example, a virtual key would be stored as: `litellm/virtual-key-name`
**Important Notes**
- Variables must be defined in a Conjur policy before setting their values
- LiteLLM automatically creates policy entries when writing new secrets
- Secret names with slashes (e.g., `litellm/key`) are automatically URL-encoded
- Session tokens are cached for 5 minutes by default to minimize API calls
## Troubleshooting
If you're experiencing issues with the LiteLLM integration, first validate that your CyberArk Conjur instance is working correctly. Run these curl commands directly against your CyberArk endpoints to verify connectivity and authentication:
**Step 1: Authenticate and get a token**
Replace `http://conjur.example.com:8080` with your `CYBERARK_API_BASE` and use your actual credentials:
```bash title="Authenticate" showLineNumbers
TOKEN=$(curl -s -X POST http://conjur.example.com:8080/authn/default/admin/authenticate \
-d "your-api-key" | base64 | tr -d '\n')
```
**Step 2: Test reading a secret**
```bash title="Read Secret" showLineNumbers
curl -H "Authorization: Token token=\"$TOKEN\"" \
"http://conjur.example.com:8080/secrets/default/variable/test-secret"
```
**Step 3: Test writing a secret**
```bash title="Write Secret" showLineNumbers
curl -X POST \
-H "Authorization: Token token=\"$TOKEN\"" \
--data "my-secret-value" \
"http://conjur.example.com:8080/secrets/default/variable/test-secret"
```
If these commands work successfully against your CyberArk instance, then CyberArk is functioning correctly and the issue is with your LiteLLM configuration. Check that:
- Your environment variables are correctly set
- The `CYBERARK_API_BASE` URL is accessible from your LiteLLM instance
- Your API key or certificates have the necessary permissions in CyberArk
## Video Walkthrough
This video walks through using CyberArk Conjur as a secret manager with LiteLLM. We create a virtual key in the LiteLLM Admin UI and verify it exists in CyberArk. Then we rotate the secret key and verify it exists in CyberArk.
<iframe width="840" height="500" src="https://www.loom.com/embed/e9892ae6cb9545d1b709b82e8695db91" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>

View file

@ -0,0 +1,43 @@
# Google Key Management Service
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
Use encrypted keys from Google KMS on the proxy
Step 1. Add keys to env
```
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/credentials.json"
export GOOGLE_KMS_RESOURCE_NAME="projects/*/locations/*/keyRings/*/cryptoKeys/*"
export PROXY_DATABASE_URL_ENCRYPTED=b'\n$\x00D\xac\xb4/\x8e\xc...'
```
Step 2: Update Config
```yaml
general_settings:
key_management_system: "google_kms"
database_url: "os.environ/PROXY_DATABASE_URL_ENCRYPTED"
master_key: sk-1234
```
Step 3: Start + test proxy
```
$ litellm --config /path/to/config.yaml
```
And in another terminal
```
$ litellm --test
```
[Quick Test Proxy](../proxy/user_keys)

View file

@ -0,0 +1,47 @@
# Google Secret Manager
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
Support for [Google Secret Manager](https://cloud.google.com/security/products/secret-manager)
1. Save Google Secret Manager details in your environment
```shell
GOOGLE_SECRET_MANAGER_PROJECT_ID="your-project-id-on-gcp" # example: adroit-crow-413218
```
Optional Params
```shell
export GOOGLE_SECRET_MANAGER_REFRESH_INTERVAL = "" # (int) defaults to 86400
export GOOGLE_SECRET_MANAGER_ALWAYS_READ_SECRET_MANAGER = "" # (str) set to "true" if you want to always read from google secret manager without using in memory caching. NOT RECOMMENDED in PROD
```
2. Add to proxy config.yaml
```yaml
model_list:
- model_name: fake-openai-endpoint
litellm_params:
model: openai/fake
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_key: os.environ/OPENAI_API_KEY # this will be read from Google Secret Manager
general_settings:
key_management_system: "google_secret_manager"
```
You can now test this by starting your proxy:
```bash
litellm --config /path/to/config.yaml
```
[Quick Test Proxy](../proxy/quick_start#using-litellm-proxy---curl-request-openai-package-langchain-langchain-js)

View file

@ -0,0 +1,196 @@
import Image from '@theme/IdealImage';
# Hashicorp Vault
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
| Feature | Support | Description |
|---------|----------|-------------|
| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` |
| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` |
| Authentication Methods to Hashicorp Vault | ✅ | AppRole, TLS Certificate, Token |
Read secrets from [Hashicorp Vault](https://developer.hashicorp.com/vault/docs/secrets/kv/kv-v2)
**Step 1.** Add Hashicorp Vault details in your environment
LiteLLM supports three methods of authentication:
1. AppRole authentication (recommended) - `HCP_VAULT_APPROLE_ROLE_ID` and `HCP_VAULT_APPROLE_SECRET_ID`
2. TLS cert authentication - `HCP_VAULT_CLIENT_CERT` and `HCP_VAULT_CLIENT_KEY`
3. Token authentication - `HCP_VAULT_TOKEN`
```bash
HCP_VAULT_ADDR="https://test-cluster-public-vault-0f98180c.e98296b2.z1.hashicorp.cloud:8200"
HCP_VAULT_NAMESPACE="admin"
# Authentication via AppRole (recommended)
HCP_VAULT_APPROLE_ROLE_ID="your-role-id"
HCP_VAULT_APPROLE_SECRET_ID="your-secret-id"
HCP_VAULT_APPROLE_MOUNT_PATH="approle" # OPTIONAL. defaults to "approle"
# OR - Authentication via TLS cert
HCP_VAULT_CLIENT_CERT="path/to/client.pem"
HCP_VAULT_CLIENT_KEY="path/to/client.key"
# OR - Authentication via token
HCP_VAULT_TOKEN="hvs.CAESIG52gL6ljBSdmq*****"
# OPTIONAL
HCP_VAULT_REFRESH_INTERVAL="86400" # defaults to 86400, frequency of cache refresh for Hashicorp Vault
```
**Step 2.** Add to proxy config.yaml
```yaml
general_settings:
key_management_system: "hashicorp_vault"
# [OPTIONAL SETTINGS]
key_management_settings:
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager
access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"]
```
**Step 3.** Start + test proxy
```
$ litellm --config /path/to/config.yaml
```
[Quick Test Proxy](../proxy/user_keys)
## Authentication Methods
LiteLLM supports three authentication methods for Hashicorp Vault, with the following priority:
1. **AppRole** - Recommended for production applications
2. **TLS Certificate** - For certificate-based authentication
3. **Token** - Direct token authentication
### 1. AppRole Authentication
To set up AppRole authentication:
1. Enable AppRole auth in Vault:
```bash
vault auth enable approle
```
2. Create a policy and role for LiteLLM:
```bash
# Create a policy file (litellm-policy.hcl)
path "secret/data/*" {
capabilities = ["create", "read", "update", "delete", "list"]
}
# Apply the policy
vault policy write litellm-policy litellm-policy.hcl
# Create an AppRole
vault write auth/approle/role/litellm \
token_policies="litellm-policy" \
token_ttl=32d \
token_max_ttl=32d
```
3. Get your Role ID and Secret ID:
```bash
# Get Role ID
vault read auth/approle/role/litellm/role-id
# Generate Secret ID
vault write -f auth/approle/role/litellm/secret-id
```
4. Set the environment variables:
```bash
export HCP_VAULT_APPROLE_ROLE_ID="your-role-id"
export HCP_VAULT_APPROLE_SECRET_ID="your-secret-id"
```
### 2. TLS Certificate Authentication
TLS Certificate authentication uses client certificates for mutual TLS authentication with Vault.
**Environment Variables:**
```bash
export HCP_VAULT_CLIENT_CERT="path/to/client.pem"
export HCP_VAULT_CLIENT_KEY="path/to/client.key"
export HCP_VAULT_CERT_ROLE="your-cert-role" # Optional
```
**How it works:**
- LiteLLM uses the client certificate and key for mutual TLS authentication
- Vault validates the certificate and issues a temporary token
- The token is cached for the duration of its lease
### 3. Token Authentication
Direct token authentication uses a static Vault token.
**Environment Variables:**
```bash
export HCP_VAULT_TOKEN="hvs.CAESIG52gL6ljBSdmq*****"
```
## How it works
**Reading Secrets**
LiteLLM reads secrets from Hashicorp Vault's KV v2 engine using the following URL format:
```
{VAULT_ADDR}/v1/{NAMESPACE}/secret/data/{SECRET_NAME}
```
For example, if you have:
- `HCP_VAULT_ADDR="https://vault.example.com:8200"`
- `HCP_VAULT_NAMESPACE="admin"`
- Secret name: `AZURE_API_KEY`
LiteLLM will look up:
```
https://vault.example.com:8200/v1/admin/secret/data/AZURE_API_KEY
```
### Expected Secret Format
LiteLLM expects all secrets to be stored as a JSON object with a `key` field containing the secret value.
For example, for `AZURE_API_KEY`, the secret should be stored as:
```json
{
"key": "sk-1234"
}
```
<Image img={require('../../img/hcorp.png')} />
**Writing Secrets**
When a Virtual Key is Created / Deleted on LiteLLM, LiteLLM will automatically create / delete the secret in Hashicorp Vault.
- Create Virtual Key on LiteLLM either through the LiteLLM Admin UI or API
<Image img={require('../../img/hcorp_create_virtual_key.png')} />
- Check Hashicorp Vault for secret
LiteLLM stores secret under the `prefix_for_stored_virtual_keys` path (default: `litellm/`)
<Image img={require('../../img/hcorp_virtual_key.png')} />

View file

@ -0,0 +1,47 @@
# Secret Managers Overview
:::info
✨ **This is an Enterprise Feature**
[Enterprise Pricing](https://www.litellm.ai/#pricing)
[Contact us here to get a free trial](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
:::
LiteLLM supports **reading secrets (eg. `OPENAI_API_KEY`)** and **writing secrets (eg. Virtual Keys)** from Azure Key Vault, Google Secret Manager, Hashicorp Vault, CyberArk Conjur, and AWS Secret Manager.
## Supported Secret Managers
- [AWS Key Management Service](./aws_kms)
- [AWS Secret Manager](./aws_secret_manager)
- [Azure Key Vault](./azure_key_vault)
- [CyberArk Conjur](./cyberark)
- [Google Secret Manager](./google_secret_manager)
- [Google Key Management Service](./google_kms)
- [Hashicorp Vault](./hashicorp_vault)
## All Secret Manager Settings
All settings related to secret management
```yaml
general_settings:
key_management_system: "aws_secret_manager" # REQUIRED
key_management_settings:
# Storing Virtual Keys Settings
store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager
prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL.I f set, this prefix will be used for stored virtual keys in the secret manager
# Access Mode Settings
access_mode: "write_only" # OPTIONAL. Literal["read_only", "write_only", "read_and_write"]. Defaults to "read_only"
# Hosted Keys Settings
hosted_keys: ["litellm_master_key"] # OPTIONAL. Specify which env keys you stored on AWS
# K/V pairs in 1 AWS Secret Settings
primary_secret_name: "litellm_secrets" # OPTIONAL. Read multiple keys from one JSON secret on AWS Secret Manager
```

View file

@ -9,7 +9,7 @@ Fallbacks | ✅ (Between supported models) |
| Guardrails Support | ✅ Content moderation and safety checks |
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
| Spend Management | ✅ Budget tracking and rate limiting |
| Supported Providers | `openai`, `azure` |
| Supported Providers | `openai`, `azure`, `gemini`, `vertex_ai` |
:::tip
@ -41,8 +41,7 @@ print(f"Initial Status: {response.status}")
# Check status until video is ready
while True:
status_response = video_status(
video_id=response.id,
custom_llm_provider="openai"
video_id=response.id
)
print(f"Current Status: {status_response.status}")
@ -57,8 +56,7 @@ while True:
# Download video content when ready
video_bytes = video_content(
video_id=response.id,
custom_llm_provider="openai"
video_id=response.id
)
# Save to file
@ -88,8 +86,7 @@ async def test_async_video():
# Check status until video is ready
while True:
status_response = await avideo_status(
video_id=response.id,
custom_llm_provider="openai"
video_id=response.id
)
print(f"Current Status: {status_response.status}")
@ -104,8 +101,7 @@ async def test_async_video():
# Download video content when ready
video_bytes = await avideo_content(
video_id=response.id,
custom_llm_provider="openai"
video_id=response.id
)
# Save to file
@ -120,21 +116,27 @@ asyncio.run(test_async_video())
```python
from litellm import video_status
# Check the status of a video generation
status_response = video_status(
video_id="video_1234567890",
custom_llm_provider="openai"
video_id="video_1234567890"
)
print(f"Video Status: {status_response.status}")
print(f"Created At: {status_response.created_at}")
print(f"Model: {status_response.model}")
```
# Possible status values:
# - "queued": Video is in the queue
# - "processing": Video is being generated
# - "completed": Video is ready for download
# - "failed": Video generation failed
### List Videos
For listing videos, you need to specify the provider since there's no video_id to decode from:
```python
from litellm import video_list
# List videos from OpenAI
videos = video_list(custom_llm_provider="openai")
for video in videos:
print(f"Video ID: {video['id']}")
```
### Video Generation with Reference Image
@ -207,7 +209,7 @@ print(f"Video ID: {response.id}")
LiteLLM provides OpenAI API compatible video endpoints for complete video generation workflow:
- `/videos/generations` - Generate new videos
- `/videos` - Generate new videos
- `/videos/remix` - Edit existing videos with reference images
- `/videos/status` - Check video generation status
- `/videos/retrieval` - Download completed videos
@ -227,7 +229,6 @@ model_list:
model: azure/sora-2
api_key: os.environ/AZURE_OPENAI_API_KEY
api_base: os.environ/AZURE_OPENAI_API_BASE
api_version: "2024-02-15-preview"
```
Start litellm
@ -253,31 +254,14 @@ curl --location 'http://localhost:4000/v1/videos' \
Test video status request
```bash
# Using custom-llm-provider header
curl --location 'http://localhost:4000/v1/videos/video_id' \
--header 'Accept: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: azure'
# Or using query parameter
curl --location 'http://localhost:4000/v1/videos/video_id?custom_llm_provider=azure' \
--header 'Accept: application/json' \
curl --location 'http://localhost:4000/v1/videos/{video_id}' \
--header 'x-litellm-api-key: sk-1234'
```
Test video retrieval request
```bash
# Using custom-llm-provider header
curl --location 'http://localhost:4000/v1/videos/video_id/content' \
--header 'Accept: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: openai' \
--output video.mp4
# Or using query parameter
curl --location 'http://localhost:4000/v1/videos/video_id/content?custom_llm_provider=openai' \
--header 'Accept: application/json' \
curl --location 'http://localhost:4000/v1/videos/{video_id}/content' \
--header 'x-litellm-api-key: sk-1234' \
--output video.mp4
```
@ -285,27 +269,27 @@ curl --location 'http://localhost:4000/v1/videos/video_id/content?custom_llm_pro
Test video remix request
```bash
# Using custom_llm_provider in request body
curl --location --request POST 'http://localhost:4000/v1/videos/video_id/remix' \
--header 'Accept: application/json' \
curl --location --request POST 'http://localhost:4000/v1/videos/{video_id}/remix' \
--header 'Content-Type: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--data '{
"prompt": "New remix instructions",
"custom_llm_provider": "azure"
}'
# Or using custom-llm-provider header
curl --location --request POST 'http://localhost:4000/v1/videos/video_id/remix' \
--header 'Accept: application/json' \
--header 'Content-Type: application/json' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: azure' \
--data '{
"prompt": "New remix instructions"
}'
```
Test video list request (requires custom_llm_provider)
```bash
# Note: video_list requires custom_llm_provider since there's no video_id to decode from
curl --location 'http://localhost:4000/v1/videos?custom_llm_provider=openai' \
--header 'x-litellm-api-key: sk-1234'
# Or using header
curl --location 'http://localhost:4000/v1/videos' \
--header 'x-litellm-api-key: sk-1234' \
--header 'custom-llm-provider: azure'
```
Test Azure video generation request
```bash
@ -618,4 +602,6 @@ The response follows OpenAI's video generation format with the following structu
| Provider | Link to Usage |
|-------------|--------------------|
| OpenAI | [Usage](providers/openai/videos) |
| Azure | [Usage](providers/azure/videos) |
| Azure | [Usage](providers/azure/videos) |
| Gemini | [Usage](providers/gemini/videos) |
| Vertex AI | [Usage](providers/vertex_ai/videos) |

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.4 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 5 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.9 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 1.8 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 605 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 980 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 273 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 784 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 603 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 552 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 151 KiB

View file

@ -1,5 +1,5 @@
---
title: "[Preview] v1.79.1-stable - FAL AI Support"
title: "v1.79.1-stable - Guardrail Playground"
slug: "v1-79-1"
date: 2025-11-01T10:00:00
authors:
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
ghcr.io/berriai/litellm:v1.80.0-stable
ghcr.io/berriai/litellm:v1.79.1-stable
```
</TabItem>

View file

@ -0,0 +1,444 @@
---
title: "[Preview] v1.79.3-stable - Built-in Guardrails on AI Gateway"
slug: "v1-79-3"
date: 2025-11-08T10:00:00
authors:
- name: Krrish Dholakia
title: CEO, LiteLLM
url: https://www.linkedin.com/in/krish-d/
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
- name: Ishaan Jaff
title: CTO, LiteLLM
url: https://www.linkedin.com/in/reffajnaahsi/
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
hide_table_of_contents: false
---
import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
## Deploy this version
<Tabs>
<TabItem value="docker" label="Docker">
``` showLineNumbers title="docker run litellm"
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
ghcr.io/berriai/litellm:v1.79.3.rc.1
```
</TabItem>
<TabItem value="pip" label="Pip">
``` showLineNumbers title="pip install litellm"
pip install litellm==1.79.3.rc.1
```
</TabItem>
</Tabs>
---
## Key Highlights
- **LiteLLM Custom Guardrail** - Built-in guardrail with UI configuration support
- **Performance Improvements** - `/responses` API 19× Lower Median Latency
- **Veo3 Video Generation (Vertex AI + Google AI Studio)** - Use OpenAI Video API to generate videos with Vertex AI and Google AI Studio Veo3 models
---
### Built-in Guardrails on AI Gateway
<Image
img={require('../../img/release_notes/built_in_guard.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
<br/>
This release introduces built-in guardrails for LiteLLM AI Gateway, allowing you to enforce protections without depending on an external guardrail API.
- **Blocking Keywords** - Block known sensitive keywords like "litellm", "python", etc.
- **Pattern Detection** - Block known sensitive patterns like emails, Social Security Numbers, API keys, etc.
- **Custom Regex Patterns** - Define custom regex patterns for your specific use case.
Get started with the built-in guardrails on AI Gateway [here](https://docs.litellm.ai/docs/proxy/guardrails/litellm_content_filter).
---
### Performance – `/responses` 19× Lower Median Latency
This update significantly improves `/responses` latency by integrating our internal network management for connection handling, eliminating per-request setup overhead.
#### Results
| Metric | Before | After | Improvement |
|--------|--------|-------|-------------|
| Median latency | 3,600 ms | **190 ms** | **−95% (~19× faster)** |
| p95 latency | 4,300 ms | **280 ms** | −93% |
| p99 latency | 4,600 ms | **590 ms** | −87% |
| Average latency | 3,571 ms | **208 ms** | −94% |
| RPS | 231 | **1,059** | +358% |
#### Test Setup
| Category | Specification |
|----------|---------------|
| **Load Testing** | Locust: 1,000 concurrent users, 500 ramp-up |
| **System** | 4 vCPUs, 8 GB RAM, 4 workers, 4 instances |
| **Database** | PostgreSQL (Redis unused) |
| **Configuration** | [config.yaml](https://gist.github.com/AlexsanderHamir/550791675fd752befcac6a9e44024652) |
| **Load Script** | [no_cache_hits.py](https://gist.github.com/AlexsanderHamir/99d673bf74cdd81fd39f59fa9048f2e8) |
---
## New Models / Updated Models
#### New Model Support
| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Features |
| -------- | ----- | -------------- | ------------------- | -------------------- | -------- |
| Azure | `azure/gpt-5-pro` | 272K | $15.00 | $120.00 | Responses API, reasoning, vision, PDF input |
| Azure | `azure/gpt-image-1-mini` | - | - | - | Image generation - per pixel pricing |
| Azure | `azure/container` | - | - | - | Container API - $0.03/session |
| OpenAI | `openai/container` | - | - | - | Container API - $0.03/session |
| Cohere | `cohere/embed-v4.0` | 128K | $0.12 | - | Embeddings with image input support |
| Gemini | `gemini/gemini-live-2.5-flash-preview-native-audio-09-2025` | 1M | $0.30 | $2.00 | Native audio, vision, web search |
| Vertex AI | `vertex_ai/minimaxai/minimax-m2-maas` | 196K | $0.30 | $1.20 | Function calling, tool choice |
| NVIDIA | `nvidia/nemotron-nano-9b-v2` | - | - | - | Chat completions |
#### OCR Models
| Provider | Model | Cost Per Page | Features |
| -------- | ----- | ------------- | -------- |
| Azure AI | `azure_ai/doc-intelligence/prebuilt-read` | $0.0015 | Document reading |
| Azure AI | `azure_ai/doc-intelligence/prebuilt-layout` | $0.01 | Layout analysis |
| Azure AI | `azure_ai/doc-intelligence/prebuilt-document` | $0.01 | Document processing |
| Vertex AI | `vertex_ai/mistral-ocr-2505` | $0.0005 | OCR processing |
#### Search Models
| Provider | Model | Pricing | Features |
| -------- | ----- | ------- | -------- |
| Firecrawl | `firecrawl/search` | Tiered: $0.00166-$0.0166/query | 10-100 results per query |
| SearXNG | `searxng/search` | Free | Open-source metasearch |
#### Features
- **[Azure](../../docs/providers/azure)**
- Add Azure GPT-5-Pro Responses API support with reasoning capabilities - [PR #16235](https://github.com/BerriAI/litellm/pull/16235)
- Add gpt-image-1-mini pricing for Azure with quality tiers (low/medium/high) - [PR #16182](https://github.com/BerriAI/litellm/pull/16182)
- Add support for returning Azure Content Policy error information when exceptions from Azure OpenAI occur - [PR #16231](https://github.com/BerriAI/litellm/pull/16231)
- Fix Azure GPT-5 incorrectly routed to O-series config (temperature parameter unsupported) - [PR #16246](https://github.com/BerriAI/litellm/pull/16246)
- Fix Azure doesn't accept extra body param - [PR #16116](https://github.com/BerriAI/litellm/pull/16116)
- Fix Azure DALL-E-3 health check content policy violation by using safe default prompt - [PR #16329](https://github.com/BerriAI/litellm/pull/16329)
- **[Bedrock](../../docs/providers/bedrock)**
- Fix empty assistant message handling in AWS Bedrock Converse API to prevent 400 Bad Request errors - [PR #15850](https://github.com/BerriAI/litellm/pull/15850)
- Fix: Filter AWS authentication params from Bedrock InvokeModel request body - [PR #16315](https://github.com/BerriAI/litellm/pull/16315)
- Fix Bedrock proxy adding name to file content, breaks when cache_control in use - [PR #16275](https://github.com/BerriAI/litellm/pull/16275)
- Fix global.anthropic.claude-haiku-4-5-20251001-v1:0 supports_reasoning flag and update pricing - [PR #16263](https://github.com/BerriAI/litellm/pull/16263)
- **[Gemini (Google AI Studio + Vertex AI)](../../docs/providers/gemini)**
- Add gemini live audio model cost in model map - [PR #16183](https://github.com/BerriAI/litellm/pull/16183)
- Fix translation problem with Gemini parallel tool calls - [PR #16194](https://github.com/BerriAI/litellm/pull/16194)
- Fix: Send Gemini API key via x-goog-api-key header with custom api_base - [PR #16085](https://github.com/BerriAI/litellm/pull/16085)
- Fix image_config.aspect_ratio not working for gemini-2.5-flash-image - [PR #15999](https://github.com/BerriAI/litellm/pull/15999)
- Fix Gemini minimal reasoning env overrides disabling thoughts - [PR #16347](https://github.com/BerriAI/litellm/pull/16347)
- Fix cache_read_input_token_cost for gemini-2.5-flash - [PR #16354](https://github.com/BerriAI/litellm/pull/16354)
- **[Anthropic](../../docs/providers/anthropic)**
- Fix Anthropic token counting for VertexAI - [PR #16171](https://github.com/BerriAI/litellm/pull/16171)
- Fix anthropic-adapter: properly translate Anthropic image format to OpenAI - [PR #16202](https://github.com/BerriAI/litellm/pull/16202)
- Enable automated prompt caching message format for Claude on Databricks - [PR #16200](https://github.com/BerriAI/litellm/pull/16200)
- Add support for Anthropic Memory Tool - [PR #16115](https://github.com/BerriAI/litellm/pull/16115)
- Propagate cache creation/read token costs for model info to fix Anthropic long context cost calculations - [PR #16376](https://github.com/BerriAI/litellm/pull/16376)
- **[Vertex AI](../../docs/providers/vertex_ai)**
- Add Vertex MiniMAX m2 model support - [PR #16373](https://github.com/BerriAI/litellm/pull/16373)
- Correctly map 429 Resource Exhausted to RateLimitError - [PR #16363](https://github.com/BerriAI/litellm/pull/16363)
- Add `vertex_credentials` support to `litellm.rerank()` for Vertex AI - [PR #16266](https://github.com/BerriAI/litellm/pull/16266)
- **[Databricks](../../docs/providers/databricks)**
- Fix databricks streaming - [PR #16368](https://github.com/BerriAI/litellm/pull/16368)
- **[Deepgram](../../docs/providers/deepgram)**
- Return the diarized transcript when it's required in the request - [PR #16133](https://github.com/BerriAI/litellm/pull/16133)
- **[Fireworks](../../docs/providers/fireworks_ai)**
- Update Fireworks audio endpoints to new `api.fireworks.ai` domains - [PR #16346](https://github.com/BerriAI/litellm/pull/16346)
- **[Cohere](../../docs/providers/cohere)**
- Add cohere embed-v4.0 model support - [PR #16358](https://github.com/BerriAI/litellm/pull/16358)
- **[Watsonx](../../docs/providers/watsonx)**
- Support `reasoning_effort` for watsonx chat models - [PR #16261](https://github.com/BerriAI/litellm/pull/16261)
- **[OpenAI](../../docs/providers/openai)**
- Remove automatic summary from reasoning_effort transformation - [PR #16210](https://github.com/BerriAI/litellm/pull/16210)
- **[XAI](../../docs/providers/xai)**
- Remove Grok 4 Models Reasoning Effort Parameter - [PR #16265](https://github.com/BerriAI/litellm/pull/16265)
- **[Hosted VLLM](../../docs/providers/vllm)**
- Fix HostedVLLMRerankConfig will not be used - [PR #16352](https://github.com/BerriAI/litellm/pull/16352)
#### New Provider Support
- **[Bedrock Agentcore](../../docs/providers/bedrock)**
- Add Bedrock Agentcore as a provider on LiteLLM Python SDK and LiteLLM AI Gateway - [PR #16252](https://github.com/BerriAI/litellm/pull/16252)
---
## LLM API Endpoints
#### Features
- **[OCR API](../../docs/ocr)**
- Add VertexAI OCR provider support + cost tracking - [PR #16216](https://github.com/BerriAI/litellm/pull/16216)
- Add Azure AI Doc Intelligence OCR support - [PR #16219](https://github.com/BerriAI/litellm/pull/16219)
- **[Search API](../../docs/search)**
- Add firecrawl search API support with tiered pricing - [PR #16257](https://github.com/BerriAI/litellm/pull/16257)
- Add searxng search API provider - [PR #16259](https://github.com/BerriAI/litellm/pull/16259)
- **[Responses API](../../docs/response_api)**
- Support responses API streaming in langfuse otel - [PR #16153](https://github.com/BerriAI/litellm/pull/16153)
- Pass extra_body parameters to provider in Responses API requests - [PR #16320](https://github.com/BerriAI/litellm/pull/16320)
- **[Container API](../../docs/container_api)**
- Add E2E Container API Support - [PR #16136](https://github.com/BerriAI/litellm/pull/16136)
- Update container documentation to be similar to others - [PR #16327](https://github.com/BerriAI/litellm/pull/16327)
- **[Video Generation API](../../docs/video_generation)**
- Add Vertex and Gemini Videos API with Cost Tracking + UI support - [PR #16323](https://github.com/BerriAI/litellm/pull/16323)
- Add `custom_llm_provider` support for video endpoints (non-generation) - [PR #16121](https://github.com/BerriAI/litellm/pull/16121)
- **[Audio API](../../docs/audio)**
- Add gpt-4o-transcribe cost tracking - [PR #16412](https://github.com/BerriAI/litellm/pull/16412)
- **[Vector Stores](../../docs/vector_stores)**
- Milvus - search vector store support + support multi-part form data on passthrough - [PR #16035](https://github.com/BerriAI/litellm/pull/16035)
- Azure AI Vector Stores - support "virtual" indexes + create vector store on passthrough API - [PR #16160](https://github.com/BerriAI/litellm/pull/16160)
- Milvus - Passthrough API support - adds create + read vector store support via passthrough API's - [PR #16170](https://github.com/BerriAI/litellm/pull/16170)
- **[Embeddings API](../../docs/embedding/supported_embedding)**
- Use valid CallTypes enum value in embeddings endpoint - [PR #16328](https://github.com/BerriAI/litellm/pull/16328)
- **[Rerank API](../../docs/rerank)**
- Generalize tiered pricing in generic cost calculator - [PR #16150](https://github.com/BerriAI/litellm/pull/16150)
#### Bugs
- **General**
- Fix index field not populated in streaming mode with n>1 and tool calls - [PR #15962](https://github.com/BerriAI/litellm/pull/15962)
- Pass aws_region_name in litellm_params - [PR #16321](https://github.com/BerriAI/litellm/pull/16321)
- Add `retry-after` header support for errors `502`, `503`, `504` - [PR #16288](https://github.com/BerriAI/litellm/pull/16288)
---
## Management Endpoints / UI
#### Features
- **Virtual Keys**
- UI - Delete Team Member with friction - [PR #16167](https://github.com/BerriAI/litellm/pull/16167)
- UI - Litellm test key audio support - [PR #16251](https://github.com/BerriAI/litellm/pull/16251)
- UI - Test Key Page Revert Model To Single Select - [PR #16390](https://github.com/BerriAI/litellm/pull/16390)
- **Models + Endpoints**
- UI - Add Model Existing Credentials Improvement - [PR #16166](https://github.com/BerriAI/litellm/pull/16166)
- UI - Add Azure AD Token field and Azure API Key optional - [PR #16331](https://github.com/BerriAI/litellm/pull/16331)
- UI - Fixed Label for vLLM in Model Create Flow - [PR #16285](https://github.com/BerriAI/litellm/pull/16285)
- UI - Include Model Access Group Models on Team Models Table - [PR #16298](https://github.com/BerriAI/litellm/pull/16298)
- Fix /model_group/info Returning Entire Model List for SSO Users - [PR #16296](https://github.com/BerriAI/litellm/pull/16296)
- Litellm non root docker Model Hub Table fix - [PR #16282](https://github.com/BerriAI/litellm/pull/16282)
- **Guardrails**
- UI - Fix regression where Guardrail Entity Could not be selected and entity was not displayed - [PR #16165](https://github.com/BerriAI/litellm/pull/16165)
- UI - Guardrail Info Page Show PII Config - [PR #16164](https://github.com/BerriAI/litellm/pull/16164)
- Change guardrail_information to list type - [PR #16127](https://github.com/BerriAI/litellm/pull/16127)
- UI - LiteLLM Guardrail - ensure you can see UI Friendly name for PII Patterns - [PR #16382](https://github.com/BerriAI/litellm/pull/16382)
- UI - Guardrails - LiteLLM Content Filter, Allow Viewing/Editing Content Filter Settings - [PR #16383](https://github.com/BerriAI/litellm/pull/16383)
- UI - Guardrails - allow updating guardrails through UI. Ensure litellm_params actually get updated in memory - [PR #16384](https://github.com/BerriAI/litellm/pull/16384)
- **SSO Settings**
- Support dot notation on ui sso - [PR #16135](https://github.com/BerriAI/litellm/pull/16135)
- UI - Prevent trailing slash in sso proxy base url input - [PR #16244](https://github.com/BerriAI/litellm/pull/16244)
- UI - SSO Proxy Base URL input validation and remove normalizing / - [PR #16332](https://github.com/BerriAI/litellm/pull/16332)
- UI - Surface SSO Create errors on create flow - [PR #16369](https://github.com/BerriAI/litellm/pull/16369)
- **Usage & Analytics**
- UI - Tag Usage Top Model Table View and Label Fix - [PR #16249](https://github.com/BerriAI/litellm/pull/16249)
- UI - Litellm usage date picker - [PR #16264](https://github.com/BerriAI/litellm/pull/16264)
- **Cache Settings**
- UI - Cache Settings Redis Add Semantic Cache Settings - [PR #16398](https://github.com/BerriAI/litellm/pull/16398)
#### Bugs
- **General**
- UI - Remove encoding_format in request for embedding models - [PR #16367](https://github.com/BerriAI/litellm/pull/16367)
- UI - Revert Changes for Test Key Multiple Model Select - [PR #16372](https://github.com/BerriAI/litellm/pull/16372)
- UI - Various Small Issues - [PR #16406](https://github.com/BerriAI/litellm/pull/16406)
---
## AI Integrations
### Logging
- **[Langfuse](../../docs/proxy/logging#langfuse)**
- Fix langfuse input tokens logic for cached tokens - [PR #16203](https://github.com/BerriAI/litellm/pull/16203)
- **[Opik](../../docs/proxy/logging#opik)**
- Fix the bug with not incorrect attachment to existing trace & refactor - [PR #15529](https://github.com/BerriAI/litellm/pull/15529)
- **[S3](../../docs/proxy/logging#s3)**
- S3 logger, add support for ssl_verify when using minio logger - [PR #16211](https://github.com/BerriAI/litellm/pull/16211)
- Strip base64 in s3 - [PR #16157](https://github.com/BerriAI/litellm/pull/16157)
- Add allowing Key based prefix to s3 path - [PR #16237](https://github.com/BerriAI/litellm/pull/16237)
- Add Prometheus metric to track callback logging failures in S3 - [PR #16209](https://github.com/BerriAI/litellm/pull/16209)
- **[OpenTelemetry](../../docs/proxy/logging#opentelemetry)**
- OTEL - Log Cost Breakdown on OTEL Logger - [PR #16334](https://github.com/BerriAI/litellm/pull/16334)
- **[DataDog](../../docs/proxy/logging#datadog)**
- Add DD Agent Host support for `datadog` callback - [PR #16379](https://github.com/BerriAI/litellm/pull/16379)
### Guardrails
- **[Noma](../../docs/proxy/guardrails)**
- Revert Noma Apply Guardrail implementation - [PR #16214](https://github.com/BerriAI/litellm/pull/16214)
- Litellm noma guardrail support images - [PR #16199](https://github.com/BerriAI/litellm/pull/16199)
- **[PANW Prisma AIRS](../../docs/proxy/guardrails)**
- PANW prisma airs guardrail deduplication and enhanced session tracking - [PR #16273](https://github.com/BerriAI/litellm/pull/16273)
- **[LiteLLM Custom Guardrail](../../docs/proxy/guardrails)**
- Add LiteLLM Gateway built in guardrail - [PR #16338](https://github.com/BerriAI/litellm/pull/16338)
- UI - Allow configuring LiteLLM Custom Guardrail - [PR #16339](https://github.com/BerriAI/litellm/pull/16339)
- Bug Fix: Content Filter Guard - [PR #16414](https://github.com/BerriAI/litellm/pull/16414)
### Secret Managers
- **[CyberArk](../../docs/secret_managers)**
- Add CyberArk Secrets Manager Integration - [PR #16278](https://github.com/BerriAI/litellm/pull/16278)
- Cyber Ark - Add Key Rotations support - [PR #16289](https://github.com/BerriAI/litellm/pull/16289)
- **[HashiCorp Vault](../../docs/secret_managers)**
- Add configurable mount name and path prefix for HashiCorp Vault - [PR #16253](https://github.com/BerriAI/litellm/pull/16253)
- Secret Manager - Hashicorp, add auth via approle - [PR #16374](https://github.com/BerriAI/litellm/pull/16374)
- **[AWS Secrets Manager](../../docs/secret_managers)**
- Add tags and descriptions support to aws secrets manager - [PR #16224](https://github.com/BerriAI/litellm/pull/16224)
- **[Custom Secret Manager](../../docs/secret_managers)**
- Add Custom Secret Manager - Allow users to define and write a custom secret manager - [PR #16297](https://github.com/BerriAI/litellm/pull/16297)
- **General**
- Email Notifications - Ensure Users get Key Rotated Email - [PR #16292](https://github.com/BerriAI/litellm/pull/16292)
- Fix verify ssl on sts boto3 - [PR #16313](https://github.com/BerriAI/litellm/pull/16313)
---
## Spend Tracking, Budgets and Rate Limiting
- **Cost Tracking**
- Fix OpenAI Responses API streaming tests usage field names and cost calculation - [PR #16236](https://github.com/BerriAI/litellm/pull/16236)
---
## MCP Gateway
- **Configuration**
- Configure static mcp header - [PR #16179](https://github.com/BerriAI/litellm/pull/16179)
- Persist mcp credentials in db - [PR #16308](https://github.com/BerriAI/litellm/pull/16308)
## Performance / Loadbalancing / Reliability improvements
- **Memory Leak Fixes**
- Resolve memory accumulation caused by Pydantic 2.11+ deprecation warnings - [PR #16110](https://github.com/BerriAI/litellm/pull/16110)
- **Session Management**
- Add shared_session support to responses API - [PR #16260](https://github.com/BerriAI/litellm/pull/16260)
- **Error Handling**
- Gracefully handle connection closed errors during streaming - [PR #16294](https://github.com/BerriAI/litellm/pull/16294)
- Handle None values in daily spend sort key - [PR #16245](https://github.com/BerriAI/litellm/pull/16245)
- **Configuration**
- Remove minimum validation for cache control injection index - [PR #16149](https://github.com/BerriAI/litellm/pull/16149)
- Improve clearing logic - only remove unvisited endpoints - [PR #16400](https://github.com/BerriAI/litellm/pull/16400)
- **Redis**
- Handle float redis_version from AWS ElastiCache Valkey - [PR #16207](https://github.com/BerriAI/litellm/pull/16207)
- **Hooks**
- Add parallel execution handling in during_call_hook - [PR #16279](https://github.com/BerriAI/litellm/pull/16279)
- **Infrastructure**
- Install runtime node for prisma - [PR #16410](https://github.com/BerriAI/litellm/pull/16410)
---
## Documentation Updates
- **Provider Documentation**
- Docs - v1.79.1 - [PR #16163](https://github.com/BerriAI/litellm/pull/16163)
- Fix broken link on model_management.md - [PR #16217](https://github.com/BerriAI/litellm/pull/16217)
- Fix image generation response format - use 'images' array instead of 'image' object - [PR #16378](https://github.com/BerriAI/litellm/pull/16378)
- **General Documentation**
- Add minimum resource requirement for production - [PR #16146](https://github.com/BerriAI/litellm/pull/16146)
- Add benchmark comparison with other AI gateways - [PR #16248](https://github.com/BerriAI/litellm/pull/16248)
- LiteLLM content filter guard documentation - [PR #16413](https://github.com/BerriAI/litellm/pull/16413)
- Fix typo of the word orginal - [PR #16255](https://github.com/BerriAI/litellm/pull/16255)
- **Security**
- Remove tornado test files (including test.key), fixes Python 3.13 security issues - [PR #16342](https://github.com/BerriAI/litellm/pull/16342)
---
## New Contributors
* @steve-gore-snapdocs made their first contribution in [PR #16149](https://github.com/BerriAI/litellm/pull/16149)
* @timbmg made their first contribution in [PR #16120](https://github.com/BerriAI/litellm/pull/16120)
* @Nivg made their first contribution in [PR #16202](https://github.com/BerriAI/litellm/pull/16202)
* @pablobgar made their first contribution in [PR #16194](https://github.com/BerriAI/litellm/pull/16194)
* @AlanPonnachan made their first contribution in [PR #16150](https://github.com/BerriAI/litellm/pull/16150)
* @Chesars made their first contribution in [PR #16236](https://github.com/BerriAI/litellm/pull/16236)
* @bowenliang123 made their first contribution in [PR #16255](https://github.com/BerriAI/litellm/pull/16255)
* @dean-zavad made their first contribution in [PR #16199](https://github.com/BerriAI/litellm/pull/16199)
* @alexkuzmik made their first contribution in [PR #15529](https://github.com/BerriAI/litellm/pull/15529)
* @Granine made their first contribution in [PR #16281](https://github.com/BerriAI/litellm/pull/16281)
* @Oodapow made their first contribution in [PR #16279](https://github.com/BerriAI/litellm/pull/16279)
* @jgoodyear made their first contribution in [PR #16275](https://github.com/BerriAI/litellm/pull/16275)
* @Qanpi made their first contribution in [PR #16321](https://github.com/BerriAI/litellm/pull/16321)
* @ShimonMimoun made their first contribution in [PR #16313](https://github.com/BerriAI/litellm/pull/16313)
* @andriykislitsyn made their first contribution in [PR #16288](https://github.com/BerriAI/litellm/pull/16288)
* @reckless-huang made their first contribution in [PR #16263](https://github.com/BerriAI/litellm/pull/16263)
* @chenmoneygithub made their first contribution in [PR #16368](https://github.com/BerriAI/litellm/pull/16368)
* @stembe-digitalex made their first contribution in [PR #16354](https://github.com/BerriAI/litellm/pull/16354)
* @jfcherng made their first contribution in [PR #16352](https://github.com/BerriAI/litellm/pull/16352)
* @xingyaoww made their first contribution in [PR #16246](https://github.com/BerriAI/litellm/pull/16246)
* @emerzon made their first contribution in [PR #16373](https://github.com/BerriAI/litellm/pull/16373)
* @wwwillchen made their first contribution in [PR #16376](https://github.com/BerriAI/litellm/pull/16376)
* @fabriciojoc made their first contribution in [PR #16203](https://github.com/BerriAI/litellm/pull/16203)
* @jroberts2600 made their first contribution in [PR #16273](https://github.com/BerriAI/litellm/pull/16273)
---
## Full Changelog
**[View complete changelog on GitHub](https://github.com/BerriAI/litellm/compare/v1.79.1-nightly...v1.79.2.rc.1)**

View file

@ -28,9 +28,10 @@ const sidebars = {
},
{
type: "category",
label: "[Beta] Guardrails",
label: "Guardrails",
items: [
"proxy/guardrails/quick_start",
"proxy/guardrails/test_playground",
...[
"adding_provider/adding_guardrail_support",
"proxy/guardrails/aim_security",
@ -41,6 +42,7 @@ const sidebars = {
"proxy/guardrails/ibm_guardrails",
"proxy/guardrails/grayswan",
"proxy/guardrails/lasso_security",
"proxy/guardrails/litellm_content_filter",
"proxy/guardrails/guardrails_ai",
"proxy/guardrails/lakera_ai",
"proxy/guardrails/model_armor",
@ -173,7 +175,6 @@ const sidebars = {
href: "https://litellm-api.up.railway.app/",
},
"proxy/enterprise",
"proxy/management_cli",
{
type: "category",
label: "Authentication",
@ -185,19 +186,9 @@ const sidebars = {
"proxy/cli_sso",
"proxy/custom_auth",
"proxy/ip_address",
"proxy/email",
"proxy/multiple_admins",
],
},
{
type: "category",
label: "Spend Tracking",
items: [
"proxy/cost_tracking",
"proxy/custom_pricing",
"proxy/billing",
],
},
{
type: "category",
label: "Budgets + Rate Limits",
@ -221,6 +212,7 @@ const sidebars = {
"proxy/rules",
]
},
"proxy/management_cli",
{
type: "link",
label: "Load Balancing, Routing, Fallbacks",
@ -233,7 +225,8 @@ const sidebars = {
"proxy/dynamic_logging",
"proxy/logging",
"proxy/logging_spec",
"proxy/team_logging"
"proxy/team_logging",
"proxy/email",
],
},
{
@ -260,10 +253,27 @@ const sidebars = {
type: "category",
label: "Secret Managers",
items: [
"secret",
"secret_managers/overview",
"secret_managers/aws_secret_manager",
"secret_managers/aws_kms",
"secret_managers/azure_key_vault",
"secret_managers/cyberark",
"secret_managers/google_secret_manager",
"secret_managers/google_kms",
"secret_managers/hashicorp_vault",
"secret_managers/custom_secret_manager",
"oidc"
]
},
{
type: "category",
label: "Spend Tracking",
items: [
"proxy/cost_tracking",
"proxy/custom_pricing",
"proxy/billing",
],
},
]
},
{
@ -398,6 +408,8 @@ const sidebars = {
"search/parallel_ai",
"search/google_pse",
"search/dataforseo",
"search/firecrawl",
"search/searxng",
]
},
{
@ -455,6 +467,7 @@ const sidebars = {
items: [
"providers/azure_ai",
"providers/azure_ocr",
"providers/azure_document_intelligence",
"providers/azure_ai_speech",
"providers/azure_ai_img",
"providers/azure_ai_vector_stores",
@ -466,10 +479,12 @@ const sidebars = {
label: "Vertex AI",
items: [
"providers/vertex",
"providers/vertex_ai/videos",
"providers/vertex_partner",
"providers/vertex_self_deployed",
"providers/vertex_image",
"providers/vertex_batch",
"providers/vertex_ocr",
]
},
{
@ -477,6 +492,7 @@ const sidebars = {
label: "Google AI Studio",
items: [
"providers/gemini",
"providers/gemini/videos",
"providers/google_ai_studio/files",
"providers/google_ai_studio/image_gen",
"providers/google_ai_studio/realtime",
@ -492,6 +508,7 @@ const sidebars = {
"providers/bedrock_embedding",
"providers/bedrock_image_gen",
"providers/bedrock_rerank",
"providers/bedrock_agentcore",
"providers/bedrock_agents",
"providers/bedrock_batches",
"providers/bedrock_vector_store",

View file

@ -8,6 +8,8 @@
import os
import sys
from litellm.types.utils import CallTypesLiteral
sys.path.insert(
0, os.path.abspath("../..")
) # Adds the parent directory to the system path
@ -166,16 +168,7 @@ class AporiaGuardrail(CustomGuardrail):
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
call_type: Literal[
"completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"responses",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
):
from litellm.proxy.common_utils.callback_utils import (
add_guardrail_to_applied_guardrails_header,

View file

@ -6,14 +6,13 @@
# +-----------------------------------------------+
# Thank you users! We ❤️ you! - Krrish & Ishaan
from typing import Literal
from fastapi import HTTPException
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.utils import CallTypesLiteral
class _ENTERPRISE_GoogleTextModeration(CustomLogger):
@ -89,16 +88,7 @@ class _ENTERPRISE_GoogleTextModeration(CustomLogger):
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
call_type: Literal[
"completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"responses",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
):
"""
- Calls Google's Text Moderation API

View file

@ -12,7 +12,6 @@ sys.path.insert(
0, os.path.abspath("../..")
) # Adds the parent directory to the system path
import sys
from typing import Literal
from fastapi import HTTPException
@ -20,6 +19,7 @@ import litellm
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.utils import CallTypesLiteral
class _ENTERPRISE_OpenAI_Moderation(CustomLogger):
@ -35,16 +35,7 @@ class _ENTERPRISE_OpenAI_Moderation(CustomLogger):
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
call_type: Literal[
"completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"responses",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
):
text = ""
if "messages" in data and isinstance(data["messages"], list):

View file

@ -23,7 +23,7 @@ import litellm
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.utils import Choices, ModelResponse
from litellm.types.utils import CallTypesLiteral, Choices, ModelResponse
class _ENTERPRISE_LlamaGuard(CustomLogger):
@ -98,16 +98,7 @@ class _ENTERPRISE_LlamaGuard(CustomLogger):
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
call_type: Literal[
"completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"responses",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
):
"""
- Calls the Llama Guard Endpoint

View file

@ -17,6 +17,7 @@ from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.secret_managers.main import get_secret_str
from litellm.types.utils import CallTypesLiteral
from litellm.utils import get_formatted_prompt
@ -120,16 +121,7 @@ class _ENTERPRISE_LLMGuard(CustomLogger):
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
call_type: Literal[
"completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"responses",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
):
"""
- Calls the LLM Guard Endpoint

View file

@ -31,6 +31,7 @@ from litellm.types.integrations.pagerduty import (
PagerDutyRequestBody,
)
from litellm.types.utils import (
CallTypesLiteral,
StandardLoggingPayload,
StandardLoggingPayloadErrorInformation,
)
@ -142,18 +143,7 @@ class PagerDutyAlerting(SlackAlerting):
user_api_key_dict: UserAPIKeyAuth,
cache: DualCache,
data: dict,
call_type: Literal[
"completion",
"text_completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"pass_through_endpoint",
"rerank",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
) -> Optional[Union[Exception, str, dict]]:
"""
Example of detecting hanging requests by waiting a given threshold.

View file

@ -11,6 +11,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
EmailEvent,
EmailParams,
SendKeyCreatedEmailEvent,
SendKeyRotatedEmailEvent,
)
from litellm._logging import verbose_proxy_logger
@ -19,10 +20,14 @@ from litellm.integrations.email_templates.email_footer import EMAIL_FOOTER
from litellm.integrations.email_templates.key_created_email import (
KEY_CREATED_EMAIL_TEMPLATE,
)
from litellm.integrations.email_templates.key_rotated_email import (
KEY_ROTATED_EMAIL_TEMPLATE,
)
from litellm.integrations.email_templates.user_invitation_email import (
USER_INVITATION_EMAIL_TEMPLATE,
)
from litellm.proxy._types import InvitationNew, UserAPIKeyAuth, WebhookEvent
from litellm.secret_managers.main import get_secret_bool
from litellm.types.integrations.slack_alerting import LITELLM_LOGO_URL
@ -32,6 +37,7 @@ class BaseEmailLogger(CustomLogger):
DEFAULT_SUBJECT_TEMPLATES = {
EmailEvent.new_user_invitation: "LiteLLM: {event_message}",
EmailEvent.virtual_key_created: "LiteLLM: {event_message}",
EmailEvent.virtual_key_rotated: "LiteLLM: {event_message}",
}
async def send_user_invitation_email(self, event: WebhookEvent):
@ -83,11 +89,58 @@ class BaseEmailLogger(CustomLogger):
f"send_key_created_email_event: {json.dumps(send_key_created_email_event, indent=4, default=str)}"
)
# Check if API key should be included in email
include_api_key = get_secret_bool(secret_name="EMAIL_INCLUDE_API_KEY", default_value=True)
if include_api_key is None:
include_api_key = True # Default to True if not set
key_token_display = send_key_created_email_event.virtual_key if include_api_key else "[Key hidden for security - retrieve from dashboard]"
email_html_content = KEY_CREATED_EMAIL_TEMPLATE.format(
email_logo_url=email_params.logo_url,
recipient_email=email_params.recipient_email,
key_budget=self._format_key_budget(send_key_created_email_event.max_budget),
key_token=send_key_created_email_event.virtual_key,
key_token=key_token_display,
base_url=email_params.base_url,
email_support_contact=email_params.support_contact,
email_footer=email_params.signature,
)
await self.send_email(
from_email=self.DEFAULT_LITELLM_EMAIL,
to_email=[email_params.recipient_email],
subject=email_params.subject,
html_body=email_html_content,
)
pass
async def send_key_rotated_email(
self, send_key_rotated_email_event: SendKeyRotatedEmailEvent
):
"""
Send email to user after rotating key for the user
"""
email_params = await self._get_email_params(
user_id=send_key_rotated_email_event.user_id,
user_email=send_key_rotated_email_event.user_email,
email_event=EmailEvent.virtual_key_rotated,
event_message=send_key_rotated_email_event.event_message,
)
verbose_proxy_logger.debug(
f"send_key_rotated_email_event: {json.dumps(send_key_rotated_email_event, indent=4, default=str)}"
)
# Check if API key should be included in email
include_api_key = get_secret_bool(secret_name="EMAIL_INCLUDE_API_KEY", default_value=True)
if include_api_key is None:
include_api_key = True # Default to True if not set
key_token_display = send_key_rotated_email_event.virtual_key if include_api_key else "[Key hidden for security - retrieve from dashboard]"
email_html_content = KEY_ROTATED_EMAIL_TEMPLATE.format(
email_logo_url=email_params.logo_url,
recipient_email=email_params.recipient_email,
key_budget=self._format_key_budget(send_key_rotated_email_event.max_budget),
key_token=key_token_display,
base_url=email_params.base_url,
email_support_contact=email_params.support_contact,
email_footer=email_params.signature,
@ -159,6 +212,13 @@ class BaseEmailLogger(CustomLogger):
self.DEFAULT_SUBJECT_TEMPLATES[EmailEvent.virtual_key_created],
"key created subject template"
)
elif email_event == EmailEvent.virtual_key_rotated:
custom_subject_key_rotated = os.getenv("EMAIL_SUBJECT_KEY_ROTATED", None)
subject_template = get_custom_or_default(
custom_subject_key_rotated,
self.DEFAULT_SUBJECT_TEMPLATES[EmailEvent.virtual_key_rotated],
"key rotated subject template"
)
else:
subject_template = "LiteLLM: {event_message}"

View file

@ -298,6 +298,13 @@ class PrometheusLogger(CustomLogger):
self.get_labels_for_metric("litellm_deployment_failed_fallbacks"),
)
# Callback Logging Failure Metrics
self.litellm_callback_logging_failures_metric = self._counter_factory(
name="litellm_callback_logging_failures_metric",
documentation="Total number of failures when emitting logs to callbacks (e.g. s3_v2, langfuse, etc)",
labelnames=["callback_name"],
)
self.litellm_llm_api_failed_requests_metric = self._counter_factory(
name="litellm_llm_api_failed_requests_metric",
documentation="deprecated - use litellm_proxy_failed_requests_metric",
@ -1723,6 +1730,17 @@ class PrometheusLogger(CustomLogger):
litellm_model_name, model_id, api_base, api_provider, exception_status
).inc()
def increment_callback_logging_failure(
self,
callback_name: str,
):
"""
Increment metric when logging to a callback fails (e.g., s3_v2, langfuse, etc.)
"""
self.litellm_callback_logging_failures_metric.labels(
callback_name=callback_name
).inc()
def track_provider_remaining_budget(
self, provider: str, spend: float, budget_limit: float
):

View file

@ -36,6 +36,7 @@ from litellm.types.llms.openai import (
OpenAIFilesPurpose,
)
from litellm.types.utils import (
CallTypesLiteral,
LiteLLMBatch,
LiteLLMFineTuningJob,
LLMResponseTypes,
@ -272,28 +273,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
user_api_key_dict: UserAPIKeyAuth,
cache: DualCache,
data: Dict,
call_type: Literal[
"completion",
"text_completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"pass_through_endpoint",
"rerank",
"acreate_batch",
"aretrieve_batch",
"acreate_file",
"afile_list",
"afile_delete",
"afile_content",
"acreate_fine_tuning_job",
"aretrieve_fine_tuning_job",
"alist_fine_tuning_jobs",
"acancel_fine_tuning_job",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
) -> Union[Exception, str, Dict, None]:
"""
- Detect litellm_proxy/ file_id

View file

@ -1,10 +1,11 @@
import enum
from typing import Dict, List
from typing import Dict, List, Optional
from pydantic import BaseModel, Field
from litellm.proxy._types import WebhookEvent
class EmailParams(BaseModel):
logo_url: str
support_contact: str
@ -22,9 +23,19 @@ class SendKeyCreatedEmailEvent(WebhookEvent):
"""
class SendKeyRotatedEmailEvent(WebhookEvent):
virtual_key: str
key_alias: Optional[str] = None
"""
The virtual key that was rotated
this will be sk-123xxx, since we will be emailing this to the user to start using the new key
"""
class EmailEvent(str, enum.Enum):
virtual_key_created = "Virtual Key Created"
new_user_invitation = "New User Invitation"
virtual_key_rotated = "Virtual Key Rotated"
class EmailEventSettings(BaseModel):
event: EmailEvent
@ -37,8 +48,9 @@ class DefaultEmailSettings(BaseModel):
"""Default settings for email events"""
settings: Dict[EmailEvent, bool] = Field(
default_factory=lambda: {
EmailEvent.virtual_key_created: False, # Off by default
EmailEvent.virtual_key_created: True, # On by default
EmailEvent.new_user_invitation: True, # On by default
EmailEvent.virtual_key_rotated: True, # On by default
}
)
def to_dict(self) -> Dict[str, bool]:

Binary file not shown.

Binary file not shown.

View file

@ -0,0 +1,2 @@
-- AlterTable
ALTER TABLE "LiteLLM_MCPServerTable" ADD COLUMN "static_headers" JSONB DEFAULT '{}';

View file

@ -0,0 +1,2 @@
-- AlterTable
ALTER TABLE "LiteLLM_MCPServerTable" ADD COLUMN "credentials" JSONB DEFAULT '{}';

View file

@ -174,6 +174,7 @@ model LiteLLM_MCPServerTable {
url String?
transport String @default("sse")
auth_type String?
credentials Json? @default("{}")
created_at DateTime? @default(now()) @map("created_at")
created_by String?
updated_at DateTime? @default(now()) @updatedAt @map("updated_at")
@ -182,6 +183,7 @@ model LiteLLM_MCPServerTable {
mcp_access_groups String[]
allowed_tools String[] @default([])
extra_headers String[] @default([])
static_headers Json? @default("{}")
// Health check status
status String? @default("unknown")
last_health_check DateTime?
@ -607,4 +609,4 @@ model LiteLLM_CacheConfig {
cache_settings Json
created_at DateTime @default(now())
updated_at DateTime @updatedAt
}
}

View file

@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm-proxy-extras"
version = "0.4.1"
version = "0.4.3"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
authors = ["BerriAI"]
readme = "README.md"
@ -22,7 +22,7 @@ requires = ["poetry-core"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "0.4.1"
version = "0.4.3"
version_files = [
"pyproject.toml:version",
"../requirements.txt:litellm-proxy-extras==",

View file

@ -4,8 +4,10 @@ import warnings
warnings.filterwarnings("ignore", message=".*conflict with protected namespace.*")
# Suppress Pydantic 2.11+ deprecation warning about accessing model_fields on instances
# This warning can accumulate during streaming and cause memory leaks
warnings.filterwarnings("ignore", message=".*Accessing the.*attribute on the instance is deprecated.*")
### INIT VARIABLES ######################
warnings.filterwarnings(
"ignore", message=".*Accessing the.*attribute on the instance is deprecated.*"
)
### INIT VARIABLES #######################
import threading
import os
from typing import (
@ -32,7 +34,7 @@ from litellm.types.utils import (
all_litellm_params as _litellm_completion_params,
CredentialItem,
PriorityReservationDict,
) # maintain backwards compatibility for root param
) # maintain backwards compatibility for root param.
from litellm._logging import (
set_verbose,
_turn_on_debug,
@ -179,22 +181,22 @@ prometheus_initialize_budget_metrics: Optional[bool] = False
require_auth_for_metrics_endpoint: Optional[bool] = False
argilla_batch_size: Optional[int] = None
datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload.
gcs_pub_sub_use_v1: Optional[bool] = (
False # if you want to use v1 gcs pubsub logged payload
)
generic_api_use_v1: Optional[bool] = (
False # if you want to use v1 generic api logged payload
)
gcs_pub_sub_use_v1: Optional[
bool
] = False # if you want to use v1 gcs pubsub logged payload
generic_api_use_v1: Optional[
bool
] = False # if you want to use v1 generic api logged payload
argilla_transformation_object: Optional[Dict[str, Any]] = None
_async_input_callback: List[Union[str, Callable, CustomLogger]] = (
[]
) # internal variable - async custom callbacks are routed here.
_async_success_callback: List[Union[str, Callable, CustomLogger]] = (
[]
) # internal variable - async custom callbacks are routed here.
_async_failure_callback: List[Union[str, Callable, CustomLogger]] = (
[]
) # internal variable - async custom callbacks are routed here.
_async_input_callback: List[
Union[str, Callable, CustomLogger]
] = [] # internal variable - async custom callbacks are routed here.
_async_success_callback: List[
Union[str, Callable, CustomLogger]
] = [] # internal variable - async custom callbacks are routed here.
_async_failure_callback: List[
Union[str, Callable, CustomLogger]
] = [] # internal variable - async custom callbacks are routed here.
pre_call_rules: List[Callable] = []
post_call_rules: List[Callable] = []
turn_off_message_logging: Optional[bool] = False
@ -202,18 +204,18 @@ log_raw_request_response: bool = False
redact_messages_in_exceptions: Optional[bool] = False
redact_user_api_key_info: Optional[bool] = False
filter_invalid_headers: Optional[bool] = False
add_user_information_to_llm_headers: Optional[bool] = (
None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers
)
add_user_information_to_llm_headers: Optional[
bool
] = None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers
store_audit_logs = False # Enterprise feature, allow users to see audit logs
### end of callbacks #############
email: Optional[str] = (
None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
token: Optional[str] = (
None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
email: Optional[
str
] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
token: Optional[
str
] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
telemetry = True
max_tokens: int = DEFAULT_MAX_TOKENS # OpenAI Defaults
drop_params = bool(os.getenv("LITELLM_DROP_PARAMS", False))
@ -269,9 +271,9 @@ use_client: bool = False
ssl_verify: Union[str, bool] = True
ssl_security_level: Optional[str] = None
ssl_certificate: Optional[str] = None
ssl_ecdh_curve: Optional[str] = (
None # Set to 'X25519' to disable PQC and improve performance
)
ssl_ecdh_curve: Optional[
str
] = None # Set to 'X25519' to disable PQC and improve performance
disable_streaming_logging: bool = False
disable_token_counter: bool = False
disable_add_transform_inline_image_block: bool = False
@ -317,24 +319,20 @@ enable_loadbalancing_on_batch_endpoints: Optional[bool] = None
enable_caching_on_provider_specific_optional_params: bool = (
False # feature-flag for caching on optional params - e.g. 'top_k'
)
caching: bool = (
False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
caching_with_models: bool = (
False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
cache: Optional[Cache] = (
None # cache object <- use this - https://docs.litellm.ai/docs/caching
)
caching: bool = False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
caching_with_models: bool = False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
cache: Optional[
Cache
] = None # cache object <- use this - https://docs.litellm.ai/docs/caching
default_in_memory_ttl: Optional[float] = None
default_redis_ttl: Optional[float] = None
default_redis_batch_cache_expiry: Optional[float] = None
model_alias_map: Dict[str, str] = {}
model_group_settings: Optional["ModelGroupSettings"] = None
max_budget: float = 0.0 # set the max budget across all providers
budget_duration: Optional[str] = (
None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
)
budget_duration: Optional[
str
] = None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
default_soft_budget: float = (
DEFAULT_SOFT_BUDGET # by default all litellm proxy keys have a soft budget of 50.0
)
@ -343,15 +341,11 @@ forward_traceparent_to_llm_provider: bool = False
_current_cost = 0.0 # private variable, used if max budget is set
error_logs: Dict = {}
add_function_to_prompt: bool = (
False # if function calling not supported by api, append function call details to system prompt
)
add_function_to_prompt: bool = False # if function calling not supported by api, append function call details to system prompt
client_session: Optional[httpx.Client] = None
aclient_session: Optional[httpx.AsyncClient] = None
model_fallbacks: Optional[List] = None # Deprecated for 'litellm.fallbacks'
model_cost_map_url: str = (
"https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json"
)
model_cost_map_url: str = "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json"
suppress_debug_info = False
dynamodb_table_name: Optional[str] = None
s3_callback_params: Optional[Dict] = None
@ -381,9 +375,7 @@ prometheus_metrics_config: Optional[List] = None
disable_add_prefix_to_prompt: bool = (
False # used by anthropic, to disable adding prefix to prompt
)
disable_copilot_system_to_assistant: bool = (
False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
)
disable_copilot_system_to_assistant: bool = False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
public_model_groups: Optional[List[str]] = None
public_model_groups_links: Dict[str, str] = {}
#### REQUEST PRIORITIZATION #######
@ -394,17 +386,13 @@ priority_reservation_settings: "PriorityReservationSettings" = (
######## Networking Settings ########
use_aiohttp_transport: bool = (
True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead.
)
use_aiohttp_transport: bool = True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead.
aiohttp_trust_env: bool = False # set to true to use HTTP_ Proxy settings
disable_aiohttp_transport: bool = False # Set this to true to use httpx instead
disable_aiohttp_trust_env: bool = (
False # When False, aiohttp will respect HTTP(S)_PROXY env vars
)
force_ipv4: bool = (
False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6.
)
force_ipv4: bool = False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6.
module_level_aclient = AsyncHTTPHandler(
timeout=request_timeout, client_alias="module level aclient"
)
@ -418,13 +406,13 @@ fallbacks: Optional[List] = None
context_window_fallbacks: Optional[List] = None
content_policy_fallbacks: Optional[List] = None
allowed_fails: int = 3
num_retries_per_request: Optional[int] = (
None # for the request overall (incl. fallbacks + model retries)
)
num_retries_per_request: Optional[
int
] = None # for the request overall (incl. fallbacks + model retries)
####### SECRET MANAGERS #####################
secret_manager_client: Optional[Any] = (
None # list of instantiated key management clients - e.g. azure kv, infisical, etc.
)
secret_manager_client: Optional[
Any
] = None # list of instantiated key management clients - e.g. azure kv, infisical, etc.
_google_kms_resource_name: Optional[str] = None
_key_management_system: Optional[KeyManagementSystem] = None
_key_management_settings: KeyManagementSettings = KeyManagementSettings()
@ -434,9 +422,9 @@ output_parse_pii: bool = False
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
model_cost = get_model_cost_map(url=model_cost_map_url)
cost_discount_config: Dict[str, float] = (
{}
) # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount
cost_discount_config: Dict[
str, float
] = {} # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount
custom_prompt_dict: Dict[str, dict] = {}
check_provider_endpoint = False
@ -492,6 +480,7 @@ vertex_deepseek_models: Set = set()
vertex_ai_ai21_models: Set = set()
vertex_mistral_models: Set = set()
vertex_openai_models: Set = set()
vertex_minimax_models: Set = set()
ai21_models: Set = set()
ai21_chat_models: Set = set()
nlp_cloud_models: Set = set()
@ -652,6 +641,9 @@ def add_known_models():
elif value.get("litellm_provider") == "vertex_ai-openai_models":
key = key.replace("vertex_ai/", "")
vertex_openai_models.add(key)
elif value.get("litellm_provider") == "vertex_ai-minimax_models":
key = key.replace("vertex_ai/", "")
vertex_minimax_models.add(key)
elif value.get("litellm_provider") == "ai21":
if value.get("mode") == "chat":
ai21_chat_models.add(key)
@ -907,7 +899,8 @@ models_by_provider: dict = {
| vertex_anthropic_models
| vertex_vision_models
| vertex_language_models
| vertex_deepseek_models,
| vertex_deepseek_models
| vertex_minimax_models,
"ai21": ai21_models,
"bedrock": bedrock_models | bedrock_converse_models,
"petals": petals_models,
@ -1105,6 +1098,7 @@ from .llms.azure_ai.rerank.transformation import AzureAIRerankConfig
from .llms.infinity.rerank.transformation import InfinityRerankConfig
from .llms.jina_ai.rerank.transformation import JinaAIRerankConfig
from .llms.deepinfra.rerank.transformation import DeepinfraRerankConfig
from .llms.hosted_vllm.rerank.transformation import HostedVLLMRerankConfig
from .llms.nvidia_nim.rerank.transformation import NvidiaNimRerankConfig
from .llms.vertex_ai.rerank.transformation import VertexAIRerankConfig
from .llms.clarifai.chat.transformation import ClarifaiConfig
@ -1236,6 +1230,7 @@ from .llms.azure.responses.transformation import AzureOpenAIResponsesAPIConfig
from .llms.azure.responses.o_series_transformation import (
AzureOpenAIOSeriesResponsesAPIConfig,
)
from .llms.xai.responses.transformation import XAIResponsesAPIConfig
from .llms.litellm_proxy.responses.transformation import (
LiteLLMProxyResponsesAPIConfig,
)
@ -1344,6 +1339,7 @@ from .exceptions import (
NotFoundError,
RateLimitError,
ServiceUnavailableError,
BadGatewayError,
OpenAIError,
ContextWindowExceededError,
ContentPolicyViolationError,
@ -1399,12 +1395,12 @@ from .types.llms.custom_llm import CustomLLMItem
from .types.utils import GenericStreamingChunk
custom_provider_map: List[CustomLLMItem] = []
_custom_providers: List[str] = (
[]
) # internal helper util, used to track names of custom providers
disable_hf_tokenizer_download: Optional[bool] = (
None # disable huggingface tokenizer download. Defaults to openai clk100
)
_custom_providers: List[
str
] = [] # internal helper util, used to track names of custom providers
disable_hf_tokenizer_download: Optional[
bool
] = None # disable huggingface tokenizer download. Defaults to openai clk100
global_disable_no_log_param: bool = False
### CLI UTILITIES ###

View file

@ -18,6 +18,7 @@ from typing import TYPE_CHECKING, Any, List, Optional, Tuple, Union, cast
import litellm
from litellm._logging import print_verbose, verbose_logger
from litellm.constants import DEFAULT_REDIS_MAJOR_VERSION
from litellm.litellm_core_utils.core_helpers import _get_parent_otel_span_from_kwargs
from litellm.litellm_core_utils.coroutine_checker import coroutine_checker
from litellm.types.caching import RedisPipelineIncrementOperation
@ -207,6 +208,35 @@ class RedisCache(BaseCache):
return key
def _parse_redis_major_version(self) -> int:
"""
Parse Redis version to extract the major version number.
Handles multiple version formats:
- Strings: "7.0.0", "6", "7.0.0-rc1", " 7.0.0 "
- Floats: 7.0 (e.g., from AWS ElastiCache Valkey)
- Integers: 7
- Malformed: "latest", "", "Unknown" (defaults to DEFAULT_REDIS_MAJOR_VERSION)
Returns:
int: The major version number (defaults to DEFAULT_REDIS_MAJOR_VERSION if unparseable)
"""
if self.redis_version == "Unknown":
return DEFAULT_REDIS_MAJOR_VERSION
try:
version_str = str(self.redis_version).strip()
# Handle cases where there's no dot (e.g., "7" or 7)
if "." in version_str:
major_version = int(version_str.split(".")[0])
else:
# Direct integer or single-digit string
major_version = int(float(version_str))
return major_version
except (ValueError, AttributeError):
# Fallback for unparseable versions (e.g., "v7.0.0", "latest")
return DEFAULT_REDIS_MAJOR_VERSION
def set_cache(self, key, value, **kwargs):
ttl = self.get_ttl(**kwargs)
print_verbose(
@ -1041,9 +1071,9 @@ class RedisCache(BaseCache):
# Test the connection
ping_result = await redis_client.ping()
# Close the connection
await redis_client.aclose()
await redis_client.aclose() # type: ignore[attr-defined]
if ping_result:
return {
@ -1259,11 +1289,7 @@ class RedisCache(BaseCache):
start_time = time.time()
print_verbose(f"LPOP from Redis list: key: {key}, count: {count}")
try:
major_version: int = 7
# Check Redis version and use appropriate method
if self.redis_version != "Unknown":
# Parse version string like "6.0.0" to get major version
major_version = int(self.redis_version.split(".")[0])
major_version = self._parse_redis_major_version()
if count is not None and major_version < 7:
# For Redis < 7.0, use pipeline to execute multiple LPOP commands

View file

@ -83,10 +83,10 @@ class RedisClusterCache(RedisCache):
)
# Test the connection
ping_result = await redis_client.ping()
ping_result = await redis_client.ping() # type: ignore[attr-defined]
# Close the connection
await redis_client.aclose()
await redis_client.aclose() # type: ignore[attr-defined]
if ping_result:
return {

View file

@ -538,16 +538,20 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
return cast(List["ALL_RESPONSES_API_TOOL_PARAMS"], responses_tools)
def _map_reasoning_effort(self, reasoning_effort: str) -> Optional[Reasoning]:
def _map_reasoning_effort(self, reasoning_effort: Union[str, Dict[str, Any]]) -> Optional[Reasoning]:
# If dict is passed, convert it directly to Reasoning object
if isinstance(reasoning_effort, dict):
return Reasoning(**reasoning_effort) # type: ignore[typeddict-item]
# If string is passed, map without summary (default)
if reasoning_effort == "high":
return Reasoning(effort="high", summary="detailed")
return Reasoning(effort="high")
elif reasoning_effort == "medium":
# docs say "summary": "concise" is also an option, but it was rejected in practice, so defaulting "auto"
return Reasoning(effort="medium", summary="auto")
return Reasoning(effort="medium")
elif reasoning_effort == "low":
return Reasoning(effort="low", summary="auto")
return Reasoning(effort="low")
elif reasoning_effort == "minimal":
return Reasoning(effort="minimal", summary="auto")
return Reasoning(effort="minimal")
return None
def _map_responses_status_to_finish_reason(self, status: Optional[str]) -> str:

View file

@ -1,6 +1,7 @@
import os
from typing import List, Literal
DEFAULT_HEALTH_CHECK_PROMPT = str(os.getenv("DEFAULT_HEALTH_CHECK_PROMPT", "test from litellm"))
AZURE_DEFAULT_RESPONSES_API_VERSION = str(
os.getenv("AZURE_DEFAULT_RESPONSES_API_VERSION", "preview")
)
@ -209,8 +210,17 @@ DEFAULT_POLLING_INTERVAL = float(
os.getenv("DEFAULT_POLLING_INTERVAL", 0.03)
) # default polling interval for the scheduler
AZURE_OPERATION_POLLING_TIMEOUT = int(os.getenv("AZURE_OPERATION_POLLING_TIMEOUT", 120))
AZURE_DOCUMENT_INTELLIGENCE_API_VERSION = str(
os.getenv("AZURE_DOCUMENT_INTELLIGENCE_API_VERSION", "2024-11-30")
)
AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI = int(
os.getenv("AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI", 96)
)
REDIS_SOCKET_TIMEOUT = float(os.getenv("REDIS_SOCKET_TIMEOUT", 0.1))
REDIS_CONNECTION_POOL_TIMEOUT = int(os.getenv("REDIS_CONNECTION_POOL_TIMEOUT", 5))
# Default Redis major version to assume when version cannot be determined
# Using 7 as it's the modern version that supports LPOP with count parameter
DEFAULT_REDIS_MAJOR_VERSION = int(os.getenv("DEFAULT_REDIS_MAJOR_VERSION", 7))
NON_LLM_CONNECTION_TIMEOUT = int(
os.getenv("NON_LLM_CONNECTION_TIMEOUT", 15)
) # timeout for adjacent services (e.g. jwt auth)
@ -270,6 +280,8 @@ ANTHROPIC_WEB_SEARCH_TOOL_MAX_USES = {
DEFAULT_IMAGE_ENDPOINT_MODEL = "dall-e-2"
DEFAULT_VIDEO_ENDPOINT_MODEL = "sora-2"
DEFAULT_GOOGLE_VIDEO_DURATION_SECONDS = int(os.getenv("DEFAULT_GOOGLE_VIDEO_DURATION_SECONDS", 8))
### DATAFORSEO CONSTANTS ###
DEFAULT_DATAFORSEO_LOCATION_CODE = int(
os.getenv("DEFAULT_DATAFORSEO_LOCATION_CODE", 2250)

View file

@ -17,6 +17,9 @@ from litellm.constants import (
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
StandardBuiltInToolCostTracking,
)
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
TranscriptionUsageObjectTransformation,
)
from litellm.litellm_core_utils.llm_cost_calc.utils import (
CostCalculatorUtils,
_generic_cost_per_character,
@ -81,6 +84,8 @@ from litellm.types.utils import (
LlmProvidersSet,
ModelInfo,
StandardBuiltInToolsParams,
TranscriptionUsageDurationObject,
TranscriptionUsageTokensObject,
Usage,
VectorStoreSearchResponse,
)
@ -319,20 +324,32 @@ def cost_per_token( # noqa: PLR0915
usage=usage_block, model=model, custom_llm_provider=custom_llm_provider
)
elif call_type == "atranscription" or call_type == "transcription":
return openai_cost_per_second(
model=model,
custom_llm_provider=custom_llm_provider,
duration=audio_transcription_file_duration,
)
if model == "gpt-4o-mini-transcribe":
return openai_cost_per_token(
model=model,
usage=usage_block,
service_tier=service_tier,
)
else:
return openai_cost_per_second(
model=model,
custom_llm_provider=custom_llm_provider,
duration=audio_transcription_file_duration,
)
elif call_type == "search" or call_type == "asearch":
# Search providers use per-query pricing
from litellm.search import search_provider_cost_per_query
return search_provider_cost_per_query(
model=model,
custom_llm_provider=custom_llm_provider,
number_of_queries=number_of_queries or 1,
optional_params=response._hidden_params if response and hasattr(response, "_hidden_params") else None
optional_params=(
response._hidden_params
if response and hasattr(response, "_hidden_params")
else None
),
)
elif custom_llm_provider == "vertex_ai":
cost_router = google_cost_router(
@ -509,16 +526,18 @@ def _select_model_name_for_cost_calc(
else:
return_model = model
if base_model is not None:
elif base_model is not None:
return_model = base_model
if completion_response_model is None and hidden_params is not None:
elif completion_response_model is None and hidden_params is not None:
if (
hidden_params.get("model", None) is not None
and len(hidden_params["model"]) > 0
):
return_model = hidden_params.get("model", model)
if hidden_params is not None and hidden_params.get("region_name", None) is not None:
elif (
hidden_params is not None and hidden_params.get("region_name", None) is not None
):
region_name = hidden_params.get("region_name", None)
if return_model is None and completion_response_model is not None:
@ -573,6 +592,19 @@ def _get_usage_object(
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
usage_obj
)
elif TranscriptionUsageObjectTransformation.is_transcription_usage_object(
usage_obj
):
return (
TranscriptionUsageObjectTransformation.transform_transcription_usage_object(
cast(
Union[
TranscriptionUsageDurationObject, TranscriptionUsageTokensObject
],
usage_obj,
)
)
)
elif isinstance(usage_obj, dict):
return Usage(**usage_obj)
elif isinstance(usage_obj, BaseModel):
@ -586,8 +618,12 @@ def _get_usage_object(
def _is_known_usage_objects(usage_obj):
"""Returns True if the usage obj is a known Usage type"""
return isinstance(usage_obj, litellm.Usage) or isinstance(
usage_obj, ResponseAPIUsage
return (
isinstance(usage_obj, litellm.Usage)
or isinstance(usage_obj, ResponseAPIUsage)
or TranscriptionUsageObjectTransformation.is_transcription_usage_object(
usage_obj
)
)
@ -827,6 +863,22 @@ def completion_cost( # noqa: PLR0915
_usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
_usage
).model_dump()
elif TranscriptionUsageObjectTransformation.is_transcription_usage_object(
_usage
):
tr_usage = TranscriptionUsageObjectTransformation.transform_transcription_usage_object(
cast(
Union[
TranscriptionUsageDurationObject,
TranscriptionUsageTokensObject,
],
_usage,
)
)
if tr_usage is not None:
_usage = tr_usage.model_dump()
else:
_usage = _usage
# get input/output tokens from completion_response
prompt_tokens = _usage.get("prompt_tokens", 0)
@ -853,15 +905,6 @@ def completion_cost( # noqa: PLR0915
"custom_llm_provider", custom_llm_provider or None
)
region_name = hidden_params.get("region_name", region_name)
size = hidden_params.get("optional_params", {}).get(
"size", "1024-x-1024"
) # openai default
quality = hidden_params.get("optional_params", {}).get(
"quality", "standard"
) # openai default
n = hidden_params.get("optional_params", {}).get(
"n", 1
) # openai default
else:
if model is None:
raise ValueError(
@ -888,7 +931,9 @@ def completion_cost( # noqa: PLR0915
str(e)
)
)
if CostCalculatorUtils._call_type_has_image_response(call_type):
if CostCalculatorUtils._call_type_has_image_response(
call_type
) and isinstance(completion_response, ImageResponse):
### IMAGE GENERATION COST CALCULATION ###
return CostCalculatorUtils.route_image_generation_cost_calculator(
model=model,
@ -906,27 +951,32 @@ def completion_cost( # noqa: PLR0915
or call_type == CallTypes.avideo_remix.value
):
### VIDEO GENERATION COST CALCULATION ###
if completion_response is not None and hasattr(completion_response, 'usage'):
usage_obj = completion_response.usage
usage_obj = getattr(completion_response, "usage", None)
if completion_response is not None and usage_obj:
# Handle both dict and Pydantic Usage object
if isinstance(usage_obj, dict):
duration_seconds = usage_obj.get('duration_seconds', None)
duration_seconds = usage_obj.get("duration_seconds", None)
else:
duration_seconds = getattr(usage_obj, 'duration_seconds', None)
duration_seconds = getattr(
usage_obj, "duration_seconds", None
)
if duration_seconds is not None:
# Calculate cost based on video duration using video-specific cost calculation
from litellm.llms.openai.cost_calculation import video_generation_cost
from litellm.llms.openai.cost_calculation import (
video_generation_cost,
)
return video_generation_cost(
model=model,
duration_seconds=duration_seconds,
custom_llm_provider=custom_llm_provider
custom_llm_provider=custom_llm_provider,
)
# Fallback to default video cost calculation if no duration available
return default_video_cost_calculator(
model=model,
duration_seconds=0.0, # Default to 0 if no duration available
custom_llm_provider=custom_llm_provider
custom_llm_provider=custom_llm_provider,
)
elif (
call_type == CallTypes.speech.value
@ -1460,13 +1510,13 @@ def default_video_cost_calculator(
model_name_without_custom_llm_provider = model.replace(
f"{custom_llm_provider}/", ""
)
base_model_name = f"{custom_llm_provider}/{model_name_without_custom_llm_provider}"
base_model_name = (
f"{custom_llm_provider}/{model_name_without_custom_llm_provider}"
)
verbose_logger.debug(
f"Looking up cost for video model: {base_model_name}"
)
verbose_logger.debug(f"Looking up cost for video model: {base_model_name}")
model_without_provider = model.split('/')[-1]
model_without_provider = model.split("/")[-1]
# Try model with provider first, fall back to base model name
cost_info: Optional[dict] = None
@ -1480,7 +1530,7 @@ def default_video_cost_calculator(
if _model is not None and _model in litellm.model_cost:
cost_info = litellm.model_cost[_model]
break
# If still not found, try with custom_llm_provider prefix
if cost_info is None and custom_llm_provider:
prefixed_model = f"{custom_llm_provider}/{model}"
@ -1495,12 +1545,12 @@ def default_video_cost_calculator(
video_cost_per_second = cost_info.get("output_cost_per_video_per_second")
if video_cost_per_second is not None:
return video_cost_per_second * duration_seconds
# Fallback to general output cost per second
output_cost_per_second = cost_info.get("output_cost_per_second")
if output_cost_per_second is not None:
return output_cost_per_second * duration_seconds
# If no cost information found, return 0
verbose_logger.info(
f"No cost information found for video model {model}. Please add pricing to model_prices_and_context_window.json"

View file

@ -450,6 +450,7 @@ class ContentPolicyViolationError(BadRequestError): # type: ignore
llm_provider,
response: Optional[httpx.Response] = None,
litellm_debug_info: Optional[str] = None,
provider_specific_fields: Optional[dict] = None,
):
self.status_code = 400
self.message = "litellm.ContentPolicyViolationError: {}".format(message)
@ -458,6 +459,8 @@ class ContentPolicyViolationError(BadRequestError): # type: ignore
self.litellm_debug_info = litellm_debug_info
request = httpx.Request(method="POST", url="https://api.openai.com/v1")
self.response = httpx.Response(status_code=400, request=request)
self.provider_specific_fields = provider_specific_fields
super().__init__(
message=self.message,
model=self.model, # type: ignore
@ -465,16 +468,18 @@ class ContentPolicyViolationError(BadRequestError): # type: ignore
response=self.response,
litellm_debug_info=self.litellm_debug_info,
) # Call the base class constructor with the parameters it needs
def __str__(self):
_message = self.message
if self.num_retries:
_message += f" LiteLLM Retried: {self.num_retries} times"
if self.max_retries:
_message += f", LiteLLM Max Retries: {self.max_retries}"
return _message
return self._transform_error_to_string()
def __repr__(self):
return self._transform_error_to_string()
def _transform_error_to_string(self) -> str:
"""
Transform the error to a string
"""
_message = self.message
if self.num_retries:
_message += f" LiteLLM Retried: {self.num_retries} times"
@ -501,8 +506,62 @@ class ServiceUnavailableError(openai.APIStatusError): # type: ignore
self.litellm_debug_info = litellm_debug_info
self.max_retries = max_retries
self.num_retries = num_retries
_response_headers = (
getattr(response, "headers", None) if response is not None else None
)
self.response = httpx.Response(
status_code=self.status_code,
headers=_response_headers,
request=httpx.Request(
method="POST",
url=" https://cloud.google.com/vertex-ai/",
),
)
super().__init__(
self.message, response=self.response, body=None
) # Call the base class constructor with the parameters it needs
def __str__(self):
_message = self.message
if self.num_retries:
_message += f" LiteLLM Retried: {self.num_retries} times"
if self.max_retries:
_message += f", LiteLLM Max Retries: {self.max_retries}"
return _message
def __repr__(self):
_message = self.message
if self.num_retries:
_message += f" LiteLLM Retried: {self.num_retries} times"
if self.max_retries:
_message += f", LiteLLM Max Retries: {self.max_retries}"
return _message
class BadGatewayError(openai.APIStatusError): # type: ignore
def __init__(
self,
message,
llm_provider,
model,
response: Optional[httpx.Response] = None,
litellm_debug_info: Optional[str] = None,
max_retries: Optional[int] = None,
num_retries: Optional[int] = None,
):
self.status_code = 502
self.message = "litellm.BadGatewayError: {}".format(message)
self.llm_provider = llm_provider
self.model = model
self.litellm_debug_info = litellm_debug_info
self.max_retries = max_retries
self.num_retries = num_retries
_response_headers = (
getattr(response, "headers", None) if response is not None else None
)
self.response = httpx.Response(
status_code=self.status_code,
headers=_response_headers,
request=httpx.Request(
method="POST",
url=" https://cloud.google.com/vertex-ai/",
@ -547,8 +606,12 @@ class InternalServerError(openai.InternalServerError): # type: ignore
self.litellm_debug_info = litellm_debug_info
self.max_retries = max_retries
self.num_retries = num_retries
_response_headers = (
getattr(response, "headers", None) if response is not None else None
)
self.response = httpx.Response(
status_code=self.status_code,
headers=_response_headers,
request=httpx.Request(
method="POST",
url=" https://cloud.google.com/vertex-ai/",
@ -754,6 +817,7 @@ LITELLM_EXCEPTION_TYPES = [
ContentPolicyViolationError,
InternalServerError,
ServiceUnavailableError,
BadGatewayError,
APIError,
APIConnectionError,
APIResponseValidationError,

View file

@ -1,6 +1,5 @@
import json
from abc import ABC
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Type, Union
from typing import TYPE_CHECKING, Any, Dict, Optional, Type
from typing_extensions import override
@ -89,19 +88,124 @@ class ArizeOTELAttributes(BaseLLMObsOTELAttributes):
)
def _set_tool_attributes(span: "Span", optional_params: dict):
"""Helper to set tool and function call attributes on span."""
from litellm.integrations._types.open_inference import (
MessageAttributes,
SpanAttributes,
ToolCallAttributes,
)
tools = optional_params.get("tools")
if tools:
for idx, tool in enumerate(tools):
function = tool.get("function")
if not function:
continue
prefix = f"{SpanAttributes.LLM_TOOLS}.{idx}"
safe_set_attribute(
span, f"{prefix}.{SpanAttributes.TOOL_NAME}", function.get("name")
)
safe_set_attribute(
span,
f"{prefix}.{SpanAttributes.TOOL_DESCRIPTION}",
function.get("description"),
)
safe_set_attribute(
span,
f"{prefix}.{SpanAttributes.TOOL_PARAMETERS}",
json.dumps(function.get("parameters")),
)
functions = optional_params.get("functions")
if functions:
for idx, function in enumerate(functions):
prefix = f"{MessageAttributes.MESSAGE_TOOL_CALLS}.{idx}"
safe_set_attribute(
span,
f"{prefix}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}",
function.get("name"),
)
def _set_response_attributes(span: "Span", response_obj):
"""Helper to set response output and token usage attributes on span."""
from litellm.integrations._types.open_inference import (
MessageAttributes,
SpanAttributes,
)
if not hasattr(response_obj, "get"):
return
for idx, choice in enumerate(response_obj.get("choices", [])):
response_message = choice.get("message", {})
safe_set_attribute(
span,
SpanAttributes.OUTPUT_VALUE,
response_message.get("content", ""),
)
prefix = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.{idx}"
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_ROLE}",
response_message.get("role"),
)
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_CONTENT}",
response_message.get("content", ""),
)
output_items = response_obj.get("output", [])
if output_items:
for i, item in enumerate(output_items):
prefix = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.{i}"
if hasattr(item, "type"):
item_type = item.type
if item_type == "reasoning" and hasattr(item, "summary"):
for summary in item.summary:
if hasattr(summary, "text"):
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_REASONING_SUMMARY}",
summary.text,
)
elif item_type == "message" and hasattr(item, "content"):
message_content = ""
content_list = item.content
if content_list and len(content_list) > 0:
first_content = content_list[0]
message_content = getattr(first_content, "text", "")
message_role = getattr(item, "role", "assistant")
safe_set_attribute(span, SpanAttributes.OUTPUT_VALUE, message_content)
safe_set_attribute(span, f"{prefix}.{MessageAttributes.MESSAGE_CONTENT}", message_content)
safe_set_attribute(span, f"{prefix}.{MessageAttributes.MESSAGE_ROLE}", message_role)
usage = response_obj and response_obj.get("usage")
if usage:
safe_set_attribute(span, SpanAttributes.LLM_TOKEN_COUNT_TOTAL, usage.get("total_tokens"))
completion_tokens = usage.get("completion_tokens") or usage.get("output_tokens")
if completion_tokens:
safe_set_attribute(span, SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, completion_tokens)
prompt_tokens = usage.get("prompt_tokens") or usage.get("input_tokens")
if prompt_tokens:
safe_set_attribute(span, SpanAttributes.LLM_TOKEN_COUNT_PROMPT, prompt_tokens)
reasoning_tokens = usage.get("output_tokens_details", {}).get("reasoning_tokens")
if reasoning_tokens:
safe_set_attribute(span, SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, reasoning_tokens)
def set_attributes(
span: "Span", kwargs, response_obj, attributes: Type[BaseLLMObsOTELAttributes]
): # noqa: PLR0915
):
"""
Populates span with OpenInference-compliant LLM attributes for Arize and Phoenix tracing.
"""
from litellm.integrations._types.open_inference import (
MessageAttributes,
OpenInferenceSpanKindValues,
SpanAttributes,
ToolCallAttributes,
)
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
try:
optional_params = kwargs.get("optional_params", {})
@ -112,11 +216,6 @@ def set_attributes(
if standard_logging_payload is None:
raise ValueError("standard_logging_object not found in kwargs")
#############################################
############ LLM CALL METADATA ##############
#############################################
# Set custom metadata for observability and trace enrichment.
metadata = (
standard_logging_payload.get("metadata")
if standard_logging_payload
@ -125,253 +224,47 @@ def set_attributes(
if metadata is not None:
safe_set_attribute(span, SpanAttributes.METADATA, safe_dumps(metadata))
#############################################
########## LLM Request Attributes ###########
#############################################
# The name of the LLM a request is being made to.
if kwargs.get("model"):
safe_set_attribute(
span,
SpanAttributes.LLM_MODEL_NAME,
kwargs.get("model"),
)
safe_set_attribute(span, SpanAttributes.LLM_MODEL_NAME, kwargs.get("model"))
# The LLM request type.
safe_set_attribute(
span,
"llm.request.type",
standard_logging_payload["call_type"],
)
safe_set_attribute(span, "llm.request.type", standard_logging_payload["call_type"])
safe_set_attribute(span, SpanAttributes.LLM_PROVIDER, litellm_params.get("custom_llm_provider", "Unknown"))
# The Generative AI Provider: Azure, OpenAI, etc.
safe_set_attribute(
span,
SpanAttributes.LLM_PROVIDER,
litellm_params.get("custom_llm_provider", "Unknown"),
)
# The maximum number of tokens the LLM generates for a request.
if optional_params.get("max_tokens"):
safe_set_attribute(
span,
"llm.request.max_tokens",
optional_params.get("max_tokens"),
)
# The temperature setting for the LLM request.
safe_set_attribute(span, "llm.request.max_tokens", optional_params.get("max_tokens"))
if optional_params.get("temperature"):
safe_set_attribute(
span,
"llm.request.temperature",
optional_params.get("temperature"),
)
# The top_p sampling setting for the LLM request.
safe_set_attribute(span, "llm.request.temperature", optional_params.get("temperature"))
if optional_params.get("top_p"):
safe_set_attribute(
span,
"llm.request.top_p",
optional_params.get("top_p"),
)
safe_set_attribute(span, "llm.request.top_p", optional_params.get("top_p"))
# Indicates whether response is streamed.
safe_set_attribute(
span,
"llm.is_streaming",
str(optional_params.get("stream", False)),
)
safe_set_attribute(span, "llm.is_streaming", str(optional_params.get("stream", False)))
# Logs the user ID if present.
if optional_params.get("user"):
safe_set_attribute(
span,
"llm.user",
optional_params.get("user"),
)
safe_set_attribute(span, "llm.user", optional_params.get("user"))
# The unique identifier for the completion.
if response_obj and response_obj.get("id"):
safe_set_attribute(span, "llm.response.id", response_obj.get("id"))
# The model used to generate the response.
if response_obj and response_obj.get("model"):
safe_set_attribute(
span,
"llm.response.model",
response_obj.get("model"),
)
safe_set_attribute(span, "llm.response.model", response_obj.get("model"))
# Required by OpenInference to mark span as LLM kind.
safe_set_attribute(
span,
SpanAttributes.OPENINFERENCE_SPAN_KIND,
OpenInferenceSpanKindValues.LLM.value,
)
safe_set_attribute(span, SpanAttributes.OPENINFERENCE_SPAN_KIND, OpenInferenceSpanKindValues.LLM.value)
attributes.set_messages(span, kwargs)
# Capture tools (function definitions) used in the LLM call.
tools = optional_params.get("tools")
if tools:
for idx, tool in enumerate(tools):
function = tool.get("function")
if not function:
continue
prefix = f"{SpanAttributes.LLM_TOOLS}.{idx}"
safe_set_attribute(
span, f"{prefix}.{SpanAttributes.TOOL_NAME}", function.get("name")
)
safe_set_attribute(
span,
f"{prefix}.{SpanAttributes.TOOL_DESCRIPTION}",
function.get("description"),
)
safe_set_attribute(
span,
f"{prefix}.{SpanAttributes.TOOL_PARAMETERS}",
json.dumps(function.get("parameters")),
)
_set_tool_attributes(span=span, optional_params=optional_params)
# Capture tool calls made during function-calling LLM flows.
functions = optional_params.get("functions")
if functions:
for idx, function in enumerate(functions):
prefix = f"{MessageAttributes.MESSAGE_TOOL_CALLS}.{idx}"
safe_set_attribute(
span,
f"{prefix}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}",
function.get("name"),
)
# Capture invocation parameters and user ID if available.
model_params = (
standard_logging_payload.get("model_parameters")
if standard_logging_payload
else None
)
if model_params:
# The Generative AI Provider: Azure, OpenAI, etc.
safe_set_attribute(
span,
SpanAttributes.LLM_INVOCATION_PARAMETERS,
safe_dumps(model_params),
)
safe_set_attribute(span, SpanAttributes.LLM_INVOCATION_PARAMETERS, safe_dumps(model_params))
if model_params.get("user"):
user_id = model_params.get("user")
if user_id is not None:
safe_set_attribute(span, SpanAttributes.USER_ID, user_id)
#############################################
########## LLM Response Attributes ##########
#############################################
# Captures response tokens, message, and content.
if hasattr(response_obj, "get"):
# Handle chat completions API (choices field)
for idx, choice in enumerate(response_obj.get("choices", [])):
response_message = choice.get("message", {})
safe_set_attribute(
span,
SpanAttributes.OUTPUT_VALUE,
response_message.get("content", ""),
)
# This shows up under `output_messages` tab on the span page.
prefix = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.{idx}"
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_ROLE}",
response_message.get("role"),
)
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_CONTENT}",
response_message.get("content", ""),
)
# Handle responses API (output field)
output_items = response_obj.get("output", [])
if output_items:
for i, item in enumerate(output_items):
prefix = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.{i}"
if hasattr(item, "type"):
item_type = item.type
# Extract reasoning summary
if item_type == "reasoning" and hasattr(item, "summary"):
for summary in item.summary:
if hasattr(summary, "text"):
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_REASONING_SUMMARY}",
summary.text,
)
# Extract message content
elif item_type == "message" and hasattr(item, "content"):
message_content = ""
content_list = item.content
if content_list and len(content_list) > 0:
first_content = content_list[0]
message_content = getattr(first_content, "text", "")
message_role = getattr(item, "role", "assistant")
safe_set_attribute(
span,
SpanAttributes.OUTPUT_VALUE,
message_content,
)
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_CONTENT}",
message_content,
)
safe_set_attribute(
span,
f"{prefix}.{MessageAttributes.MESSAGE_ROLE}",
message_role,
)
# Token usage info.
usage = response_obj and response_obj.get("usage")
if usage:
safe_set_attribute(
span,
SpanAttributes.LLM_TOKEN_COUNT_TOTAL,
usage.get("total_tokens"),
)
# The number of tokens used in the LLM response (completion).
# Responses API uses "output_tokens", chat completions uses "completion_tokens"
completion_tokens = usage.get("completion_tokens") or usage.get("output_tokens")
if completion_tokens:
safe_set_attribute(
span,
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION,
completion_tokens,
)
# The number of tokens used in the LLM prompt.
# Responses API uses "input_tokens", chat completions uses "prompt_tokens"
prompt_tokens = usage.get("prompt_tokens") or usage.get("input_tokens")
if prompt_tokens:
safe_set_attribute(
span,
SpanAttributes.LLM_TOKEN_COUNT_PROMPT,
prompt_tokens,
)
# The number of reasoning tokens in the output, if available.
reasoning_tokens = usage.get("output_tokens_details", {}).get("reasoning_tokens")
if reasoning_tokens:
safe_set_attribute(
span,
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING,
reasoning_tokens,
)
_set_response_attributes(span=span, response_obj=response_obj)
except Exception as e:
verbose_logger.error(

View file

@ -8,7 +8,6 @@ from typing import (
AsyncGenerator,
Dict,
List,
Literal,
Optional,
Tuple,
Union,
@ -24,6 +23,7 @@ from litellm.types.llms.openai import AllMessageValues, ChatCompletionRequest
from litellm.types.utils import (
AdapterCompletionStreamWrapper,
CallTypes,
CallTypesLiteral,
LLMResponseTypes,
ModelResponse,
ModelResponseStream,
@ -65,12 +65,11 @@ _BASE64_INLINE_PATTERN = re.compile(
class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callback#callback-class
# Class variables or attributes
def __init__(
self,
self,
turn_off_message_logging: bool = False,
# deprecated param, use `turn_off_message_logging` instead
message_logging: bool = True,
**kwargs
**kwargs,
) -> None:
"""
Args:
@ -221,7 +220,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
) -> Optional[Any]:
"""
Allow modifying streaming chunks just before they're returned to the user.
This is called for each streaming chunk in the response.
"""
pass
@ -292,18 +291,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
user_api_key_dict: UserAPIKeyAuth,
cache: DualCache,
data: dict,
call_type: Literal[
"completion",
"text_completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"pass_through_endpoint",
"rerank",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
) -> Optional[
Union[Exception, str, dict]
]: # raise exception if invalid, return a str for the user to receive - if rejected, or return a modified dictionary for passing into litellm
@ -342,16 +330,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
call_type: Literal[
"completion",
"embeddings",
"image_generation",
"moderation",
"audio_transcription",
"responses",
"mcp_call",
"anthropic_messages",
],
call_type: CallTypesLiteral,
) -> Any:
pass
@ -435,7 +414,6 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
# MCP TOOL CALL HOOKS
#########################################################
async def async_post_mcp_tool_call_hook(
self, kwargs, response_obj: MCPPostCallResponseObject, start_time, end_time
) -> Optional[MCPPostCallResponseObject]:
@ -519,33 +497,32 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
if LITELLM_METADATA_FIELD in request_kwargs:
return LITELLM_METADATA_FIELD
return OLD_LITELLM_METADATA_FIELD
def redact_standard_logging_payload_from_model_call_details(
self, model_call_details: Dict
) -> Dict:
"""
Only redacts messages and responses when self.turn_off_message_logging is True
By default, self.turn_off_message_logging is False and this does nothing.
Return a redacted deepcopy of the provided logging payload.
This is useful for logging payloads that contain sensitive information.
"""
from copy import copy
from litellm import Choices, Message, ModelResponse
from litellm.types.utils import LiteLLMCommonStrings
turn_off_message_logging: bool = getattr(self, "turn_off_message_logging", False)
if turn_off_message_logging is False:
return model_call_details
# Only make a shallow copy of the top-level dict to avoid deepcopy issues
# with complex objects like AuthenticationError that may be present
model_call_details_copy = copy(model_call_details)
redacted_str = LiteLLMCommonStrings.redacted_by_litellm.value
redacted_str = "redacted-by-litellm"
standard_logging_object = model_call_details.get("standard_logging_object")
if standard_logging_object is None:
return model_call_details_copy
@ -554,20 +531,40 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
standard_logging_object_copy = copy(standard_logging_object)
if standard_logging_object_copy.get("messages") is not None:
standard_logging_object_copy["messages"] = [Message(content=redacted_str).model_dump()]
standard_logging_object_copy["messages"] = [
Message(content=redacted_str).model_dump()
]
if standard_logging_object_copy.get("response") is not None:
model_response = ModelResponse(
choices=[Choices(message=Message(content=redacted_str))]
)
model_response_dict = model_response.model_dump()
standard_logging_object_copy["response"] = model_response_dict
response = standard_logging_object_copy["response"]
# Check if this is a ResponsesAPIResponse (has "output" field)
if isinstance(response, dict) and "output" in response:
# Make a copy to avoid modifying the original
from copy import deepcopy
response_copy = deepcopy(response)
# Redact content in output array
if isinstance(response_copy.get("output"), list):
for output_item in response_copy["output"]:
if isinstance(output_item, dict) and "content" in output_item:
if isinstance(output_item["content"], list):
# Redact text in content items
for content_item in output_item["content"]:
if isinstance(content_item, dict) and "text" in content_item:
content_item["text"] = redacted_str
standard_logging_object_copy["response"] = response_copy
else:
# Standard ModelResponse format
model_response = ModelResponse(
choices=[Choices(message=Message(content=redacted_str))]
)
model_response_dict = model_response.model_dump()
standard_logging_object_copy["response"] = model_response_dict
model_call_details_copy["standard_logging_object"] = standard_logging_object_copy
model_call_details_copy["standard_logging_object"] = (
standard_logging_object_copy
)
return model_call_details_copy
async def get_proxy_server_request_from_cold_storage_with_object_key(
self,
object_key: str,
@ -577,9 +574,37 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
"""
pass
def handle_callback_failure(self, callback_name: str):
"""
Handle callback logging failures by incrementing Prometheus metrics.
Call this method in exception handlers within your callback when logging fails.
"""
try:
import litellm
from litellm._logging import verbose_logger
all_callbacks = litellm.logging_callback_manager._get_all_callbacks()
for callback_obj in all_callbacks:
if hasattr(callback_obj, 'increment_callback_logging_failure'):
verbose_logger.debug(f"Incrementing callback failure metric for {callback_name}")
callback_obj.increment_callback_logging_failure(callback_name=callback_name) # type: ignore
return
verbose_logger.debug(
f"No callback with increment_callback_logging_failure method found for {callback_name}. "
"Ensure 'prometheus' is in your callbacks config."
)
except Exception as e:
from litellm._logging import verbose_logger
verbose_logger.debug(f"Error in handle_callback_failure for {callback_name}: {str(e)}")
async def _strip_base64_from_messages(
self, payload: "StandardLoggingPayload", max_depth: int = DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER
self,
payload: "StandardLoggingPayload",
max_depth: int = DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER,
) -> "StandardLoggingPayload":
"""
Removes or redacts base64-encoded file data (e.g., PDFs, images, audio)
@ -609,9 +634,47 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
f"[CustomLogger] Completed base64 strip; retained {total_items} content items"
)
return payload
def _redact_base64(self, value: Any, depth: int = 0, max_depth: int = DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER) -> Any:
def _strip_base64_from_messages_sync(
self, payload: "StandardLoggingPayload", max_depth: int = DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER
) -> "StandardLoggingPayload":
"""
Removes or redacts base64-encoded file data (e.g., PDFs, images, audio)
from messages and responses before sending to SQS.
Behavior:
• Drop entries with a 'file' key.
• Drop entries with type == 'file' or any non-text type.
• Keep untyped or text content.
• Recursively redact inline base64 blobs in *any* string field, at any depth.
"""
raw_messages: Any = payload.get("messages", [])
messages: List[Any] = raw_messages if isinstance(raw_messages, list) else []
verbose_logger.debug(f"[CustomLogger] Stripping base64 from {len(messages)} messages")
if messages:
payload["messages"] = self._process_messages(
messages=messages, max_depth=max_depth
)
total_items = 0
for m in payload.get("messages", []) or []:
if isinstance(m, dict):
content = m.get("content", [])
if isinstance(content, list):
total_items += len(content)
verbose_logger.debug(
f"[CustomLogger] Completed base64 strip; retained {total_items} content items"
)
return payload
def _redact_base64(
self,
value: Any,
depth: int = 0,
max_depth: int = DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER,
) -> Any:
"""Recursively redact inline base64 from any nested structure with a max recursion depth limit."""
if depth > max_depth:
verbose_logger.warning(
@ -628,10 +691,16 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
return value
if isinstance(value, list):
return [self._redact_base64(value=v, depth=depth + 1, max_depth=max_depth) for v in value]
return [
self._redact_base64(value=v, depth=depth + 1, max_depth=max_depth)
for v in value
]
if isinstance(value, dict):
return {k: self._redact_base64(value=v, depth=depth + 1, max_depth=max_depth) for k, v in value.items()}
return {
k: self._redact_base64(value=v, depth=depth + 1, max_depth=max_depth)
for k, v in value.items()
}
return value
@ -654,10 +723,14 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
cleaned: List[Any] = []
for c in contents:
if self._should_keep_content(content=c):
cleaned.append(self._redact_base64(value=c, max_depth=max_depth))
cleaned.append(
self._redact_base64(value=c, max_depth=max_depth)
)
msg["content"] = cleaned
else:
msg["content"] = self._redact_base64(value=contents, max_depth=max_depth)
msg["content"] = self._redact_base64(
value=contents, max_depth=max_depth
)
for key, val in list(msg.items()):
if key != "content":

View file

@ -0,0 +1,254 @@
"""
Custom Secret Manager Integration
This module provides a base class for implementing custom secret managers in LiteLLM.
Usage:
from litellm.integrations.custom_secret_manager import CustomSecretManager
class MySecretManager(CustomSecretManager):
def __init__(self):
super().__init__(secret_manager_name="my_secret_manager")
async def async_read_secret(
self,
secret_name: str,
optional_params=None,
timeout=None,
):
# Your implementation here
return await self._fetch_secret_from_service(secret_name)
def sync_read_secret(
self,
secret_name: str,
optional_params=None,
timeout=None,
):
# Your implementation here
return self._fetch_secret_from_service_sync(secret_name)
# Set your custom secret manager
import litellm
from litellm.types.secret_managers.main import KeyManagementSystem
litellm.secret_manager_client = MySecretManager()
litellm._key_management_system = KeyManagementSystem.CUSTOM
"""
from abc import abstractmethod
from typing import Any, Dict, Optional, Union
import httpx
from litellm._logging import verbose_logger
from litellm.secret_managers.base_secret_manager import BaseSecretManager
class CustomSecretManager(BaseSecretManager):
"""
Base class for implementing custom secret managers.
This class provides a standard interface for implementing custom secret management
integrations in LiteLLM. Users can extend this class to integrate their own secret
management systems.
Example:
```python
from litellm.integrations.custom_secret_manager import CustomSecretManager
class MyVaultSecretManager(CustomSecretManager):
def __init__(self, vault_url: str, token: str):
super().__init__(secret_manager_name="my_vault")
self.vault_url = vault_url
self.token = token
async def async_read_secret(self, secret_name: str, optional_params=None, timeout=None):
# Implementation for reading secrets from your vault
async with httpx.AsyncClient() as client:
response = await client.get(
f"{self.vault_url}/v1/secret/{secret_name}",
headers={"X-Vault-Token": self.token},
timeout=timeout
)
return response.json()["data"]["value"]
def sync_read_secret(self, secret_name: str, optional_params=None, timeout=None):
# Sync implementation
with httpx.Client() as client:
response = client.get(
f"{self.vault_url}/v1/secret/{secret_name}",
headers={"X-Vault-Token": self.token},
timeout=timeout
)
return response.json()["data"]["value"]
```
"""
def __init__(
self,
secret_manager_name: Optional[str] = None,
**kwargs,
):
"""
Initialize the CustomSecretManager.
Args:
secret_manager_name: A descriptive name for your secret manager.
This is used for logging and debugging purposes.
**kwargs: Additional keyword arguments to pass to your secret manager.
"""
super().__init__()
self.secret_manager_name = secret_manager_name or "custom_secret_manager"
verbose_logger.info(
"Initialized custom secret manager"
)
@abstractmethod
async def async_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""
Asynchronously read a secret from your custom secret manager.
Args:
secret_name: Name/path of the secret to read
optional_params: Additional parameters specific to your secret manager
timeout: Request timeout
Returns:
The secret value if found, None otherwise
Raises:
Exception: If there's an error reading the secret
"""
pass
@abstractmethod
def sync_read_secret(
self,
secret_name: str,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Optional[str]:
"""
Synchronously read a secret from your custom secret manager.
Args:
secret_name: Name/path of the secret to read
optional_params: Additional parameters specific to your secret manager
timeout: Request timeout
Returns:
The secret value if found, None otherwise
Raises:
Exception: If there's an error reading the secret
"""
pass
async def async_write_secret(
self,
secret_name: str,
secret_value: str,
description: Optional[str] = None,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
tags: Optional[Union[dict, list]] = None,
) -> Dict[str, Any]:
"""
Asynchronously write a secret to your custom secret manager.
This is optional to implement. If your secret manager supports writing secrets,
you can override this method.
Args:
secret_name: Name/path of the secret to write
secret_value: Value to store
description: Description of the secret
optional_params: Additional parameters specific to your secret manager
timeout: Request timeout
tags: Optional tags to apply to the secret
Returns:
Response from the secret manager containing write operation details
Raises:
NotImplementedError: If write operations are not supported
"""
raise NotImplementedError(
f"Write operations are not implemented for {self.secret_manager_name}. "
"Override async_write_secret() to add write support."
)
async def async_delete_secret(
self,
secret_name: str,
recovery_window_in_days: Optional[int] = 7,
optional_params: Optional[dict] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> dict:
"""
Asynchronously delete a secret from your custom secret manager.
This is optional to implement. If your secret manager supports deleting secrets,
you can override this method.
Args:
secret_name: Name of the secret to delete
recovery_window_in_days: Number of days before permanent deletion (if supported)
optional_params: Additional parameters specific to your secret manager
timeout: Request timeout
Returns:
Response from the secret manager containing deletion details
Raises:
NotImplementedError: If delete operations are not supported
"""
raise NotImplementedError(
f"Delete operations are not implemented for {self.secret_manager_name}. "
"Override async_delete_secret() to add delete support."
)
def validate_environment(self) -> bool:
"""
Validate that all required environment variables and configuration are present.
Override this method to validate your secret manager's configuration.
Returns:
True if the environment is valid
Raises:
ValueError: If required configuration is missing
"""
verbose_logger.debug(
"No environment validation configured for custom secret manager"
)
return True
async def async_health_check(
self, timeout: Optional[Union[float, httpx.Timeout]] = None
) -> bool:
"""
Perform a health check on your secret manager.
This is optional to implement. Override this method to add health check support.
Args:
timeout: Request timeout
Returns:
True if the secret manager is healthy, False otherwise
"""
verbose_logger.debug(
f"Health check not implemented for {self.secret_manager_name}"
)
return True
def __repr__(self) -> str:
return f"<{self.__class__.__name__}(name={self.secret_manager_name})>"

View file

@ -17,7 +17,6 @@ import asyncio
import datetime
import os
import traceback
from litellm._uuid import uuid
from datetime import datetime as datetimeObj
from typing import Any, Dict, List, Optional, Union
@ -26,6 +25,7 @@ from httpx import Response
import litellm
from litellm._logging import verbose_logger
from litellm._uuid import uuid
from litellm.integrations.custom_batch_logger import CustomBatchLogger
from litellm.llms.custom_httpx.http_handler import (
_get_httpx_client,
@ -60,17 +60,19 @@ class DataDogLogger(
"""
Initializes the datadog logger, checks if the correct env variables are set
Required environment variables:
Required environment variables (Direct API):
`DD_API_KEY` - your datadog api key
`DD_SITE` - your datadog site, example = `"us5.datadoghq.com"`
Optional environment variables (DataDog Agent):
`DD_AGENT_HOST` - hostname or IP of DataDog agent, example = `"localhost"`
`DD_AGENT_PORT` - port of DataDog agent (default: 10518 for logs)
Note: If DD_AGENT_HOST is set, logs will be sent to the agent instead of directly to DataDog API.
In this case, DD_API_KEY and DD_SITE are not required (agent handles authentication).
"""
try:
verbose_logger.debug("Datadog: in init datadog logger")
# check if the correct env variables are set
if os.getenv("DD_API_KEY", None) is None:
raise Exception("DD_API_KEY is not set, set 'DD_API_KEY=<>")
if os.getenv("DD_SITE", None) is None:
raise Exception("DD_SITE is not set in .env, set 'DD_SITE=<>")
#########################################################
# Handle datadog_params set as litellm.datadog_params
@ -81,21 +83,16 @@ class DataDogLogger(
self.async_client = get_async_httpx_client(
llm_provider=httpxSpecialProvider.LoggingCallback
)
self.DD_API_KEY = os.getenv("DD_API_KEY")
self.intake_url = (
f"https://http-intake.logs.{os.getenv('DD_SITE')}/api/v2/logs"
)
###################################
# OPTIONAL -only used for testing
dd_base_url: Optional[str] = (
os.getenv("_DATADOG_BASE_URL")
or os.getenv("DATADOG_BASE_URL")
or os.getenv("DD_BASE_URL")
)
if dd_base_url is not None:
self.intake_url = f"{dd_base_url}/api/v2/logs"
###################################
# Configure DataDog endpoint (Agent or Direct API)
dd_agent_host = os.getenv("DD_AGENT_HOST")
if dd_agent_host:
self._configure_dd_agent(dd_agent_host=dd_agent_host)
else:
self._configure_dd_direct_api()
# Optional override for testing
self._apply_dd_base_url_override()
self.sync_client = _get_httpx_client()
asyncio.create_task(self.periodic_flush())
self.flush_lock = asyncio.Lock()
@ -123,6 +120,47 @@ class DataDogLogger(
dict_datadog_params = DatadogInitParams(**litellm.datadog_params).model_dump()
return dict_datadog_params
def _configure_dd_agent(self, dd_agent_host: str) -> None:
"""
Configure DataDog Agent for log forwarding
Args:
dd_agent_host: Hostname or IP of DataDog agent
"""
dd_agent_port = os.getenv("DD_AGENT_PORT", "10518") # default port for logs
self.intake_url = f"http://{dd_agent_host}:{dd_agent_port}/api/v2/logs"
self.DD_API_KEY = os.getenv("DD_API_KEY") # Optional when using agent
verbose_logger.debug(f"Datadog: Using DD Agent at {self.intake_url}")
def _configure_dd_direct_api(self) -> None:
"""
Configure direct DataDog API connection
Raises:
Exception: If required environment variables are not set
"""
if os.getenv("DD_API_KEY", None) is None:
raise Exception("DD_API_KEY is not set, set 'DD_API_KEY=<>")
if os.getenv("DD_SITE", None) is None:
raise Exception("DD_SITE is not set in .env, set 'DD_SITE=<>")
self.DD_API_KEY = os.getenv("DD_API_KEY")
self.intake_url = (
f"https://http-intake.logs.{os.getenv('DD_SITE')}/api/v2/logs"
)
def _apply_dd_base_url_override(self) -> None:
"""
Apply base URL override for testing purposes
"""
dd_base_url: Optional[str] = (
os.getenv("_DATADOG_BASE_URL")
or os.getenv("DATADOG_BASE_URL")
or os.getenv("DD_BASE_URL")
)
if dd_base_url is not None:
self.intake_url = f"{dd_base_url}/api/v2/logs"
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
"""
Async Log success events to Datadog
@ -226,12 +264,16 @@ class DataDogLogger(
end_time=end_time,
)
# Build headers
headers = {}
# Add API key if available (required for direct API, optional for agent)
if self.DD_API_KEY:
headers["DD-API-KEY"] = self.DD_API_KEY
response = self.sync_client.post(
url=self.intake_url,
json=dd_payload, # type: ignore
headers={
"DD-API-KEY": self.DD_API_KEY,
},
headers=headers,
)
response.raise_for_status()
@ -342,14 +384,21 @@ class DataDogLogger(
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
compressed_data = gzip.compress(safe_dumps(data).encode("utf-8"))
# Build headers
headers = {
"Content-Encoding": "gzip",
"Content-Type": "application/json",
}
# Add API key if available (required for direct API, optional for agent)
if self.DD_API_KEY:
headers["DD-API-KEY"] = self.DD_API_KEY
response = await self.async_client.post(
url=self.intake_url,
data=compressed_data, # type: ignore
headers={
"DD-API-KEY": self.DD_API_KEY,
"Content-Encoding": "gzip",
"Content-Type": "application/json",
},
headers=headers,
)
return response

View file

@ -0,0 +1,225 @@
"""
Modern Email Templates for LiteLLM Email Service with professional styling
"""
KEY_ROTATED_EMAIL_TEMPLATE = """
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Your API Key Has Been Rotated</title>
<style>
body, html {{
margin: 0;
padding: 0;
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Helvetica, Arial, sans-serif;
color: #333333;
background-color: #f8fafc;
line-height: 1.5;
}}
.container {{
max-width: 560px;
margin: 20px auto;
background-color: #ffffff;
border-radius: 8px;
overflow: hidden;
box-shadow: 0 1px 3px rgba(0,0,0,0.1);
}}
.header {{
padding: 24px 0;
text-align: center;
border-bottom: 1px solid #f1f5f9;
}}
.content {{
padding: 32px 40px;
}}
.greeting {{
font-size: 16px;
margin-bottom: 20px;
color: #333333;
}}
.message {{
font-size: 16px;
color: #333333;
margin-bottom: 20px;
}}
.key-container {{
margin: 28px 0;
}}
.key-label {{
font-size: 14px;
font-weight: 500;
margin-bottom: 8px;
color: #4b5563;
}}
.key {{
font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, monospace;
word-break: break-all;
background-color: #f9fafb;
border-radius: 6px;
padding: 16px;
font-size: 14px;
border: 1px solid #e5e7eb;
color: #4338ca;
}}
h2 {{
font-size: 18px;
font-weight: 600;
margin-top: 36px;
margin-bottom: 16px;
color: #333333;
}}
.budget-info {{
background-color: #f0fdf4;
border-radius: 6px;
padding: 14px 16px;
margin: 24px 0;
font-size: 14px;
border: 1px solid #dcfce7;
}}
.code-block {{
background-color: #f8fafc;
color: #334155;
border-radius: 8px;
padding: 20px;
font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, monospace;
font-size: 13px;
overflow-x: auto;
margin: 20px 0;
line-height: 1.6;
border: 1px solid #e2e8f0;
}}
.code-comment {{
color: #64748b;
}}
.code-string {{
color: #0369a1;
}}
.code-keyword {{
color: #7e22ce;
}}
.btn {{
display: inline-block;
padding: 8px 20px;
background-color: #6366f1;
color: #ffffff !important;
text-decoration: none;
border-radius: 6px;
font-weight: 500;
margin-top: 24px;
text-align: center;
font-size: 14px;
transition: background-color 0.2s;
}}
.btn:hover {{
background-color: #4f46e5;
color: #ffffff !important;
}}
.separator {{
height: 1px;
background-color: #f1f5f9;
margin: 40px 0 30px;
}}
.footer {{
padding: 24px 40px 32px;
text-align: center;
color: #64748b;
font-size: 13px;
background-color: #f8fafc;
border-top: 1px solid #f1f5f9;
}}
.social-links {{
margin-top: 12px;
}}
.social-links a {{
display: inline-block;
margin: 0 8px;
color: #64748b;
text-decoration: none;
}}
@media only screen and (max-width: 620px) {{
.container {{
width: 100%;
margin: 0;
border-radius: 0;
}}
.content {{
padding: 24px 20px;
}}
.footer {{
padding: 20px;
}}
}}
</style>
</head>
<body>
<div class="container">
<div class="header">
<img src="{email_logo_url}" alt="LiteLLM Logo" style="height: 32px; width: auto;">
</div>
<div class="content">
<div class="greeting">
<p>Hi {recipient_email},</p>
</div>
<div class="message">
<p><strong>Your LiteLLM API key has been rotated</strong> as part of our ongoing commitment to security best practices.</p>
<p style="margin-top: 16px;">Your previous API key has been deactivated and will no longer work. Please update your applications with the new key below.</p>
</div>
<div class="key-container">
<div class="key-label">Your New API Key</div>
<div class="key">{key_token}</div>
</div>
<div class="budget-info">
<p style="margin: 0;"><strong>Monthly Budget:</strong> {key_budget}</p>
</div>
<h2>Action Required</h2>
<p>Update your applications and systems with the new API key. Here's an example:</p>
<div class="code-block">
<span class="code-keyword">import</span> openai<br>
<br>
client = openai.OpenAI(<br>
&nbsp;&nbsp;api_key=<span class="code-string">"{key_token}"</span>,<br>
&nbsp;&nbsp;base_url=<span class="code-string">"{base_url}"</span><br>
)<br>
<br>
response = client.chat.completions.create(<br>
&nbsp;&nbsp;model=<span class="code-string">"gpt-3.5-turbo"</span>, <span class="code-comment"># model to send to the proxy</span><br>
&nbsp;&nbsp;messages = [<br>
&nbsp;&nbsp;&nbsp;&nbsp;{{<br>
&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;<span class="code-string">"role"</span>: <span class="code-string">"user"</span>,<br>
&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;<span class="code-string">"content"</span>: <span class="code-string">"this is a test request, write a short poem"</span><br>
&nbsp;&nbsp;&nbsp;&nbsp;}}<br>
&nbsp;&nbsp;]<br>
)
</div>
<div class="separator"></div>
<h2>Security Best Practices</h2>
<p style="margin-bottom: 12px;">To keep your API key secure:</p>
<ul style="margin: 0; padding-left: 20px; color: #333333;">
<li style="margin-bottom: 8px;">Never share your API key publicly or commit it to version control</li>
<li style="margin-bottom: 8px;">Store it securely using environment variables or secret management systems</li>
<li style="margin-bottom: 8px;">Monitor your API usage regularly for any unusual activity</li>
<li style="margin-bottom: 8px;">Rotate your keys periodically as a security best practice</li>
</ul>
<a href="https://docs.litellm.ai/docs/proxy/user_keys" class="btn" style="color: #ffffff;">View Documentation</a>
<div class="separator"></div>
<h2>Need Help?</h2>
<p>If you have any questions or need assistance updating your systems, please contact us at {email_support_contact}.</p>
</div>
{email_footer}
</div>
</body>
</html>
"""

View file

@ -688,16 +688,19 @@ class LangFuseLogger:
"completion_tokens": _usage_obj.completion_tokens,
"total_cost": cost if self._supports_costs() else None,
}
cache_read_input_tokens = _usage_obj.get(
"cache_read_input_tokens", 0
)
# According to langfuse documentation: "the input value must be reduced by the number of cache_read_input_tokens"
input_tokens = _usage_obj.prompt_tokens - cache_read_input_tokens
usage_details = LangfuseUsageDetails(
input=_usage_obj.prompt_tokens,
input=input_tokens,
output=_usage_obj.completion_tokens,
total=_usage_obj.total_tokens,
cache_creation_input_tokens=_usage_obj.get(
"cache_creation_input_tokens", 0
),
cache_read_input_tokens=_usage_obj.get(
"cache_read_input_tokens", 0
),
cache_read_input_tokens=cache_read_input_tokens,
)
generation_name = clean_metadata.pop("generation_name", None)

View file

@ -87,31 +87,10 @@ class LangfuseOtelLogger(OpenTelemetry):
return metadata
@staticmethod
def _set_langfuse_specific_attributes(span: Span, kwargs, response_obj):
"""
Sets Langfuse specific metadata attributes onto the OTEL span.
All keys supported by the vanilla Langfuse integration are mapped to
OTEL-safe attribute names defined in LangfuseSpanAttributes. Complex
values (lists/dicts) are serialised to JSON strings for OTEL
compatibility.
"""
def _set_metadata_attributes(span: Span, metadata: dict):
"""Helper to set metadata attributes from mapping."""
from litellm.integrations.arize._utils import safe_set_attribute
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
# 1) Environment variable override
langfuse_environment = os.environ.get("LANGFUSE_TRACING_ENVIRONMENT")
if langfuse_environment:
safe_set_attribute(
span,
LangfuseSpanAttributes.LANGFUSE_ENVIRONMENT.value,
langfuse_environment,
)
# 2) Dynamic metadata from kwargs / headers
metadata = LangfuseOtelLogger._extract_langfuse_metadata(kwargs)
# Mapping from metadata key -> OTEL attribute enum
mapping = {
"generation_name": LangfuseSpanAttributes.GENERATION_NAME,
"generation_id": LangfuseSpanAttributes.GENERATION_ID,
@ -135,7 +114,6 @@ class LangfuseOtelLogger(OpenTelemetry):
for key, enum_attr in mapping.items():
if key in metadata and metadata[key] is not None:
value = metadata[key]
# Lists / dicts must be stringified for OTEL
if isinstance(value, (list, dict)):
try:
value = json.dumps(value)
@ -143,117 +121,105 @@ class LangfuseOtelLogger(OpenTelemetry):
value = str(value)
safe_set_attribute(span, enum_attr.value, value)
# 3) Set observation input/output for better UI display
#
# These Langfuse-specific attributes provide better UI display,
# especially for tool calls and function calling.
# Set observation input (messages)
messages = kwargs.get("messages")
if messages:
safe_set_attribute(
span,
LangfuseSpanAttributes.OBSERVATION_INPUT.value,
safe_dumps(messages),
)
@staticmethod
def _set_observation_output(span: Span, response_obj):
"""Helper to set observation output attributes."""
from litellm.integrations.arize._utils import safe_set_attribute
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
# Set observation output (response with tool_calls if present)
if response_obj and hasattr(response_obj, "get"):
# Handle chat completions API (choices field)
choices = response_obj.get("choices", [])
if choices:
# Extract the first choice's message
first_choice = choices[0]
message = first_choice.get("message", {})
if not response_obj or not hasattr(response_obj, "get"):
return
# Check if there are tool_calls
tool_calls = message.get("tool_calls")
if tool_calls:
# Transform tool_calls to Langfuse-expected format
transformed_tool_calls = []
for tool_call in tool_calls:
function = tool_call.get("function", {})
arguments_str = function.get("arguments", "{}")
choices = response_obj.get("choices", [])
if choices:
first_choice = choices[0]
message = first_choice.get("message", {})
tool_calls = message.get("tool_calls")
if tool_calls:
transformed_tool_calls = []
for tool_call in tool_calls:
function = tool_call.get("function", {})
arguments_str = function.get("arguments", "{}")
try:
arguments_obj = (
json.loads(arguments_str)
if isinstance(arguments_str, str)
else arguments_str
)
except json.JSONDecodeError:
arguments_obj = {}
langfuse_tool_call = {
"id": response_obj.get("id", ""),
"name": function.get("name", ""),
"call_id": tool_call.get("id", ""),
"type": "function_call",
"arguments": arguments_obj,
}
transformed_tool_calls.append(langfuse_tool_call)
safe_set_attribute(span, LangfuseSpanAttributes.OBSERVATION_OUTPUT.value, safe_dumps(transformed_tool_calls))
else:
output_data = {}
if message.get("role"):
output_data["role"] = message.get("role")
if message.get("content") is not None:
output_data["content"] = message.get("content")
if output_data:
safe_set_attribute(span, LangfuseSpanAttributes.OBSERVATION_OUTPUT.value, safe_dumps(output_data))
# Parse arguments from JSON string to object
try:
arguments_obj = (
json.loads(arguments_str)
if isinstance(arguments_str, str)
else arguments_str
)
except json.JSONDecodeError:
arguments_obj = {}
# Create Langfuse-compatible tool call object
output = response_obj.get("output", [])
if output:
output_items_data: list[dict] = []
for item in output:
if hasattr(item, "type"):
item_type = item.type
if item_type == "reasoning" and hasattr(item, "summary"):
for summary in item.summary:
if hasattr(summary, "text"):
output_items_data.append({"role": "reasoning_summary", "content": summary.text})
elif item_type == "message":
output_items_data.append({
"role": getattr(item, "role", "assistant"),
"content": getattr(getattr(item, "content", [{}])[0], "text", "")
})
elif item_type == "function_call":
arguments_str = getattr(item, "arguments", "{}")
arguments_obj = json.loads(arguments_str) if isinstance(arguments_str, str) else arguments_str
langfuse_tool_call = {
"id": response_obj.get("id", ""),
"name": function.get("name", ""),
"call_id": tool_call.get("id", ""),
"id": getattr(item, "id", ""),
"name": getattr(item, "name", ""),
"call_id": getattr(item, "call_id", ""),
"type": "function_call",
"arguments": arguments_obj,
}
transformed_tool_calls.append(langfuse_tool_call)
output_items_data.append(langfuse_tool_call)
if output_items_data:
safe_set_attribute(span, LangfuseSpanAttributes.OBSERVATION_OUTPUT.value, safe_dumps(output_items_data))
# Set the observation output with transformed tool_calls
safe_set_attribute(
span,
LangfuseSpanAttributes.OBSERVATION_OUTPUT.value,
safe_dumps(transformed_tool_calls),
)
else:
# No tool_calls, use regular content-based output
output_data = {}
@staticmethod
def _set_langfuse_specific_attributes(span: Span, kwargs, response_obj):
"""
Sets Langfuse specific metadata attributes onto the OTEL span.
if message.get("role"):
output_data["role"] = message.get("role")
All keys supported by the vanilla Langfuse integration are mapped to
OTEL-safe attribute names defined in LangfuseSpanAttributes. Complex
values (lists/dicts) are serialised to JSON strings for OTEL
compatibility.
"""
from litellm.integrations.arize._utils import safe_set_attribute
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
if message.get("content") is not None:
output_data["content"] = message.get("content")
langfuse_environment = os.environ.get("LANGFUSE_TRACING_ENVIRONMENT")
if langfuse_environment:
safe_set_attribute(span, LangfuseSpanAttributes.LANGFUSE_ENVIRONMENT.value, langfuse_environment)
if output_data:
safe_set_attribute(
span,
LangfuseSpanAttributes.OBSERVATION_OUTPUT.value,
safe_dumps(output_data),
)
metadata = LangfuseOtelLogger._extract_langfuse_metadata(kwargs)
LangfuseOtelLogger._set_metadata_attributes(span=span, metadata=metadata)
# Handle responses API (output field)
output = response_obj.get("output", [])
if output:
output_data = []
for item in output:
if hasattr(item, "type"):
item_type = item.type
if item_type == "reasoning" and hasattr(item, "summary"):
for summary in item.summary:
if hasattr(summary, "text"):
output_data.append({
"role": "reasoning_summary",
"content": summary.text
})
elif item_type == "message":
output_data.append({
"role": getattr(item, "role", "assistant"),
"content": getattr(getattr(item, "content", [{}])[0], "text", "")
})
elif item_type == "function_call":
arguments_str = getattr(item, "arguments", "{}")
arguments_obj = json.loads(arguments_str) if isinstance(arguments_str, str) else arguments_str
langfuse_tool_call = {
"id": getattr(item, "id", ""),
"name": getattr(item, "name", ""),
"call_id": getattr(item, "call_id", ""),
"type": "function_call",
"arguments": arguments_obj,
}
output_data.append(langfuse_tool_call)
if output_data:
safe_set_attribute(
span,
LangfuseSpanAttributes.OBSERVATION_OUTPUT.value,
safe_dumps(output_data),
)
messages = kwargs.get("messages")
if messages:
safe_set_attribute(span, LangfuseSpanAttributes.OBSERVATION_INPUT.value, safe_dumps(messages))
LangfuseOtelLogger._set_observation_output(span=span, response_obj=response_obj)
@staticmethod
def _get_langfuse_otel_host() -> Optional[str]:

View file

@ -5,13 +5,11 @@ Relevant Issue: https://github.com/BerriAI/litellm/issues/13764
"""
import json
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union
from typing import TYPE_CHECKING, Any, Dict, Optional, Union
from numpy import isin
from pydantic import BaseModel
from typing_extensions import override
import litellm
from litellm.integrations.opentelemetry_utils.base_otel_llm_obs_attributes import (
BaseLLMObsOTELAttributes,
safe_set_attribute,

View file

@ -10,6 +10,7 @@ from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.types.services import ServiceLoggerPayload
from litellm.types.utils import (
ChatCompletionMessageToolCall,
CostBreakdown,
Function,
StandardCallbackDynamicParams,
StandardLoggingPayload,
@ -1076,6 +1077,16 @@ class OpenTelemetry(CustomLogger):
self.safe_set_attribute(
span=span, key="hidden_params", value=safe_dumps(hidden_params)
)
# Cost breakdown tracking
cost_breakdown: Optional[CostBreakdown] = standard_logging_payload.get("cost_breakdown")
if cost_breakdown:
for key, value in cost_breakdown.items():
if value is not None:
self.safe_set_attribute(
span=span,
key=f"gen_ai.cost.{key}",
value=value,
)
#############################################
########## LLM Request Attributes ###########
#############################################

Some files were not shown because too many files have changed in this diff Show more