diff --git a/.github/workflows/test-litellm-matrix.yml b/.github/workflows/test-litellm-matrix.yml
index d83fedcb2ae..1672a193161 100644
--- a/.github/workflows/test-litellm-matrix.yml
+++ b/.github/workflows/test-litellm-matrix.yml
@@ -12,44 +12,59 @@ concurrency:
jobs:
test:
runs-on: ubuntu-latest
- timeout-minutes: 15
+ timeout-minutes: 20 # Increased from 15 to 20
strategy:
fail-fast: false
matrix:
test-group:
# tests/test_litellm split by subdirectory (~560 files total)
- - name: "llms"
- path: "tests/test_litellm/llms"
- workers: 4
+ # Vertex AI tests separated for better isolation (prevent auth/env pollution)
+ - name: "llms-vertex"
+ path: "tests/test_litellm/llms/vertex_ai"
+ workers: 1
+ reruns: 2
+ - name: "llms-other"
+ path: "tests/test_litellm/llms --ignore=tests/test_litellm/llms/vertex_ai"
+ workers: 2
+ reruns: 2
# tests/test_litellm/proxy split by subdirectory (~180 files total)
- name: "proxy-guardrails"
path: "tests/test_litellm/proxy/guardrails tests/test_litellm/proxy/management_endpoints tests/test_litellm/proxy/management_helpers"
- workers: 4
+ workers: 2
+ reruns: 2
- name: "proxy-core"
path: "tests/test_litellm/proxy/auth tests/test_litellm/proxy/client tests/test_litellm/proxy/db tests/test_litellm/proxy/hooks tests/test_litellm/proxy/policy_engine"
- workers: 4
+ workers: 2
+ reruns: 2
- name: "proxy-misc"
path: "tests/test_litellm/proxy/_experimental tests/test_litellm/proxy/agent_endpoints tests/test_litellm/proxy/anthropic_endpoints tests/test_litellm/proxy/common_utils tests/test_litellm/proxy/discovery_endpoints tests/test_litellm/proxy/experimental tests/test_litellm/proxy/google_endpoints tests/test_litellm/proxy/health_endpoints tests/test_litellm/proxy/image_endpoints tests/test_litellm/proxy/middleware tests/test_litellm/proxy/openai_files_endpoint tests/test_litellm/proxy/pass_through_endpoints tests/test_litellm/proxy/prompts tests/test_litellm/proxy/public_endpoints tests/test_litellm/proxy/response_api_endpoints tests/test_litellm/proxy/spend_tracking tests/test_litellm/proxy/ui_crud_endpoints tests/test_litellm/proxy/vector_store_endpoints tests/test_litellm/proxy/test_*.py"
- workers: 4
+ workers: 2
+ reruns: 2
- name: "integrations"
path: "tests/test_litellm/integrations"
- workers: 4
+ workers: 2
+ reruns: 3 # Integration tests tend to be flakier
- name: "core-utils"
path: "tests/test_litellm/litellm_core_utils"
workers: 2
+ reruns: 1
- name: "other"
path: "tests/test_litellm/caching tests/test_litellm/responses tests/test_litellm/secret_managers tests/test_litellm/vector_stores tests/test_litellm/a2a_protocol tests/test_litellm/anthropic_interface tests/test_litellm/completion_extras tests/test_litellm/containers tests/test_litellm/enterprise tests/test_litellm/experimental_mcp_client tests/test_litellm/google_genai tests/test_litellm/images tests/test_litellm/interactions tests/test_litellm/passthrough tests/test_litellm/router_strategy tests/test_litellm/router_utils tests/test_litellm/types"
- workers: 4
+ workers: 2
+ reruns: 2
- name: "root"
path: "tests/test_litellm/test_*.py"
- workers: 4
+ workers: 2
+ reruns: 2
# tests/proxy_unit_tests split alphabetically (~48 files total)
- name: "proxy-unit-a"
path: "tests/proxy_unit_tests/test_[a-o]*.py"
workers: 2
+ reruns: 1
- name: "proxy-unit-b"
path: "tests/proxy_unit_tests/test_[p-z]*.py"
workers: 2
+ reruns: 1
name: test (${{ matrix.test-group.name }})
@@ -79,7 +94,8 @@ jobs:
run: |
poetry config virtualenvs.in-project true
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
- poetry run pip install pytest-retry==1.6.3 pytest-xdist google-genai==1.22.0 \
+ # pytest-rerunfailures and pytest-xdist are in pyproject.toml dev dependencies
+ poetry run pip install google-genai==1.22.0 \
google-cloud-aiplatform>=1.38 fastapi-offline==1.7.3 python-multipart==0.0.22 openapi-core
- name: Setup litellm-enterprise
@@ -92,4 +108,7 @@ jobs:
--tb=short -vv \
--maxfail=10 \
-n ${{ matrix.test-group.workers }} \
+ --reruns ${{ matrix.test-group.reruns }} \
+ --reruns-delay 1 \
+ --dist=loadscope \
--durations=20
diff --git a/.github/workflows/test_server_root_path.yml b/.github/workflows/test_server_root_path.yml
new file mode 100644
index 00000000000..bc559817503
--- /dev/null
+++ b/.github/workflows/test_server_root_path.yml
@@ -0,0 +1,96 @@
+name: Test Proxy SERVER_ROOT_PATH Routing
+permissions:
+ contents: read
+
+on:
+ pull_request:
+ branches: [main]
+
+jobs:
+ test-server-root-path:
+ runs-on: ubuntu-latest
+ timeout-minutes: 15
+
+ strategy:
+ matrix:
+ root_path: ["/api/v1", "/llmproxy"]
+
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v4
+
+ - name: Set up Docker Buildx
+ uses: docker/setup-buildx-action@v3
+
+ - name: Build Docker image
+ uses: docker/build-push-action@v5
+ with:
+ context: .
+ file: ./docker/Dockerfile.database
+ tags: litellm-test:${{ github.sha }}
+ load: true
+ cache-from: type=gha
+ cache-to: type=gha,mode=max
+
+ - name: Start LiteLLM container with SERVER_ROOT_PATH
+ run: |
+ docker run -d \
+ --name litellm-test \
+ -p 4000:4000 \
+ -e SERVER_ROOT_PATH="${{ matrix.root_path }}" \
+ -e LITELLM_MASTER_KEY="sk-1234" \
+ litellm-test:${{ github.sha }} \
+ --detailed_debug
+
+ - name: Wait for container to be healthy
+ run: |
+ echo "Waiting for LiteLLM to start..."
+ max_attempts=30
+ attempt=0
+
+ while [ $attempt -lt $max_attempts ]; do
+ if docker logs litellm-test 2>&1 | grep -q "Uvicorn running"; then
+ echo "LiteLLM started successfully"
+ break
+ fi
+ attempt=$((attempt + 1))
+ echo "Attempt $attempt/$max_attempts - waiting for server to start..."
+ sleep 2
+ done
+
+ if [ $attempt -eq $max_attempts ]; then
+ echo "Server failed to start within timeout"
+ docker logs litellm-test
+ exit 1
+ fi
+
+ sleep 5
+
+ - name: Show container logs
+ if: always()
+ run: docker logs litellm-test
+
+ - name: Test UI endpoint with root path
+ run: |
+ ROOT_PATH="${{ matrix.root_path }}"
+ echo "Testing UI at: http://localhost:4000${ROOT_PATH}/ui/"
+
+ for i in 1 2 3; do
+ content=$(curl -sL --max-time 5 -H "Authorization: Bearer sk-1234" "http://localhost:4000${ROOT_PATH}/ui/")
+ if echo "$content" | grep -q -E "(html|>LP: Request with beta headers
+ Note over CC,LP: anthropic-beta: header1,header2,header3
+
+ LP->>Provider: Forward ALL headers (no validation)
+ Note over LP,Provider: anthropic-beta: header1,header2,header3
+
+ Provider-->>LP: ❌ Error: invalid beta flag
+ LP-->>CC: Request fails
+```
+
+Requests succeeded for Anthropic (native support) but failed for other providers when Claude Code sent headers those providers didn't support.
+
+---
+
+## Root cause
+
+LiteLLM lacked provider-specific beta header validation. When Claude Code introduced new beta features or sent headers that specific providers didn't support, those headers were blindly forwarded, causing provider API errors.
+
+---
+
+## Remediation
+
+| # | Action | Status | Code |
+|---|---|---|---|
+| 1 | Create `anthropic_beta_headers_config.json` with provider-specific mappings | ✅ Done | [`anthropic_beta_headers_config.json`](https://github.com/BerriAI/litellm/blob/main/litellm/anthropic_beta_headers_config.json) |
+| 2 | Implement strict validation: headers must be explicitly mapped to be forwarded | ✅ Done | [`litellm_logging.py`](https://github.com/BerriAI/litellm/blob/main/litellm/litellm_core_utils/litellm_logging.py) |
+| 3 | Add `/reload/anthropic_beta_headers` endpoint for dynamic config updates | ✅ Done | Proxy management endpoints |
+| 4 | Add `/schedule/anthropic_beta_headers_reload` for automatic periodic updates | ✅ Done | Proxy management endpoints |
+| 5 | Support `LITELLM_ANTHROPIC_BETA_HEADERS_URL` for custom config sources | ✅ Done | Environment configuration |
+| 6 | Support `LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS` for air-gapped deployments | ✅ Done | Environment configuration |
+
+Now LiteLLM validates and transforms headers per-provider:
+
+```mermaid
+sequenceDiagram
+ participant CC as Claude Code
+ participant LP as LiteLLM (new behavior)
+ participant Config as Beta Headers Config
+ participant Provider as Provider (Bedrock/Azure/Vertex)
+
+ CC->>LP: Request with beta headers
+ Note over CC,LP: anthropic-beta: header1,header2,header3
+
+ LP->>Config: Load header mapping for provider
+ Config-->>LP: Returns mapping (header→value or null)
+
+ Note over LP: Validate & Transform:
1. Check if header exists in mapping
2. Filter out null values
3. Map to provider-specific names
+
+ LP->>Provider: Request with filtered & mapped headers
+ Note over LP,Provider: anthropic-beta: mapped-header2
(header1, header3 filtered out)
+
+ Provider-->>LP: ✅ Success response
+ LP-->>CC: Response
+```
+
+---
+
+## Dynamic configuration updates
+
+A key improvement is zero-downtime configuration updates. When Anthropic releases new beta features, users can update their configuration without restarting:
+
+```bash
+# Manually trigger reload (no restart needed)
+curl -X POST "https://your-proxy-url/reload/anthropic_beta_headers" \
+ -H "Authorization: Bearer YOUR_ADMIN_TOKEN"
+
+# Or schedule automatic reloads every 24 hours
+curl -X POST "https://your-proxy-url/schedule/anthropic_beta_headers_reload?hours=24" \
+ -H "Authorization: Bearer YOUR_ADMIN_TOKEN"
+```
+
+This prevents future incidents where Claude Code introduces new headers before LiteLLM configuration is updated.
+
+---
+
+## Configuration format
+
+The `anthropic_beta_headers_config.json` file maps input headers to provider-specific output headers:
+
+```json
+{
+ "description": "Mapping of Anthropic beta headers for each provider.",
+ "anthropic": {
+ "advanced-tool-use-2025-11-20": "advanced-tool-use-2025-11-20",
+ "computer-use-2025-01-24": "computer-use-2025-01-24"
+ },
+ "bedrock_converse": {
+ "advanced-tool-use-2025-11-20": null,
+ "computer-use-2025-01-24": "computer-use-2025-01-24"
+ },
+ "azure_ai": {
+ "advanced-tool-use-2025-11-20": "advanced-tool-use-2025-11-20",
+ "computer-use-2025-01-24": "computer-use-2025-01-24"
+ }
+}
+```
+
+**Validation rules:**
+1. Headers must exist in the mapping for the target provider
+2. Headers with `null` values are filtered out (unsupported)
+3. Header names can be transformed per-provider (e.g., Bedrock uses different names for some features)
+
+---
+
+## Resolution steps for users
+
+For users still experiencing issues, update to the latest LiteLLM version if < v1.81.11-nightly:
+
+```bash
+pip install --upgrade litellm
+```
+
+Or manually reload the configuration without restarting:
+
+```bash
+curl -X POST "https://your-proxy-url/reload/anthropic_beta_headers" \
+ -H "Authorization: Bearer YOUR_ADMIN_TOKEN"
+```
+
+---
+
+## Related documentation
+
+- [Managing Anthropic Beta Headers](../proxy/sync_anthropic_beta_headers.md) - Complete configuration guide
+- [`anthropic_beta_headers_config.json`](https://github.com/BerriAI/litellm/blob/main/litellm/anthropic_beta_headers_config.json) - Current configuration file
diff --git a/docs/my-website/blog/claude_sonnet_4_6/index.md b/docs/my-website/blog/claude_sonnet_4_6/index.md
new file mode 100644
index 00000000000..df54fa09792
--- /dev/null
+++ b/docs/my-website/blog/claude_sonnet_4_6/index.md
@@ -0,0 +1,283 @@
+---
+slug: claude_sonnet_4_6
+title: "Day 0 Support: Claude Sonnet 4.6"
+date: 2026-02-17T10:00:00
+authors:
+ - name: Ishaan Jaff
+ title: "CTO, LiteLLM"
+ url: https://www.linkedin.com/in/reffajnaahsi/
+ image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
+ - name: Krrish Dholakia
+ title: "CEO, LiteLLM"
+ url: https://www.linkedin.com/in/krish-d/
+ image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
+description: "Day 0 support for Claude Sonnet 4.6 on LiteLLM AI Gateway - use across Anthropic, Azure, Vertex AI, and Bedrock."
+tags: [anthropic, claude, sonnet 4.6]
+hide_table_of_contents: false
+---
+
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+LiteLLM now supports Claude Sonnet 4.6 on Day 0. Use it across Anthropic, Azure, Vertex AI, and Bedrock through the LiteLLM AI Gateway.
+
+## Docker Image
+
+```bash
+docker pull ghcr.io/berriai/litellm:v1.81.3-stable.sonnet-4-6
+```
+
+## Usage - Anthropic
+
+
+
+
+**1. Setup config.yaml**
+
+```yaml
+model_list:
+ - model_name: claude-sonnet-4-6
+ litellm_params:
+ model: anthropic/claude-sonnet-4-6
+ api_key: os.environ/ANTHROPIC_API_KEY
+```
+
+**2. Start the proxy**
+
+```bash
+docker run -d \
+ -p 4000:4000 \
+ -e ANTHROPIC_API_KEY=$ANTHROPIC_API_KEY \
+ -v $(pwd)/config.yaml:/app/config.yaml \
+ ghcr.io/berriai/litellm:v1.81.3-stable.sonnet-4-6 \
+ --config /app/config.yaml
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer $LITELLM_KEY' \
+--data '{
+ "model": "claude-sonnet-4-6",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ]
+}'
+```
+
+
+
+
+
+```python
+from litellm import completion
+
+response = completion(
+ model="anthropic/claude-sonnet-4-6",
+ messages=[{"role": "user", "content": "what llm are you"}]
+)
+print(response.choices[0].message.content)
+```
+
+
+
+
+## Usage - Azure
+
+
+
+
+**1. Setup config.yaml**
+
+```yaml
+model_list:
+ - model_name: claude-sonnet-4-6
+ litellm_params:
+ model: azure_ai/claude-sonnet-4-6
+ api_key: os.environ/AZURE_AI_API_KEY
+ api_base: os.environ/AZURE_AI_API_BASE # https://.services.ai.azure.com
+```
+
+**2. Start the proxy**
+
+```bash
+docker run -d \
+ -p 4000:4000 \
+ -e AZURE_AI_API_KEY=$AZURE_AI_API_KEY \
+ -e AZURE_AI_API_BASE=$AZURE_AI_API_BASE \
+ -v $(pwd)/config.yaml:/app/config.yaml \
+ ghcr.io/berriai/litellm:v1.81.3-stable.sonnet-4-6 \
+ --config /app/config.yaml
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer $LITELLM_KEY' \
+--data '{
+ "model": "claude-sonnet-4-6",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ]
+}'
+```
+
+
+
+
+
+```python
+from litellm import completion
+
+response = completion(
+ model="azure_ai/claude-sonnet-4-6",
+ api_key="your-azure-api-key",
+ api_base="https://.services.ai.azure.com",
+ messages=[{"role": "user", "content": "what llm are you"}]
+)
+print(response.choices[0].message.content)
+```
+
+
+
+
+## Usage - Vertex AI
+
+
+
+
+**1. Setup config.yaml**
+
+```yaml
+model_list:
+ - model_name: claude-sonnet-4-6
+ litellm_params:
+ model: vertex_ai/claude-sonnet-4-6
+ vertex_project: os.environ/VERTEX_PROJECT
+ vertex_location: us-east5
+```
+
+**2. Start the proxy**
+
+```bash
+docker run -d \
+ -p 4000:4000 \
+ -e VERTEX_PROJECT=$VERTEX_PROJECT \
+ -e GOOGLE_APPLICATION_CREDENTIALS=/app/credentials.json \
+ -v $(pwd)/config.yaml:/app/config.yaml \
+ -v $(pwd)/credentials.json:/app/credentials.json \
+ ghcr.io/berriai/litellm:v1.81.3-stable.sonnet-4-6 \
+ --config /app/config.yaml
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer $LITELLM_KEY' \
+--data '{
+ "model": "claude-sonnet-4-6",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ]
+}'
+```
+
+
+
+
+
+```python
+from litellm import completion
+
+response = completion(
+ model="vertex_ai/claude-sonnet-4-6",
+ vertex_project="your-project-id",
+ vertex_location="us-east5",
+ messages=[{"role": "user", "content": "what llm are you"}]
+)
+print(response.choices[0].message.content)
+```
+
+
+
+
+## Usage - Bedrock
+
+
+
+
+**1. Setup config.yaml**
+
+```yaml
+model_list:
+ - model_name: claude-sonnet-4-6
+ litellm_params:
+ model: bedrock/anthropic.claude-sonnet-4-6-v1
+ aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
+ aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
+ aws_region_name: us-east-1
+```
+
+**2. Start the proxy**
+
+```bash
+docker run -d \
+ -p 4000:4000 \
+ -e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \
+ -e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \
+ -v $(pwd)/config.yaml:/app/config.yaml \
+ ghcr.io/berriai/litellm:v1.81.3-stable.sonnet-4-6 \
+ --config /app/config.yaml
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer $LITELLM_KEY' \
+--data '{
+ "model": "claude-sonnet-4-6",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ]
+}'
+```
+
+
+
+
+
+```python
+from litellm import completion
+
+response = completion(
+ model="bedrock/anthropic.claude-sonnet-4-6-v1",
+ aws_access_key_id="your-access-key",
+ aws_secret_access_key="your-secret-key",
+ aws_region_name="us-east-1",
+ messages=[{"role": "user", "content": "what llm are you"}]
+)
+print(response.choices[0].message.content)
+```
+
+
+
diff --git a/docs/my-website/docs/evals_api.md b/docs/my-website/docs/evals_api.md
new file mode 100644
index 00000000000..bb66e9fdc0a
--- /dev/null
+++ b/docs/my-website/docs/evals_api.md
@@ -0,0 +1,441 @@
+# /evals
+
+LiteLLM Proxy supports OpenAI's Evaluations (Evals) API, allowing you to create, manage, and run evaluations to measure model performance against defined testing criteria.
+
+## What are Evals?
+
+OpenAI Evals API provides a structured way to:
+- **Create Evaluations**: Define testing criteria and data sources for evaluating model outputs
+- **Run Evaluations**: Execute evaluations against specific models and datasets
+- **Track Results**: Monitor evaluation progress and review detailed results
+
+## Quick Start
+
+### Setup LiteLLM Proxy
+
+First, start your LiteLLM Proxy server:
+
+```bash
+litellm --config config.yaml
+
+# Proxy will run on http://localhost:4000
+```
+
+### Initialize OpenAI Client
+
+```python
+from openai import OpenAI
+
+# Point to your LiteLLM Proxy
+client = OpenAI(
+ api_key="sk-1234", # Your LiteLLM proxy API key
+ base_url="http://localhost:4000" # Your proxy URL
+)
+```
+
+
+For async operations:
+
+```python
+from openai import AsyncOpenAI
+
+client = AsyncOpenAI(
+ api_key="sk-1234",
+ base_url="http://localhost:4000"
+)
+```
+
+---
+
+## Evaluation Management
+
+### Create an Evaluation
+
+Create an evaluation with testing criteria and data source configuration.
+
+#### Example: Sentiment Classification Eval
+
+```python
+from openai import OpenAI
+
+client = OpenAI(
+ api_key="sk-1234",
+ base_url="http://localhost:4000"
+)
+
+# Create evaluation with label model grader
+eval_obj = client.evals.create(
+ name="Sentiment Classification",
+ data_source_config={
+ "type": "stored_completions",
+ "metadata": {"usecase": "chatbot"}
+ },
+ testing_criteria=[
+ {
+ "type": "label_model",
+ "model": "gpt-4o-mini",
+ "input": [
+ {
+ "role": "developer",
+ "content": "Classify the sentiment of the following statement as one of 'positive', 'neutral', or 'negative'"
+ },
+ {
+ "role": "user",
+ "content": "Statement: {{item.input}}"
+ }
+ ],
+ "passing_labels": ["positive"],
+ "labels": ["positive", "neutral", "negative"],
+ "name": "Sentiment Grader"
+ }
+ ]
+)
+
+# Note: If you want to use model-specific credentials for this evaluation, you can specify the model name in the extra body parameters.
+
+print(f"Created eval: {eval_obj.id}")
+print(f"Eval name: {eval_obj.name}")
+```
+
+#### Example: Push Notifications Summarizer Monitoring
+
+This example shows how to monitor prompt changes for regressions in a push notifications summarizer:
+
+```python
+from openai import AsyncOpenAI
+
+client = AsyncOpenAI(
+ api_key="sk-1234",
+ base_url="http://localhost:4000"
+)
+
+# Define data source for stored completions
+data_source_config = {
+ "type": "stored_completions",
+ "metadata": {
+ "usecase": "push_notifications_summarizer"
+ }
+}
+
+# Define grader criteria
+GRADER_DEVELOPER_PROMPT = """
+Label the following push notification summary as either correct or incorrect.
+The push notification and the summary will be provided below.
+A good push notification summary is concise and snappy.
+If it is good, then label it as correct, if not, then incorrect.
+"""
+
+GRADER_TEMPLATE_PROMPT = """
+Push notifications: {{item.input}}
+Summary: {{sample.output_text}}
+"""
+
+push_notification_grader = {
+ "name": "Push Notification Summary Grader",
+ "type": "label_model",
+ "model": "gpt-4o-mini",
+ "input": [
+ {
+ "role": "developer",
+ "content": GRADER_DEVELOPER_PROMPT,
+ },
+ {
+ "role": "user",
+ "content": GRADER_TEMPLATE_PROMPT,
+ },
+ ],
+ "passing_labels": ["correct"],
+ "labels": ["correct", "incorrect"],
+}
+
+# Create the evaluation
+eval_result = await client.evals.create(
+ name="Push Notification Completion Monitoring",
+ metadata={"description": "This eval monitors completions"},
+ data_source_config=data_source_config,
+ testing_criteria=[push_notification_grader],
+)
+
+eval_id = eval_result.id
+print(f"Created eval: {eval_id}")
+```
+
+### List Evaluations
+
+Retrieve a list of all your evaluations with pagination support.
+
+```python
+# List all evaluations
+evals_response = client.evals.list(
+ limit=20,
+ order="desc"
+)
+
+for eval in evals_response.data:
+ print(f"Eval ID: {eval.id}, Name: {eval.name}")
+
+# Check if there are more evals
+if evals_response.has_more:
+ # Fetch next page
+ next_evals = client.evals.list(
+ after=evals_response.last_id,
+ limit=20
+ )
+```
+
+### Get a Specific Evaluation
+
+Retrieve details of a specific evaluation by ID.
+
+```python
+eval = client.evals.retrieve(
+ eval_id="eval_abc123"
+)
+
+print(f"Eval ID: {eval.id}")
+print(f"Name: {eval.name}")
+print(f"Data Source: {eval.data_source_config}")
+print(f"Testing Criteria: {eval.testing_criteria}")
+```
+
+### Update an Evaluation
+
+Update evaluation metadata or name.
+
+```python
+updated_eval = client.evals.update(
+ eval_id="eval_abc123",
+ name="Updated Evaluation Name",
+ metadata={
+ "version": "2.0",
+ "updated_by": "user@example.com"
+ }
+)
+
+print(f"Updated eval: {updated_eval.name}")
+```
+
+### Delete an Evaluation
+
+Permanently delete an evaluation.
+
+```python
+delete_response = client.evals.delete(
+ eval_id="eval_abc123"
+)
+
+print(f"Deleted: {delete_response.deleted}") # True
+```
+
+---
+
+## Evaluation Runs
+
+### Create a Run
+
+Execute an evaluation by creating a run. The run processes your data through the model and applies testing criteria.
+
+#### Using Stored Completions
+
+First, generate some test data by making chat completions with metadata:
+
+```python
+from openai import AsyncOpenAI
+import asyncio
+
+client = AsyncOpenAI(
+ api_key="sk-1234",
+ base_url="http://localhost:4000"
+)
+
+# Generate test data with different prompt versions
+push_notification_data = [
+ """
+- New message from Sarah: "Can you call me later?"
+- Your package has been delivered!
+- Flash sale: 20% off electronics for the next 2 hours!
+""",
+ """
+- Weather alert: Thunderstorm expected in your area.
+- Reminder: Doctor's appointment at 3 PM.
+- John liked your photo on Instagram.
+"""
+]
+
+PROMPTS = [
+ (
+ """
+ You are a helpful assistant that summarizes push notifications.
+ You are given a list of push notifications and you need to collapse them into a single one.
+ Output only the final summary, nothing else.
+ """,
+ "v1"
+ ),
+ (
+ """
+ You are a helpful assistant that summarizes push notifications.
+ You are given a list of push notifications and you need to collapse them into a single one.
+ The summary should be longer than it needs to be and include more information than is necessary.
+ Output only the final summary, nothing else.
+ """,
+ "v2"
+ )
+]
+
+# Create completions with metadata for tracking
+tasks = []
+for notifications in push_notification_data:
+ for (prompt, version) in PROMPTS:
+ tasks.append(client.chat.completions.create(
+ model="gpt-4o-mini",
+ messages=[
+ {"role": "developer", "content": prompt},
+ {"role": "user", "content": notifications},
+ ],
+ metadata={
+ "prompt_version": version,
+ "usecase": "push_notifications_summarizer"
+ }
+ ))
+
+await asyncio.gather(*tasks)
+```
+
+Now create runs to evaluate different prompt versions:
+
+```python
+# Grade prompt_version=v1
+eval_run_result = await client.evals.runs.create(
+ eval_id=eval_id,
+ name="v1-run",
+ data_source={
+ "type": "completions",
+ "source": {
+ "type": "stored_completions",
+ "metadata": {
+ "prompt_version": "v1",
+ }
+ }
+ }
+)
+
+print(f"Run ID: {eval_run_result.id}")
+print(f"Status: {eval_run_result.status}")
+print(f"Report URL: {eval_run_result.report_url}")
+
+# Grade prompt_version=v2
+eval_run_result_v2 = await client.evals.runs.create(
+ eval_id=eval_id,
+ name="v2-run",
+ data_source={
+ "type": "completions",
+ "source": {
+ "type": "stored_completions",
+ "metadata": {
+ "prompt_version": "v2",
+ }
+ }
+ }
+)
+
+print(f"Run ID: {eval_run_result_v2.id}")
+print(f"Report URL: {eval_run_result_v2.report_url}")
+```
+
+#### Using Completions with Different Models
+
+Test how different models perform on the same inputs:
+
+```python
+# Test with GPT-4o using stored completions as input
+tasks = []
+for prompt_version in ["v1", "v2"]:
+ tasks.append(client.evals.runs.create(
+ eval_id=eval_id,
+ name=f"gpt-4o-run-{prompt_version}",
+ data_source={
+ "type": "completions",
+ "input_messages": {
+ "type": "item_reference",
+ "item_reference": "item.input",
+ },
+ "model": "gpt-4o",
+ "source": {
+ "type": "stored_completions",
+ "metadata": {
+ "prompt_version": prompt_version,
+ }
+ }
+ }
+ ))
+
+results = await asyncio.gather(*tasks)
+for run in results:
+ print(f"Report URL: {run.report_url}")
+```
+
+### List Runs
+
+Get all runs for a specific evaluation.
+
+```python
+# List all runs for an evaluation
+runs_response = client.evals.runs.list(
+ eval_id="eval_abc123",
+ limit=20,
+ order="desc"
+)
+
+for run in runs_response.data:
+ print(f"Run ID: {run.id}")
+ print(f"Status: {run.status}")
+ print(f"Name: {run.name}")
+ if run.result_counts:
+ print(f"Results: {run.result_counts.passed}/{run.result_counts.total} passed")
+```
+
+### Get Run Details
+
+Retrieve detailed information about a specific run, including results.
+
+```python
+run = client.evals.runs.retrieve(
+ eval_id="eval_abc123",
+ run_id="run_def456"
+)
+
+print(f"Run ID: {run.id}")
+print(f"Status: {run.status}")
+print(f"Started: {run.started_at}")
+print(f"Completed: {run.completed_at}")
+
+# Check results
+if run.result_counts:
+ print(f"\nOverall Results:")
+ print(f"Total: {run.result_counts.total}")
+ print(f"Passed: {run.result_counts.passed}")
+ print(f"Failed: {run.result_counts.failed}")
+ print(f"Error: {run.result_counts.errored}")
+
+# Per-criteria results
+if run.per_testing_criteria_results:
+ for criteria_result in run.per_testing_criteria_results:
+ print(f"\nCriteria {criteria_result.testing_criteria_index}:")
+ print(f" Passed: {criteria_result.result_counts.passed}")
+ print(f" Average Score: {criteria_result.average_score}")
+```
+
+### Delete a Run
+
+Permanently delete a run and its results.
+
+```python
+delete_response = await client.evals.runs.delete(
+ eval_id="eval_abc123",
+ run_id="run_def456"
+)
+
+print(f"Deleted: {delete_response.deleted}") # True
+print(f"Run ID: {delete_response.run_id}")
+```
+
diff --git a/docs/my-website/docs/projects/openai-agents.md b/docs/my-website/docs/projects/openai-agents.md
index 95a2191b883..86983e7e510 100644
--- a/docs/my-website/docs/projects/openai-agents.md
+++ b/docs/my-website/docs/projects/openai-agents.md
@@ -1,22 +1,121 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
# OpenAI Agents SDK
-The [OpenAI Agents SDK](https://github.com/openai/openai-agents-python) is a lightweight framework for building multi-agent workflows.
-It includes an official LiteLLM extension that lets you use any of the 100+ supported providers (Anthropic, Gemini, Mistral, Bedrock, etc.)
+Use OpenAI Agents SDK with any LLM provider through LiteLLM Proxy.
+
+The [OpenAI Agents SDK](https://github.com/openai/openai-agents-python) is a lightweight framework for building multi-agent workflows. It includes an official LiteLLM extension that lets you use any of the 100+ supported providers.
+
+## Quick Start
+
+### 1. Install Dependencies
+
+```bash
+pip install "openai-agents[litellm]"
+```
+
+### 2. Add Model to Config
+
+```yaml title="config.yaml"
+model_list:
+ - model_name: gpt-4o
+ litellm_params:
+ model: "openai/gpt-4o"
+ api_key: "os.environ/OPENAI_API_KEY"
+
+ - model_name: claude-sonnet
+ litellm_params:
+ model: "anthropic/claude-3-5-sonnet-20241022"
+ api_key: "os.environ/ANTHROPIC_API_KEY"
+
+ - model_name: gemini-pro
+ litellm_params:
+ model: "gemini/gemini-2.0-flash-exp"
+ api_key: "os.environ/GEMINI_API_KEY"
+```
+
+### 3. Start LiteLLM Proxy
+
+```bash
+litellm --config config.yaml
+```
+
+### 4. Use with Proxy
+
+
+
```python
from agents import Agent, Runner
from agents.extensions.models.litellm_model import LitellmModel
+# Point to LiteLLM proxy
agent = Agent(
name="Assistant",
instructions="You are a helpful assistant.",
- model=LitellmModel(model="provider/model-name")
+ model=LitellmModel(
+ model="claude-sonnet", # Model from config.yaml
+ api_key="sk-1234", # LiteLLM API key
+ base_url="http://localhost:4000"
+ )
)
-result = Runner.run_sync(agent, "your_prompt_here")
-print("Result:", result.final_output)
+result = await Runner.run(agent, "What is LiteLLM?")
+print(result.final_output)
```
-- [GitHub](https://github.com/openai/openai-agents-python)
-- [LiteLLM Extension Docs](https://openai.github.io/openai-agents-python/ref/extensions/litellm/)
+
+
+
+```python
+from agents import Agent, Runner
+from agents.extensions.models.litellm_model import LitellmModel
+
+# Use any provider directly
+agent = Agent(
+ name="Assistant",
+ instructions="You are a helpful assistant.",
+ model=LitellmModel(
+ model="anthropic/claude-3-5-sonnet-20241022",
+ api_key="your-anthropic-key"
+ )
+)
+
+result = await Runner.run(agent, "What is LiteLLM?")
+print(result.final_output)
+```
+
+
+
+
+## Track Usage
+
+Enable usage tracking to monitor token consumption:
+
+```python
+from agents import Agent, ModelSettings
+from agents.extensions.models.litellm_model import LitellmModel
+
+agent = Agent(
+ name="Assistant",
+ model=LitellmModel(model="claude-sonnet", api_key="sk-1234"),
+ model_settings=ModelSettings(include_usage=True)
+)
+
+result = await Runner.run(agent, "Hello")
+print(result.context_wrapper.usage) # Token counts
+```
+
+## Environment Variables
+
+| Variable | Value | Description |
+|----------|-------|-------------|
+| `LITELLM_BASE_URL` | `http://localhost:4000` | LiteLLM proxy URL |
+| `LITELLM_API_KEY` | `sk-1234` | Your LiteLLM API key |
+
+## Related Resources
+
+- [OpenAI Agents SDK Documentation](https://openai.github.io/openai-agents-python/)
+- [LiteLLM Extension Docs](https://openai.github.io/openai-agents-python/models/litellm/)
+- [LiteLLM Proxy Quick Start](../proxy/quick_start)
diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md
index ac554b09174..a2371232302 100644
--- a/docs/my-website/docs/proxy/config_settings.md
+++ b/docs/my-website/docs/proxy/config_settings.md
@@ -450,6 +450,7 @@ router_settings:
| BATCH_STATUS_POLL_INTERVAL_SECONDS | Interval in seconds for polling batch status. Default is 3600 (1 hour)
| BATCH_STATUS_POLL_MAX_ATTEMPTS | Maximum number of attempts for polling batch status. Default is 24 (for 24 hours)
| BEDROCK_MAX_POLICY_SIZE | Maximum size for Bedrock policy. Default is 75
+| BEDROCK_MIN_THINKING_BUDGET_TOKENS | Minimum thinking budget in tokens for Bedrock reasoning models. Bedrock returns a 400 error if budget_tokens is below this value. Requests with lower values are clamped to this minimum. Default is 1024
| BERRISPEND_ACCOUNT_ID | Account ID for BerriSpend service
| BRAINTRUST_API_KEY | API key for Braintrust integration
| BRAINTRUST_API_BASE | Base URL for Braintrust API. Default is https://api.braintrustdata.com/v1
@@ -602,7 +603,6 @@ router_settings:
| EMAIL_BUDGET_ALERT_TTL | Time-to-live for budget alert deduplication in seconds. Default is 86400 (24 hours)
| ENKRYPTAI_API_BASE | Base URL for EnkryptAI Guardrails API. **Default is https://api.enkryptai.com**
| ENKRYPTAI_API_KEY | API key for EnkryptAI Guardrails service
-| EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING | Flag to enable new multi-instance rate limiting. **Default is False**
| FIREWORKS_AI_4_B | Size parameter for Fireworks AI 4B model. Default is 4
| FIREWORKS_AI_16_B | Size parameter for Fireworks AI 16B model. Default is 16
| FIREWORKS_AI_56_B_MOE | Size parameter for Fireworks AI 56B MOE model. Default is 56
@@ -769,6 +769,7 @@ router_settings:
| LITELM_ENVIRONMENT | Environment of LiteLLM Instance, used by logging services. Currently only used by DeepEval.
| LITELLM_KEY_ROTATION_ENABLED | Enable auto-key rotation for LiteLLM (boolean). Default is false.
| LITELLM_KEY_ROTATION_CHECK_INTERVAL_SECONDS | Interval in seconds for how often to run job that auto-rotates keys. Default is 86400 (24 hours).
+| LITELLM_KEY_ROTATION_GRACE_PERIOD | Duration to keep old key valid after rotation (e.g. "24h", "2d"). Default is empty (immediate revoke). Used for scheduled rotations and as fallback when not specified in regenerate request.
| LITELLM_LICENSE | License key for LiteLLM usage
| LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS | Set to `True` to use the local bundled Anthropic beta headers config only, disabling remote fetching. Default is `False`
| LITELLM_LOCAL_MODEL_COST_MAP | Local configuration for model cost mapping in LiteLLM
diff --git a/docs/my-website/docs/proxy/logging.md b/docs/my-website/docs/proxy/logging.md
index 56fb420e6cf..1abb127dfda 100644
--- a/docs/my-website/docs/proxy/logging.md
+++ b/docs/my-website/docs/proxy/logging.md
@@ -1338,6 +1338,7 @@ litellm_settings:
s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for S3
s3_path: my-test-path # [OPTIONAL] set path in bucket you want to write logs to
s3_endpoint_url: https://s3.amazonaws.com # [OPTIONAL] S3 endpoint URL, if you want to use Backblaze/cloudflare s3 buckets
+ s3_use_virtual_hosted_style: false # [OPTIONAL] use virtual-hosted-style URLs (bucket.endpoint/key) instead of path-style (endpoint/bucket/key). Useful for S3-compatible services like MinIO
s3_strip_base64_files: false # [OPTIONAL] remove base64 files before storing in s3
```
diff --git a/docs/my-website/docs/proxy/release_cycle.md b/docs/my-website/docs/proxy/release_cycle.md
index 10dd6d8b3c5..b3e056b0243 100644
--- a/docs/my-website/docs/proxy/release_cycle.md
+++ b/docs/my-website/docs/proxy/release_cycle.md
@@ -22,4 +22,10 @@ Stable releases come out every week (typically Sunday)
- 'patch' bumps: extremely minor addition that doesn't affect any existing functionality or add any user-facing features. (e.g. a 'created_at' column in a database table)
- 'minor' bumps: add a new feature or a new database table that is backward compatible.
-- 'major' bumps: break backward compatibility.
\ No newline at end of file
+- 'major' bumps: break backward compatibility.
+
+### Enterprise Support
+
+
+- Stable releases come out every week. Once a new one is available, we no longer provide support for an older one.
+- If there is a MAJOR change (according to semvar conventions - e.g. 1.x.x -> 2.x.x), we can provide support for upto 90 days on the prior stable image.
diff --git a/docs/my-website/docs/proxy/team_budgets.md b/docs/my-website/docs/proxy/team_budgets.md
index 03d18797133..01b07f23a33 100644
--- a/docs/my-website/docs/proxy/team_budgets.md
+++ b/docs/my-website/docs/proxy/team_budgets.md
@@ -8,7 +8,6 @@ import TabItem from '@theme/TabItem';
# Pre-Requisites
- You must set up a Postgres database (e.g. Supabase, Neon, etc.)
-- To enable team member rate limits, set the environment variable `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING=true` **before starting the proxy server**. Without this, team member rate limits will not be enforced.
## Default Budget for Auto-Generated JWT Teams
diff --git a/docs/my-website/docs/proxy/users.md b/docs/my-website/docs/proxy/users.md
index a389f0bd443..8517db51a8f 100644
--- a/docs/my-website/docs/proxy/users.md
+++ b/docs/my-website/docs/proxy/users.md
@@ -68,13 +68,6 @@ You can:
**Step-by step tutorial on setting, resetting budgets on Teams here (API or using Admin UI)**
-> **Prerequisite:**
-> To enable team member rate limits, you must set the environment variable `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING=true` before starting the proxy server. Without this, team member rate limits will not be enforced.
-
-👉 [https://docs.litellm.ai/docs/proxy/team_budgets](https://docs.litellm.ai/docs/proxy/team_budgets)
-
-:::
-
#### **Add budgets to teams**
```shell
@@ -822,12 +815,10 @@ Expected Response:
}
```
-### [BETA] Multi-instance rate limiting
+### Multi-instance rate limiting
-Enable multi-instance rate limiting with the env var `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"`
**Important Notes:**
-- Setting `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"` is required for team member rate limits to function, not just for multi-instance scenarios.
- **Rate limits do not apply to proxy admin users.**
- When testing rate limits, use internal user roles (non-admin) to ensure limits are enforced as expected.
diff --git a/docs/my-website/docs/proxy/virtual_keys.md b/docs/my-website/docs/proxy/virtual_keys.md
index 38ff4ede280..c74aa75ff4a 100644
--- a/docs/my-website/docs/proxy/virtual_keys.md
+++ b/docs/my-website/docs/proxy/virtual_keys.md
@@ -549,11 +549,14 @@ curl 'http://localhost:4000/key/sk-1234/regenerate' \
"models": [
"gpt-4",
"gpt-3.5-turbo"
- ]
+ ],
+ "grace_period": "48h"
}'
```
+**Grace period (optional)**: Set `grace_period` (e.g. `"24h"`, `"2d"`, `"1w"`) to keep the old key valid for a transitional period. Both old and new keys work until the grace period elapses, enabling seamless cutover without production downtime. Omitted or empty = immediate revoke. Can also be set via `LITELLM_KEY_ROTATION_GRACE_PERIOD` env var for scheduled rotations.
+
**Read More**
- [Write rotated keys to secrets manager](https://docs.litellm.ai/docs/secret#aws-secret-manager)
@@ -640,11 +643,13 @@ Set these environment variables when starting the proxy:
|----------|-------------|---------|
| `LITELLM_KEY_ROTATION_ENABLED` | Enable the rotation worker | `false` |
| `LITELLM_KEY_ROTATION_CHECK_INTERVAL_SECONDS` | How often to scan for keys to rotate (in seconds) | `86400` (24 hours) |
+| `LITELLM_KEY_ROTATION_GRACE_PERIOD` | Duration to keep old key valid after rotation (e.g. `24h`, `2d`) | `""` (immediate revoke) |
**Example:**
```bash
export LITELLM_KEY_ROTATION_ENABLED=true
export LITELLM_KEY_ROTATION_CHECK_INTERVAL_SECONDS=3600 # Check every hour
+export LITELLM_KEY_ROTATION_GRACE_PERIOD=48h # Keep old key valid for 48h during cutover
litellm --config config.yaml
```
diff --git a/docs/my-website/release_notes/v1.81.12.md b/docs/my-website/release_notes/v1.81.12.md
index 4f94cef6b68..c68b23488c0 100644
--- a/docs/my-website/release_notes/v1.81.12.md
+++ b/docs/my-website/release_notes/v1.81.12.md
@@ -48,6 +48,13 @@ pip install litellm==1.81.12.rc1
- **Responses API `shell` Tool & `context_management` support** - [Server-side context management (compaction) and Shell tool support for the OpenAI Responses API](../../docs/response_api)
- **Access Groups** - [Create access groups to manage model, MCP server, and agent access across teams and keys](../../docs/proxy/access_groups)
- **50+ New Bedrock Regional Model Entries** - DeepSeek V3.2, MiniMax M2.1, Kimi K2.5, Qwen3 Coder Next, and NVIDIA Nemotron Nano across multiple regions
+- **Add Semgrep & fix OOMs** - [Static analysis rules and out-of-memory fixes](#add-semgrep--fix-ooms) - [PR #20912](https://github.com/BerriAI/litellm/pull/20912)
+
+---
+
+## Add Semgrep & fix OOMs
+
+This release fixes out-of-memory (OOM) risks from unbounded `asyncio.Queue()` usage. Log queues (e.g. GCS bucket) and DB spend-update queues were previously unbounded and could grow without limit under load. They now use a configurable max size (`LITELLM_ASYNCIO_QUEUE_MAXSIZE`, default 1000); when full, queues flush immediately to make room instead of growing memory. A Semgrep rule (`.semgrep/rules/python/unbounded-memory.yml`) was added to flag similar unbounded-memory patterns in future code. [PR #20912](https://github.com/BerriAI/litellm/pull/20912)
---
diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js
index 4efb2475755..f1376a46159 100644
--- a/docs/my-website/sidebars.js
+++ b/docs/my-website/sidebars.js
@@ -176,6 +176,7 @@ const sidebars = {
"tutorials/copilotkit_sdk",
"tutorials/google_adk",
"tutorials/livekit_xai_realtime",
+ "projects/openai-agents"
]
},
@@ -572,6 +573,7 @@ const sidebars = {
"proxy/managed_finetuning",
]
},
+ "evals_api",
"generateContent",
"apply_guardrail",
"bedrock_invoke",
@@ -1125,6 +1127,11 @@ const sidebars = {
type: "category",
label: "Blog",
items: [
+ {
+ type: "link",
+ label: "Day 0 Support: Claude Sonnet 4.6",
+ href: "/blog/claude_sonnet_4_6",
+ },
{
type: "link",
label: "Incident: Broken Model Cost Map",
diff --git a/docs/my-website/src/pages/troubleshoot.md b/docs/my-website/src/pages/troubleshoot.md
deleted file mode 100644
index 05dbf56caae..00000000000
--- a/docs/my-website/src/pages/troubleshoot.md
+++ /dev/null
@@ -1,11 +0,0 @@
-# Troubleshooting
-
-## Stable Version
-
-If you're running into problems with installation / Usage
-Use the stable version of litellm
-
-```
-pip install litellm==0.1.345
-```
-
diff --git a/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py b/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py
index bb25e4f0626..bf8bc46f723 100644
--- a/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py
+++ b/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py
@@ -4,7 +4,7 @@ Polls LiteLLM_ManagedObjectTable to check if the batch job is complete, and if t
from litellm._uuid import uuid
from datetime import datetime
-from typing import TYPE_CHECKING, Optional, cast
+from typing import TYPE_CHECKING, Optional
from litellm._logging import verbose_proxy_logger
@@ -35,14 +35,11 @@ class CheckBatchCost:
- if not, return False
- if so, return True
"""
- from litellm_enterprise.proxy.hooks.managed_files import (
- _PROXY_LiteLLMManagedFiles,
- )
-
from litellm.batches.batch_utils import (
_get_file_content_as_dictionary,
calculate_batch_cost_and_usage,
)
+ from litellm.files.main import afile_content
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
from litellm.proxy.openai_files_endpoints.common_utils import (
@@ -102,31 +99,41 @@ class CheckBatchCost:
continue
## RETRIEVE THE BATCH JOB OUTPUT FILE
- managed_files_obj = cast(
- Optional[_PROXY_LiteLLMManagedFiles],
- self.proxy_logging_obj.get_proxy_hook("managed_files"),
- )
if (
response.status == "completed"
and response.output_file_id is not None
- and managed_files_obj is not None
):
verbose_proxy_logger.info(
f"Batch ID: {batch_id} is complete, tracking cost and usage"
)
- # track cost
- model_file_id_mapping = {
- response.output_file_id: {model_id: response.output_file_id}
- }
- _file_content = await managed_files_obj.afile_content(
- file_id=response.output_file_id,
- litellm_parent_otel_span=None,
- llm_router=self.llm_router,
- model_file_id_mapping=model_file_id_mapping,
+
+ # This background job runs as default_user_id, so going through the HTTP endpoint
+ # would trigger check_managed_file_id_access and get 403. Instead, extract the raw
+ # provider file ID and call afile_content directly with deployment credentials.
+ raw_output_file_id = response.output_file_id
+ decoded = _is_base64_encoded_unified_file_id(raw_output_file_id)
+ if decoded:
+ try:
+ raw_output_file_id = decoded.split("llm_output_file_id,")[1].split(";")[0]
+ except (IndexError, AttributeError):
+ pass
+
+ credentials = self.llm_router.get_deployment_credentials_with_provider(model_id) or {}
+ _file_content = await afile_content(
+ file_id=raw_output_file_id,
+ **credentials,
)
+ # Access content - handle both direct attribute and method call
+ if hasattr(_file_content, 'content'):
+ content_bytes = _file_content.content
+ elif hasattr(_file_content, 'read'):
+ content_bytes = await _file_content.read()
+ else:
+ content_bytes = _file_content
+
file_content_as_dict = _get_file_content_as_dictionary(
- _file_content.content
+ content_bytes
)
deployment_info = self.llm_router.get_deployment(model_id=model_id)
@@ -143,11 +150,15 @@ class CheckBatchCost:
custom_llm_provider=custom_llm_provider,
)
+ # Pass deployment model_info so custom batch pricing
+ # (input_cost_per_token_batches etc.) is used for cost calc
+ deployment_model_info = deployment_info.model_info.model_dump() if deployment_info.model_info else {}
batch_cost, batch_usage, batch_models = (
await calculate_batch_cost_and_usage(
file_content_dictionary=file_content_as_dict,
custom_llm_provider=llm_provider, # type: ignore
model_name=model_name,
+ model_info=deployment_model_info,
)
)
logging_obj = LiteLLMLogging(
diff --git a/enterprise/litellm_enterprise/proxy/hooks/managed_files.py b/enterprise/litellm_enterprise/proxy/hooks/managed_files.py
index a41b3f3bf6f..b1cbeecd1ec 100644
--- a/enterprise/litellm_enterprise/proxy/hooks/managed_files.py
+++ b/enterprise/litellm_enterprise/proxy/hooks/managed_files.py
@@ -230,12 +230,14 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
if managed_file:
return managed_file.created_by == user_id
- return False
+ raise HTTPException(
+ status_code=404,
+ detail=f"File not found: {unified_file_id}",
+ )
async def can_user_call_unified_object_id(
self, unified_object_id: str, user_api_key_dict: UserAPIKeyAuth
) -> bool:
- ## check if the user has access to the unified object id
## check if the user has access to the unified object id
user_id = user_api_key_dict.user_id
managed_object = (
@@ -246,7 +248,10 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
if managed_object:
return managed_object.created_by == user_id
- return True # don't raise error if managed object is not found
+ raise HTTPException(
+ status_code=404,
+ detail=f"Object not found: {unified_object_id}",
+ )
async def list_user_batches(
self,
@@ -911,15 +916,24 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
)
setattr(response, file_attr, unified_file_id)
- # Fetch the actual file object from the provider
+ # Use llm_router credentials when available. Without credentials,
+ # Azure and other auth-required providers return 500/401.
file_object = None
try:
- # Use litellm to retrieve the file object from the provider
- from litellm import afile_retrieve
- file_object = await afile_retrieve(
- custom_llm_provider=model_name.split("/")[0] if model_name and "/" in model_name else "openai",
- file_id=original_file_id
- )
+ # Import module and use getattr for better testability with mocks
+ import litellm.proxy.proxy_server as proxy_server_module
+ _llm_router = getattr(proxy_server_module, 'llm_router', None)
+ if _llm_router is not None and model_id:
+ _creds = _llm_router.get_deployment_credentials_with_provider(model_id) or {}
+ file_object = await litellm.afile_retrieve(
+ file_id=original_file_id,
+ **_creds,
+ )
+ else:
+ file_object = await litellm.afile_retrieve(
+ custom_llm_provider=model_name.split("/")[0] if model_name and "/" in model_name else "openai",
+ file_id=original_file_id,
+ )
verbose_logger.debug(
f"Successfully retrieved file object for {file_attr}={original_file_id}"
)
@@ -1004,8 +1018,12 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
raise Exception(f"LiteLLM Managed File object with id={file_id} not found")
# Case 2: Managed file and the file object exists in the database
+ # The stored file_object has the raw provider ID. Replace with the unified ID
+ # so callers see a consistent ID (matching Case 3 which does response.id = file_id).
if stored_file_object and stored_file_object.file_object:
- return stored_file_object.file_object
+ # Use model_copy to ensure the ID update persists (Pydantic v2 compatibility)
+ response = stored_file_object.file_object.model_copy(update={"id": file_id})
+ return response
# Case 3: Managed file exists in the database but not the file object (for. e.g the batch task might not have run)
# So we fetch the file object from the provider. We deliberately do not store the result to avoid interfering with batch cost tracking code.
diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.4.40-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.4.40-py3-none-any.whl
new file mode 100644
index 00000000000..9f2ad8fd317
Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.4.40-py3-none-any.whl differ
diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.4.40.tar.gz b/litellm-proxy-extras/dist/litellm_proxy_extras-0.4.40.tar.gz
new file mode 100644
index 00000000000..fdab43c01a3
Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.4.40.tar.gz differ
diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260203120000_add_deprecated_verification_token_table/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260203120000_add_deprecated_verification_token_table/migration.sql
new file mode 100644
index 00000000000..51d88444191
--- /dev/null
+++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260203120000_add_deprecated_verification_token_table/migration.sql
@@ -0,0 +1,19 @@
+-- CreateTable
+CREATE TABLE "LiteLLM_DeprecatedVerificationToken" (
+ "id" TEXT NOT NULL,
+ "token" TEXT NOT NULL,
+ "active_token_id" TEXT NOT NULL,
+ "revoke_at" TIMESTAMP(3) NOT NULL,
+ "created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
+
+ CONSTRAINT "LiteLLM_DeprecatedVerificationToken_pkey" PRIMARY KEY ("id")
+);
+
+-- CreateIndex
+CREATE UNIQUE INDEX "LiteLLM_DeprecatedVerificationToken_token_key" ON "LiteLLM_DeprecatedVerificationToken"("token");
+
+-- CreateIndex
+CREATE INDEX "LiteLLM_DeprecatedVerificationToken_token_revoke_at_idx" ON "LiteLLM_DeprecatedVerificationToken"("token", "revoke_at");
+
+-- CreateIndex
+CREATE INDEX "LiteLLM_DeprecatedVerificationToken_revoke_at_idx" ON "LiteLLM_DeprecatedVerificationToken"("revoke_at");
diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260214124140_baseline_diff/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260214124140_baseline_diff/migration.sql
new file mode 100644
index 00000000000..2f725d83806
--- /dev/null
+++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260214124140_baseline_diff/migration.sql
@@ -0,0 +1,2 @@
+-- This is an empty migration.
+
diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
index af52d998949..8bd46672ae6 100644
--- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
+++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
@@ -326,6 +326,19 @@ model LiteLLM_VerificationToken {
@@index([budget_reset_at, expires])
}
+// Deprecated keys during grace period - allows old key to work until revoke_at
+model LiteLLM_DeprecatedVerificationToken {
+ id String @id @default(uuid())
+ token String // Hashed old key
+ active_token_id String // Current token hash in LiteLLM_VerificationToken
+ revoke_at DateTime // When the old key stops working
+ created_at DateTime @default(now()) @map("created_at")
+
+ @@unique([token])
+ @@index([token, revoke_at])
+ @@index([revoke_at])
+}
+
// Audit table for deleted keys - preserves spend and key information for historical tracking
model LiteLLM_DeletedVerificationToken {
id String @id @default(uuid())
diff --git a/litellm-proxy-extras/pyproject.toml b/litellm-proxy-extras/pyproject.toml
index 28786969747..7ef0409b6b8 100644
--- a/litellm-proxy-extras/pyproject.toml
+++ b/litellm-proxy-extras/pyproject.toml
@@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm-proxy-extras"
-version = "0.4.39"
+version = "0.4.40"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
authors = ["BerriAI"]
readme = "README.md"
@@ -22,7 +22,7 @@ requires = ["poetry-core"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
-version = "0.4.39"
+version = "0.4.40"
version_files = [
"pyproject.toml:version",
"../requirements.txt:litellm-proxy-extras==",
diff --git a/litellm/__init__.py b/litellm/__init__.py
index 4aaddc3da76..0f16fd5625c 100644
--- a/litellm/__init__.py
+++ b/litellm/__init__.py
@@ -1152,6 +1152,28 @@ from .skills.main import (
delete_skill,
adelete_skill,
)
+from .evals.main import (
+ create_eval,
+ acreate_eval,
+ list_evals,
+ alist_evals,
+ get_eval,
+ aget_eval,
+ delete_eval,
+ adelete_eval,
+ cancel_eval,
+ acancel_eval,
+ create_run,
+ acreate_run,
+ list_runs,
+ alist_runs,
+ get_run,
+ aget_run,
+ delete_run,
+ adelete_run,
+ cancel_run,
+ acancel_run,
+)
from .integrations import *
from .llms.custom_httpx.async_client_cleanup import close_litellm_async_clients
from .exceptions import (
@@ -1732,6 +1754,37 @@ def __getattr__(name: str) -> Any:
_globals["_service_logger"] = litellm._service_logger
return _globals["_service_logger"]
+ # Lazy load evals module functions
+ if name in ["acreate_eval", "alist_evals", "aget_eval", "aupdate_eval", "adelete_eval", "acancel_eval",
+ "create_eval", "list_evals", "get_eval", "update_eval", "delete_eval", "cancel_eval",
+ "acreate_run", "alist_runs", "aget_run", "acancel_run", "adelete_run",
+ "create_run", "list_runs", "get_run", "cancel_run", "delete_run"]:
+ from litellm.evals.main import (
+ acreate_eval,
+ alist_evals,
+ aget_eval,
+ aupdate_eval,
+ adelete_eval,
+ acancel_eval,
+ create_eval,
+ list_evals,
+ get_eval,
+ update_eval,
+ delete_eval,
+ cancel_eval,
+ acreate_run,
+ alist_runs,
+ aget_run,
+ acancel_run,
+ adelete_run,
+ create_run,
+ list_runs,
+ get_run,
+ cancel_run,
+ delete_run,
+ )
+ return locals()[name]
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
diff --git a/litellm/_service_logger.py b/litellm/_service_logger.py
index b67d0d86063..8f9a3c5083f 100644
--- a/litellm/_service_logger.py
+++ b/litellm/_service_logger.py
@@ -312,10 +312,12 @@ class ServiceLogging(CustomLogger):
_duration, type(_duration)
)
) # invalid _duration value
+ # Batch polling callbacks (check_batch_cost) don't include call_type in kwargs.
+ # Use .get() to avoid KeyError.
await self.async_service_success_hook(
service=ServiceTypes.LITELLM,
duration=_duration,
- call_type=kwargs["call_type"],
+ call_type=kwargs.get("call_type", "unknown")
)
except Exception as e:
raise e
diff --git a/litellm/batches/batch_utils.py b/litellm/batches/batch_utils.py
index 16a467e00cb..29bd99c2a60 100644
--- a/litellm/batches/batch_utils.py
+++ b/litellm/batches/batch_utils.py
@@ -8,7 +8,7 @@ import litellm
from litellm._logging import verbose_logger
from litellm._uuid import uuid
from litellm.types.llms.openai import Batch
-from litellm.types.utils import CallTypes, ModelResponse, Usage
+from litellm.types.utils import CallTypes, ModelInfo, ModelResponse, Usage
from litellm.utils import token_counter
@@ -16,14 +16,22 @@ async def calculate_batch_cost_and_usage(
file_content_dictionary: List[dict],
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "hosted_vllm", "anthropic"],
model_name: Optional[str] = None,
+ model_info: Optional[ModelInfo] = None,
) -> Tuple[float, Usage, List[str]]:
"""
- Calculate the cost and usage of a batch
+ Calculate the cost and usage of a batch.
+
+ Args:
+ model_info: Optional deployment-level model info with custom batch
+ pricing. Threaded through to batch_cost_calculator so that
+ deployment-specific pricing (e.g. input_cost_per_token_batches)
+ is used instead of the global cost map.
"""
batch_cost = _batch_cost_calculator(
custom_llm_provider=custom_llm_provider,
file_content_dictionary=file_content_dictionary,
model_name=model_name,
+ model_info=model_info,
)
batch_usage = _get_batch_job_total_usage_from_file_content(
file_content_dictionary=file_content_dictionary,
@@ -94,6 +102,7 @@ def _batch_cost_calculator(
file_content_dictionary: List[dict],
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "hosted_vllm", "anthropic"] = "openai",
model_name: Optional[str] = None,
+ model_info: Optional[ModelInfo] = None,
) -> float:
"""
Calculate the cost of a batch based on the output file id
@@ -108,6 +117,7 @@ def _batch_cost_calculator(
total_cost = _get_batch_job_cost_from_file_content(
file_content_dictionary=file_content_dictionary,
custom_llm_provider=custom_llm_provider,
+ model_info=model_info,
)
verbose_logger.debug("total_cost=%s", total_cost)
return total_cost
@@ -290,10 +300,13 @@ def _get_file_content_as_dictionary(file_content: bytes) -> List[dict]:
def _get_batch_job_cost_from_file_content(
file_content_dictionary: List[dict],
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "hosted_vllm", "anthropic"] = "openai",
+ model_info: Optional[ModelInfo] = None,
) -> float:
"""
Get the cost of a batch job from the file content
"""
+ from litellm.cost_calculator import batch_cost_calculator
+
try:
total_cost: float = 0.0
# parse the file content as json
@@ -303,11 +316,22 @@ def _get_batch_job_cost_from_file_content(
for _item in file_content_dictionary:
if _batch_response_was_successful(_item):
_response_body = _get_response_from_batch_job_output_file(_item)
- total_cost += litellm.completion_cost(
- completion_response=_response_body,
- custom_llm_provider=custom_llm_provider,
- call_type=CallTypes.aretrieve_batch.value,
- )
+ if model_info is not None:
+ usage = _get_batch_job_usage_from_response_body(_response_body)
+ model = _response_body.get("model", "")
+ prompt_cost, completion_cost = batch_cost_calculator(
+ usage=usage,
+ model=model,
+ custom_llm_provider=custom_llm_provider,
+ model_info=model_info,
+ )
+ total_cost += prompt_cost + completion_cost
+ else:
+ total_cost += litellm.completion_cost(
+ completion_response=_response_body,
+ custom_llm_provider=custom_llm_provider,
+ call_type=CallTypes.aretrieve_batch.value,
+ )
verbose_logger.debug("total_cost=%s", total_cost)
return total_cost
except Exception as e:
diff --git a/litellm/constants.py b/litellm/constants.py
index 03f80a8cb78..3c11eb701dc 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -319,6 +319,9 @@ NON_LLM_CONNECTION_TIMEOUT = int(
MAX_EXCEPTION_MESSAGE_LENGTH = int(os.getenv("MAX_EXCEPTION_MESSAGE_LENGTH", 2000))
MAX_STRING_LENGTH_PROMPT_IN_DB = int(os.getenv("MAX_STRING_LENGTH_PROMPT_IN_DB", 2048))
BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
+BEDROCK_MIN_THINKING_BUDGET_TOKENS = int(
+ os.getenv("BEDROCK_MIN_THINKING_BUDGET_TOKENS", 1024)
+)
REPLICATE_POLLING_DELAY_SECONDS = float(
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)
)
@@ -1036,6 +1039,7 @@ BEDROCK_CONVERSE_MODELS = [
"anthropic.claude-sonnet-4-5-20250929-v1:0",
"anthropic.claude-opus-4-6-v1:0",
"anthropic.claude-opus-4-6-v1",
+ "anthropic.claude-sonnet-4-6",
"anthropic.claude-opus-4-1-20250805-v1:0",
"anthropic.claude-opus-4-20250514-v1:0",
"anthropic.claude-sonnet-4-20250514-v1:0",
@@ -1258,6 +1262,9 @@ LITELLM_KEY_ROTATION_ENABLED = os.getenv("LITELLM_KEY_ROTATION_ENABLED", "false"
LITELLM_KEY_ROTATION_CHECK_INTERVAL_SECONDS = int(
os.getenv("LITELLM_KEY_ROTATION_CHECK_INTERVAL_SECONDS", 86400)
) # 24 hours default
+LITELLM_KEY_ROTATION_GRACE_PERIOD: str = os.getenv(
+ "LITELLM_KEY_ROTATION_GRACE_PERIOD", ""
+) # Duration to keep old key valid after rotation (e.g. "24h", "2d"); empty = immediate revoke (default)
UI_SESSION_TOKEN_TEAM_ID = "litellm-dashboard"
LITELLM_PROXY_ADMIN_NAME = "default_user_id"
diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py
index fe082843306..dae0bb1c2c0 100644
--- a/litellm/cost_calculator.py
+++ b/litellm/cost_calculator.py
@@ -1896,9 +1896,16 @@ def batch_cost_calculator(
usage: Usage,
model: str,
custom_llm_provider: Optional[str] = None,
+ model_info: Optional[ModelInfo] = None,
) -> Tuple[float, float]:
"""
- Calculate the cost of a batch job
+ Calculate the cost of a batch job.
+
+ Args:
+ model_info: Optional deployment-level model info containing custom
+ batch pricing (e.g. input_cost_per_token_batches). When provided,
+ skips the global litellm.get_model_info() lookup so that
+ deployment-specific pricing is used.
"""
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
@@ -1911,12 +1918,13 @@ def batch_cost_calculator(
custom_llm_provider,
)
- try:
- model_info: Optional[ModelInfo] = litellm.get_model_info(
- model=model, custom_llm_provider=custom_llm_provider
- )
- except Exception:
- model_info = None
+ if model_info is None:
+ try:
+ model_info = litellm.get_model_info(
+ model=model, custom_llm_provider=custom_llm_provider
+ )
+ except Exception:
+ model_info = None
if not model_info:
return 0.0, 0.0
diff --git a/litellm/evals/__init__.py b/litellm/evals/__init__.py
new file mode 100644
index 00000000000..89dfb62b2b7
--- /dev/null
+++ b/litellm/evals/__init__.py
@@ -0,0 +1,33 @@
+"""
+Evals API operations
+"""
+
+from .main import (
+ acancel_eval,
+ acreate_eval,
+ adelete_eval,
+ aget_eval,
+ alist_evals,
+ aupdate_eval,
+ cancel_eval,
+ create_eval,
+ delete_eval,
+ get_eval,
+ list_evals,
+ update_eval,
+)
+
+__all__ = [
+ "acreate_eval",
+ "alist_evals",
+ "aget_eval",
+ "aupdate_eval",
+ "adelete_eval",
+ "acancel_eval",
+ "create_eval",
+ "list_evals",
+ "get_eval",
+ "update_eval",
+ "delete_eval",
+ "cancel_eval",
+]
diff --git a/litellm/evals/main.py b/litellm/evals/main.py
new file mode 100644
index 00000000000..a39c2839150
--- /dev/null
+++ b/litellm/evals/main.py
@@ -0,0 +1,1944 @@
+"""
+Main entry point for Evals API operations
+Provides create, list, get, update, delete, and cancel operations for evals
+"""
+
+import asyncio
+import contextvars
+from functools import partial
+from typing import Any, Coroutine, Dict, List, Optional, Union
+
+import httpx
+
+import litellm
+from litellm.constants import request_timeout
+from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
+from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig
+from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
+from litellm.types.llms.openai_evals import (
+ CancelEvalResponse,
+ CancelRunResponse,
+ CreateEvalRequest,
+ CreateRunRequest,
+ DeleteEvalResponse,
+ Eval,
+ ListEvalsParams,
+ ListEvalsResponse,
+ ListRunsParams,
+ ListRunsResponse,
+ Run,
+ RunDeleteResponse,
+ UpdateEvalRequest,
+)
+from litellm.types.router import GenericLiteLLMParams
+from litellm.utils import ProviderConfigManager, client
+
+# Initialize HTTP handler
+base_llm_http_handler = BaseLLMHTTPHandler()
+DEFAULT_OPENAI_API_BASE = "https://api.openai.com"
+
+
+@client
+async def acreate_eval(
+ data_source_config: Dict[str, Any],
+ testing_criteria: List[Dict[str, Any]],
+ name: Optional[str] = None,
+ metadata: Optional[Dict[str, Any]] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ extra_body: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Eval:
+ """
+ Async: Create a new evaluation
+
+ Args:
+ data_source_config: Configuration for the data source
+ testing_criteria: List of graders for all eval runs
+ name: Optional name for the evaluation
+ metadata: Optional additional metadata (max 16 key-value pairs)
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ extra_body: Additional body parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Eval object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["acreate_eval"] = True
+
+ func = partial(
+ create_eval,
+ data_source_config=data_source_config,
+ testing_criteria=testing_criteria,
+ name=name,
+ metadata=metadata,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def create_eval(
+ data_source_config: Dict[str, Any],
+ testing_criteria: List[Dict[str, Any]],
+ name: Optional[str] = None,
+ metadata: Optional[Dict[str, Any]] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ extra_body: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[Eval, Coroutine[Any, Any, Eval]]:
+ """
+ Create a new evaluation
+
+ Args:
+ data_source_config: Configuration for the data source
+ testing_criteria: List of graders for all eval runs
+ name: Optional name for the evaluation
+ metadata: Optional additional metadata (max 16 key-value pairs)
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ extra_body: Additional body parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Eval object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("acreate_eval", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(
+ f"CREATE eval is not supported for {custom_llm_provider}"
+ )
+
+ # Build create request
+ create_request: CreateEvalRequest = {
+ "data_source_config": data_source_config, # type: ignore
+ "testing_criteria": testing_criteria, # type: ignore
+ }
+ if name is not None:
+ create_request["name"] = name
+
+ # Merge extra_body if provided
+ if extra_body:
+ create_request.update(extra_body) # type: ignore
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ request_body = evals_api_provider_config.transform_create_eval_request(
+ create_request=create_request,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Get API base and URL
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url = evals_api_provider_config.get_complete_url(
+ api_base=api_base, endpoint="evals"
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params=request_body,
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.create_eval_handler( # type: ignore
+ url=url,
+ request_body=request_body,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def alist_evals(
+ limit: Optional[int] = None,
+ after: Optional[str] = None,
+ before: Optional[str] = None,
+ order: Optional[str] = None,
+ order_by: Optional[str] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> ListEvalsResponse:
+ """
+ Async: List all evaluations
+
+ Args:
+ limit: Number of results to return per page (max 100, default 20)
+ after: Cursor for pagination - returns evals after this ID
+ before: Cursor for pagination - returns evals before this ID
+ order: Sort order ('asc' or 'desc', default 'desc')
+ order_by: Field to sort by ('created_at' or 'updated_at', default 'created_at')
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ ListEvalsResponse object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["alist_evals"] = True
+
+ func = partial(
+ list_evals,
+ limit=limit,
+ after=after,
+ before=before,
+ order=order,
+ order_by=order_by,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def list_evals(
+ limit: Optional[int] = None,
+ after: Optional[str] = None,
+ before: Optional[str] = None,
+ order: Optional[str] = None,
+ order_by: Optional[str] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[ListEvalsResponse, Coroutine[Any, Any, ListEvalsResponse]]:
+ """
+ List all evaluations
+
+ Args:
+ limit: Number of results to return per page (max 100, default 20)
+ after: Cursor for pagination - returns evals after this ID
+ before: Cursor for pagination - returns evals before this ID
+ order: Sort order ('asc' or 'desc', default 'desc')
+ order_by: Field to sort by ('created_at' or 'updated_at', default 'created_at')
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ ListEvalsResponse object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("alist_evals", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"LIST evals is not supported for {custom_llm_provider}")
+
+ # Build list parameters
+ list_params: ListEvalsParams = {}
+ if limit is not None:
+ list_params["limit"] = limit
+ if after is not None:
+ list_params["after"] = after
+ if before is not None:
+ list_params["before"] = before
+ if order is not None:
+ list_params["order"] = order # type: ignore
+ if order_by is not None:
+ list_params["order_by"] = order_by # type: ignore
+
+ # Merge extra_query if provided
+ if extra_query:
+ list_params.update(extra_query) # type: ignore
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ url, query_params = evals_api_provider_config.transform_list_evals_request(
+ list_params=list_params,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params=query_params,
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.list_evals_handler( # type: ignore
+ url=url,
+ query_params=query_params,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def aget_eval(
+ eval_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Eval:
+ """
+ Async: Get an evaluation by ID
+
+ Args:
+ eval_id: The ID of the evaluation to fetch
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Eval object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["aget_eval"] = True
+
+ func = partial(
+ get_eval,
+ eval_id=eval_id,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def get_eval(
+ eval_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[Eval, Coroutine[Any, Any, Eval]]:
+ """
+ Get an evaluation by ID
+
+ Args:
+ eval_id: The ID of the evaluation to fetch
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Eval object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("aget_eval", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"GET eval is not supported for {custom_llm_provider}")
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers = evals_api_provider_config.transform_get_eval_request(
+ eval_id=eval_id,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.get_eval_handler( # type: ignore
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def aupdate_eval(
+ eval_id: str,
+ name: Optional[str] = None,
+ metadata: Optional[Dict[str, Any]] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ extra_body: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Eval:
+ """
+ Async: Update an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to update
+ name: Updated name
+ metadata: Updated metadata
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ extra_body: Additional body parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Eval object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["aupdate_eval"] = True
+
+ func = partial(
+ update_eval,
+ eval_id=eval_id,
+ name=name,
+ metadata=metadata,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def update_eval(
+ eval_id: str,
+ name: Optional[str] = None,
+ metadata: Optional[Dict[str, Any]] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ extra_body: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[Eval, Coroutine[Any, Any, Eval]]:
+ """
+ Update an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to update
+ name: Updated name
+ metadata: Updated metadata
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ extra_body: Additional body parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Eval object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("aupdate_eval", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(
+ f"UPDATE eval is not supported for {custom_llm_provider}"
+ )
+
+ # Build update request
+ update_request: UpdateEvalRequest = {}
+ if name is not None:
+ update_request["name"] = name
+
+ # Filter metadata to exclude internal LiteLLM fields
+ if metadata is not None:
+ # List of internal LiteLLM metadata keys that should NOT be sent to OpenAI
+ internal_keys = {
+ "headers", "requester_metadata", "user_api_key_hash", "user_api_key_alias",
+ "user_api_key_spend", "user_api_key_max_budget", "user_api_key_team_id",
+ "user_api_key_user_id", "user_api_key_org_id", "user_api_key_team_alias",
+ "user_api_key_end_user_id", "user_api_key_user_email", "user_api_key_request_route",
+ "user_api_key_budget_reset_at", "user_api_key_auth_metadata", "user_api_key",
+ "user_api_end_user_max_budget", "user_api_key_auth", "litellm_api_version",
+ "global_max_parallel_requests", "user_api_key_team_max_budget",
+ "user_api_key_team_spend", "user_api_key_model_max_budget",
+ "user_api_key_user_spend", "user_api_key_user_max_budget",
+ "user_api_key_metadata", "endpoint", "litellm_parent_otel_span",
+ "requester_ip_address", "user_agent",
+ }
+ # Only include user-provided metadata keys
+ filtered_metadata = {k: v for k, v in metadata.items() if k not in internal_keys}
+ if filtered_metadata: # Only add if there's user metadata
+ update_request["metadata"] = filtered_metadata
+
+ # Merge extra_body if provided
+ if extra_body:
+ update_request.update(extra_body) # type: ignore
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers, request_body = evals_api_provider_config.transform_update_eval_request(
+ eval_id=eval_id,
+ update_request=update_request,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params=request_body,
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.update_eval_handler( # type: ignore
+ url=url,
+ request_body=request_body,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def adelete_eval(
+ eval_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> DeleteEvalResponse:
+ """
+ Async: Delete an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to delete
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ DeleteEvalResponse object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["adelete_eval"] = True
+
+ func = partial(
+ delete_eval,
+ eval_id=eval_id,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def delete_eval(
+ eval_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[DeleteEvalResponse, Coroutine[Any, Any, DeleteEvalResponse]]:
+ """
+ Delete an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to delete
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ DeleteEvalResponse object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("adelete_eval", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"DELETE eval is not supported for {custom_llm_provider}")
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers = evals_api_provider_config.transform_delete_eval_request(
+ eval_id=eval_id,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.delete_eval_handler( # type: ignore
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def acancel_eval(
+ eval_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> CancelEvalResponse:
+ """
+ Async: Cancel a running evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to cancel
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ CancelEvalResponse object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["acancel_eval"] = True
+
+ func = partial(
+ cancel_eval,
+ eval_id=eval_id,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def cancel_eval(
+ eval_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[CancelEvalResponse, Coroutine[Any, Any, CancelEvalResponse]]:
+ """
+ Cancel a running evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to cancel
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ CancelEvalResponse object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("acancel_eval", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"CANCEL eval is not supported for {custom_llm_provider}")
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers, request_body = evals_api_provider_config.transform_cancel_eval_request(
+ eval_id=eval_id,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.cancel_eval_handler( # type: ignore
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+# ===================================
+# Run API Functions
+# ===================================
+
+
+@client
+async def acreate_run(
+ eval_id: str,
+ data_source: Dict[str, Any],
+ name: Optional[str] = None,
+ metadata: Optional[Dict[str, Any]] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ extra_body: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Run:
+ """
+ Async: Create a new run for an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to run
+ data_source: Data source configuration for the run (can be jsonl, completions, or responses type)
+ name: Optional name for the run
+ metadata: Optional additional metadata
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ extra_body: Additional body parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Run object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["acreate_run"] = True
+
+ func = partial(
+ create_run,
+ eval_id=eval_id,
+ data_source=data_source,
+ name=name,
+ metadata=metadata,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def create_run(
+ eval_id: str,
+ data_source: Dict[str, Any],
+ name: Optional[str] = None,
+ metadata: Optional[Dict[str, Any]] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ extra_body: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[Run, Coroutine[Any, Any, Run]]:
+ """
+ Create a new run for an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation to run
+ data_source: Data source configuration for the run (can be jsonl, completions, or responses type)
+ name: Optional name for the run
+ metadata: Optional additional metadata
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ extra_body: Additional body parameters
+ timeout: Request timeout (default 600s for long-running operations)
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Run object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("acreate_run", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(
+ f"CREATE run is not supported for {custom_llm_provider}"
+ )
+
+ # Build create request
+ create_request: CreateRunRequest = {
+ "data_source": data_source, # type: ignore
+ }
+ if name is not None:
+ create_request["name"] = name
+ # if metadata is not None:
+ # create_request["metadata"] = metadata
+
+ # Merge extra_body if provided
+ if extra_body:
+ create_request.update(extra_body) # type: ignore
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, request_body = evals_api_provider_config.transform_create_run_request(
+ eval_id=eval_id,
+ create_request=create_request,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params=request_body,
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request (default 600s timeout for long-running operations)
+ response = base_llm_http_handler.create_run_handler( # type: ignore
+ url=url,
+ request_body=request_body,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or httpx.Timeout(timeout=600.0, connect=5.0),
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def alist_runs(
+ eval_id: str,
+ limit: Optional[int] = None,
+ after: Optional[str] = None,
+ before: Optional[str] = None,
+ order: Optional[str] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> ListRunsResponse:
+ """
+ Async: List all runs for an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation
+ limit: Number of results to return per page (max 100, default 20)
+ after: Cursor for pagination - returns runs after this ID
+ before: Cursor for pagination - returns runs before this ID
+ order: Sort order ('asc' or 'desc', default 'desc')
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ ListRunsResponse object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["alist_runs"] = True
+
+ func = partial(
+ list_runs,
+ eval_id=eval_id,
+ limit=limit,
+ after=after,
+ before=before,
+ order=order,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def list_runs(
+ eval_id: str,
+ limit: Optional[int] = None,
+ after: Optional[str] = None,
+ before: Optional[str] = None,
+ order: Optional[str] = None,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[ListRunsResponse, Coroutine[Any, Any, ListRunsResponse]]:
+ """
+ List all runs for an evaluation
+
+ Args:
+ eval_id: The ID of the evaluation
+ limit: Number of results to return per page (max 100, default 20)
+ after: Cursor for pagination - returns runs after this ID
+ before: Cursor for pagination - returns runs before this ID
+ order: Sort order ('asc' or 'desc', default 'desc')
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ ListRunsResponse object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("alist_runs", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"LIST runs is not supported for {custom_llm_provider}")
+
+ # Build list parameters
+ list_params: ListRunsParams = {}
+ if limit is not None:
+ list_params["limit"] = limit
+ if after is not None:
+ list_params["after"] = after
+ if before is not None:
+ list_params["before"] = before
+ if order is not None:
+ list_params["order"] = order # type: ignore
+
+ # Merge extra_query if provided
+ if extra_query:
+ list_params.update(extra_query) # type: ignore
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ url, query_params = evals_api_provider_config.transform_list_runs_request(
+ eval_id=eval_id,
+ list_params=list_params,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id, **query_params},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.list_runs_handler( # type: ignore
+ url=url,
+ query_params=query_params,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def aget_run(
+ eval_id: str,
+ run_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Run:
+ """
+ Async: Get a specific run
+
+ Args:
+ eval_id: The ID of the evaluation
+ run_id: The ID of the run to retrieve
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Run object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["aget_run"] = True
+
+ func = partial(
+ get_run,
+ eval_id=eval_id,
+ run_id=run_id,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def get_run(
+ eval_id: str,
+ run_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[Run, Coroutine[Any, Any, Run]]:
+ """
+ Get a specific run
+
+ Args:
+ eval_id: The ID of the evaluation
+ run_id: The ID of the run to retrieve
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ Run object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("aget_run", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"GET run is not supported for {custom_llm_provider}")
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers = evals_api_provider_config.transform_get_run_request(
+ eval_id=eval_id,
+ run_id=run_id,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id, "run_id": run_id},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.get_run_handler( # type: ignore
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+async def acancel_run(
+ eval_id: str,
+ run_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> CancelRunResponse:
+ """
+ Async: Cancel a running run
+
+ Args:
+ eval_id: The ID of the evaluation
+ run_id: The ID of the run to cancel
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ CancelRunResponse object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["acancel_run"] = True
+
+ func = partial(
+ cancel_run,
+ eval_id=eval_id,
+ run_id=run_id,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def cancel_run(
+ eval_id: str,
+ run_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[CancelRunResponse, Coroutine[Any, Any, CancelRunResponse]]:
+ """
+ Cancel a running run
+
+ Args:
+ eval_id: The ID of the evaluation
+ run_id: The ID of the run to cancel
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ CancelRunResponse object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("acancel_run", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"CANCEL run is not supported for {custom_llm_provider}")
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers, request_body = evals_api_provider_config.transform_cancel_run_request(
+ eval_id=eval_id,
+ run_id=run_id,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id, "run_id": run_id},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.cancel_run_handler( # type: ignore
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+# ===================================
+# Delete Run API Functions
+# ===================================
+
+
+@client
+async def adelete_run(
+ eval_id: str,
+ run_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> RunDeleteResponse:
+ """
+ Async: Delete a run
+
+ Args:
+ eval_id: The ID of the evaluation
+ run_id: The ID of the run to delete
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ RunDeleteResponse object
+ """
+ local_vars = locals()
+ try:
+ loop = asyncio.get_event_loop()
+ kwargs["adelete_run"] = True
+
+ func = partial(
+ delete_run,
+ eval_id=eval_id,
+ run_id=run_id,
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ timeout=timeout,
+ custom_llm_provider=custom_llm_provider,
+ **kwargs,
+ )
+
+ ctx = contextvars.copy_context()
+ func_with_context = partial(ctx.run, func)
+ init_response = await loop.run_in_executor(None, func_with_context)
+
+ if asyncio.iscoroutine(init_response):
+ response = await init_response
+ else:
+ response = init_response
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
+
+
+@client
+def delete_run(
+ eval_id: str,
+ run_id: str,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ extra_query: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ custom_llm_provider: Optional[str] = None,
+ **kwargs,
+) -> Union[RunDeleteResponse, Coroutine[Any, Any, RunDeleteResponse]]:
+ """
+ Delete a run
+
+ Args:
+ eval_id: The ID of the evaluation
+ run_id: The ID of the run to delete
+ extra_headers: Additional headers for the request
+ extra_query: Additional query parameters
+ timeout: Request timeout
+ custom_llm_provider: Provider name (e.g., 'openai')
+ **kwargs: Additional parameters
+
+ Returns:
+ RunDeleteResponse object
+ """
+ local_vars = locals()
+ try:
+ litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
+ litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
+ _is_async = kwargs.pop("adelete_run", False) is True
+
+ # Get LiteLLM parameters
+ litellm_params = GenericLiteLLMParams(**kwargs)
+
+ # Determine provider
+ if custom_llm_provider is None:
+ custom_llm_provider = "openai"
+
+ # Get provider config
+ evals_api_provider_config: Optional[BaseEvalsAPIConfig] = (
+ ProviderConfigManager.get_provider_evals_api_config( # type: ignore
+ provider=litellm.LlmProviders(custom_llm_provider),
+ )
+ )
+
+ if evals_api_provider_config is None:
+ raise ValueError(f"DELETE run is not supported for {custom_llm_provider}")
+
+ # Validate environment and get headers
+ headers = extra_headers or {}
+ headers = evals_api_provider_config.validate_environment(
+ headers=headers, litellm_params=litellm_params
+ )
+
+ # Transform request
+ api_base = litellm_params.api_base or DEFAULT_OPENAI_API_BASE
+ url, headers, request_body = evals_api_provider_config.transform_delete_run_request(
+ eval_id=eval_id,
+ run_id=run_id,
+ api_base=api_base,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ # Pre-call logging
+ litellm_logging_obj.update_environment_variables(
+ model=None,
+ optional_params={"eval_id": eval_id, "run_id": run_id},
+ litellm_params={
+ "litellm_call_id": litellm_call_id,
+ },
+ custom_llm_provider=custom_llm_provider,
+ )
+
+ # Make HTTP request
+ response = base_llm_http_handler.delete_run_handler( # type: ignore
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=litellm_logging_obj,
+ extra_headers=headers,
+ timeout=timeout or request_timeout,
+ _is_async=_is_async,
+ client=kwargs.get("client"),
+ shared_session=kwargs.get("shared_session"),
+ )
+
+ return response
+ except Exception as e:
+ raise litellm.exception_type(
+ model=None,
+ custom_llm_provider=custom_llm_provider,
+ original_exception=e,
+ completion_kwargs=local_vars,
+ extra_kwargs=kwargs,
+ )
diff --git a/litellm/integrations/custom_guardrail.py b/litellm/integrations/custom_guardrail.py
index 0b7b95e7bbc..4a1e3e41e96 100644
--- a/litellm/integrations/custom_guardrail.py
+++ b/litellm/integrations/custom_guardrail.py
@@ -26,6 +26,7 @@ from litellm.types.utils import (
CallTypes,
GenericGuardrailAPIInputs,
GuardrailStatus,
+ GuardrailTracingDetail,
LLMResponseTypes,
StandardLoggingGuardrailInformation,
)
@@ -520,9 +521,15 @@ class CustomGuardrail(CustomLogger):
masked_entity_count: Optional[Dict[str, int]] = None,
guardrail_provider: Optional[str] = None,
event_type: Optional[GuardrailEventHooks] = None,
+ tracing_detail: Optional[GuardrailTracingDetail] = None,
) -> None:
"""
Builds `StandardLoggingGuardrailInformation` and adds it to the request metadata so it can be used for logging to DataDog, Langfuse, etc.
+
+ Args:
+ tracing_detail: Optional typed dict with provider-specific tracing fields
+ (guardrail_id, policy_template, detection_method, confidence_score,
+ classification, match_details, patterns_checked, alert_recipients).
"""
if isinstance(guardrail_json_response, Exception):
guardrail_json_response = str(guardrail_json_response)
@@ -559,6 +566,7 @@ class CustomGuardrail(CustomLogger):
end_time=end_time,
duration=duration,
masked_entity_count=masked_entity_count,
+ **(tracing_detail or {}),
)
def _append_guardrail_info(container: dict) -> None:
@@ -814,8 +822,8 @@ def log_guardrail_information(func):
- during_call
- post_call
"""
- import asyncio
import functools
+ import inspect
def _infer_event_type_from_function_name(
func_name: str,
@@ -896,7 +904,7 @@ def log_guardrail_information(func):
@functools.wraps(func)
def wrapper(*args, **kwargs):
- if asyncio.iscoroutinefunction(func):
+ if inspect.iscoroutinefunction(func):
return async_wrapper(*args, **kwargs)
return sync_wrapper(*args, **kwargs)
diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py
index b847180174a..35362a71ccd 100644
--- a/litellm/integrations/opentelemetry.py
+++ b/litellm/integrations/opentelemetry.py
@@ -1051,23 +1051,15 @@ class OpenTelemetry(CustomLogger):
# See: https://github.com/open-telemetry/opentelemetry-python/pull/4676
# TODO: Refactor to use the proper OTEL Logs API instead of directly creating SDK LogRecords
- from opentelemetry._logs import (
- SeverityNumber,
- get_logger,
- )
-
- # MyPy evaluates both branches of try/except imports and can fail when
- # newer OTEL stubs remove/relocate symbols. Gate the typing import so
- # only the canonical location is type-checked.
- if TYPE_CHECKING:
- from opentelemetry.sdk._logs._internal import LogRecord as SdkLogRecord
- else:
- try:
- from opentelemetry.sdk._logs import (
- LogRecord as SdkLogRecord, # type: ignore[attr-defined]
- )
- except ImportError:
- from opentelemetry.sdk._logs._internal import LogRecord as SdkLogRecord
+ from opentelemetry._logs import SeverityNumber, get_logger
+ try:
+ from opentelemetry.sdk._logs import ( # type: ignore[attr-defined] # OTEL < 1.39.0
+ LogRecord as SdkLogRecord,
+ )
+ except ImportError:
+ from opentelemetry.sdk._logs._internal import (
+ LogRecord as SdkLogRecord, # type: ignore[attr-defined] # OTEL >= 1.39.0
+ )
otel_logger = get_logger(LITELLM_LOGGER_NAME)
diff --git a/litellm/integrations/s3_v2.py b/litellm/integrations/s3_v2.py
index 534b85e4752..eddc80dbc1f 100644
--- a/litellm/integrations/s3_v2.py
+++ b/litellm/integrations/s3_v2.py
@@ -51,6 +51,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
s3_use_team_prefix: bool = False,
s3_strip_base64_files: bool = False,
s3_use_key_prefix: bool = False,
+ s3_use_virtual_hosted_style: bool = False,
**kwargs,
):
try:
@@ -78,7 +79,8 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
s3_path=s3_path,
s3_use_team_prefix=s3_use_team_prefix,
s3_strip_base64_files=s3_strip_base64_files,
- s3_use_key_prefix=s3_use_key_prefix
+ s3_use_key_prefix=s3_use_key_prefix,
+ s3_use_virtual_hosted_style=s3_use_virtual_hosted_style
)
verbose_logger.debug(f"s3 logger using endpoint url {s3_endpoint_url}")
@@ -135,6 +137,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
s3_use_team_prefix: bool = False,
s3_strip_base64_files: bool = False,
s3_use_key_prefix: bool = False,
+ s3_use_virtual_hosted_style: bool = False,
):
"""
Initialize the s3 params for this logging callback
@@ -217,6 +220,11 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
or s3_strip_base64_files
)
+ self.s3_use_virtual_hosted_style = (
+ bool(litellm.s3_callback_params.get("s3_use_virtual_hosted_style", False))
+ or s3_use_virtual_hosted_style
+ )
+
return
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
@@ -247,8 +255,14 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
standard_logging_payload=kwargs.get("standard_logging_object", None),
)
+ # afile_delete and other non-model call types never produce a standard_logging_object,
+ # so s3_batch_logging_element is None. Skip gracefully instead of raising ValueError.
if s3_batch_logging_element is None:
- raise ValueError("s3_batch_logging_element is None")
+ verbose_logger.debug(
+ "s3 Logging - skipping event, no standard_logging_object for call_type=%s",
+ kwargs.get("call_type", "unknown"),
+ )
+ return
verbose_logger.debug(
"\ns3 Logger - Logging payload = %s", s3_batch_logging_element
@@ -302,13 +316,20 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{batch_logging_element.s3_object_key}"
if self.s3_endpoint_url and self.s3_bucket_name:
- url = (
- self.s3_endpoint_url
- + "/"
- + self.s3_bucket_name
- + "/"
- + batch_logging_element.s3_object_key
- )
+ if self.s3_use_virtual_hosted_style:
+ # Virtual-hosted-style: bucket.endpoint/key
+ endpoint_host = self.s3_endpoint_url.replace("https://", "").replace("http://", "")
+ protocol = "https://" if self.s3_endpoint_url.startswith("https://") else "http://"
+ url = f"{protocol}{self.s3_bucket_name}.{endpoint_host}/{batch_logging_element.s3_object_key}"
+ else:
+ # Path-style: endpoint/bucket/key
+ url = (
+ self.s3_endpoint_url
+ + "/"
+ + self.s3_bucket_name
+ + "/"
+ + batch_logging_element.s3_object_key
+ )
# Convert JSON to string
json_string = safe_dumps(batch_logging_element.payload)
@@ -456,13 +477,20 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{batch_logging_element.s3_object_key}"
if self.s3_endpoint_url and self.s3_bucket_name:
- url = (
- self.s3_endpoint_url
- + "/"
- + self.s3_bucket_name
- + "/"
- + batch_logging_element.s3_object_key
- )
+ if self.s3_use_virtual_hosted_style:
+ # Virtual-hosted-style: bucket.endpoint/key
+ endpoint_host = self.s3_endpoint_url.replace("https://", "").replace("http://", "")
+ protocol = "https://" if self.s3_endpoint_url.startswith("https://") else "http://"
+ url = f"{protocol}{self.s3_bucket_name}.{endpoint_host}/{batch_logging_element.s3_object_key}"
+ else:
+ # Path-style: endpoint/bucket/key
+ url = (
+ self.s3_endpoint_url
+ + "/"
+ + self.s3_bucket_name
+ + "/"
+ + batch_logging_element.s3_object_key
+ )
# Convert JSON to string
json_string = safe_dumps(batch_logging_element.payload)
@@ -550,13 +578,20 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{s3_object_key}"
if self.s3_endpoint_url and self.s3_bucket_name:
- url = (
- self.s3_endpoint_url
- + "/"
- + self.s3_bucket_name
- + "/"
- + s3_object_key
- )
+ if self.s3_use_virtual_hosted_style:
+ # Virtual-hosted-style: bucket.endpoint/key
+ endpoint_host = self.s3_endpoint_url.replace("https://", "").replace("http://", "")
+ protocol = "https://" if self.s3_endpoint_url.startswith("https://") else "http://"
+ url = f"{protocol}{self.s3_bucket_name}.{endpoint_host}/{s3_object_key}"
+ else:
+ # Path-style: endpoint/bucket/key
+ url = (
+ self.s3_endpoint_url
+ + "/"
+ + self.s3_bucket_name
+ + "/"
+ + s3_object_key
+ )
# Prepare the request for GET operation
# For GET requests, we need x-amz-content-sha256 with hash of empty string
@@ -618,4 +653,4 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
verbose_logger.exception(
f"Error retrieving object {object_key} from cold storage: {str(e)}"
)
- return None
+ return None
\ No newline at end of file
diff --git a/litellm/litellm_core_utils/logging_utils.py b/litellm/litellm_core_utils/logging_utils.py
index bf43519afc6..8cde8ccef1c 100644
--- a/litellm/litellm_core_utils/logging_utils.py
+++ b/litellm/litellm_core_utils/logging_utils.py
@@ -1,5 +1,6 @@
import asyncio
import functools
+import inspect
import time
from datetime import datetime
from typing import TYPE_CHECKING, Any, List, Optional, Union
@@ -270,7 +271,7 @@ def track_llm_api_timing():
verbose_logger.debug(f"Error in service logging: {str(e)}")
# Check if the function is async or sync
- if asyncio.iscoroutinefunction(func):
+ if inspect.iscoroutinefunction(func):
return async_wrapper
return sync_wrapper
diff --git a/litellm/litellm_core_utils/redact_messages.py b/litellm/litellm_core_utils/redact_messages.py
index 5d6d1fbc1c5..ad68f3851a8 100644
--- a/litellm/litellm_core_utils/redact_messages.py
+++ b/litellm/litellm_core_utils/redact_messages.py
@@ -9,6 +9,7 @@
import asyncio
import copy
+import inspect
from typing import TYPE_CHECKING, Any, Optional
import litellm
@@ -101,8 +102,8 @@ def perform_redaction(model_call_details: dict, result):
# Redact result
if result is not None:
# Check if result is a coroutine, async generator, or other async object - these cannot be deepcopied
- if (asyncio.iscoroutine(result) or
- asyncio.iscoroutinefunction(result) or
+ if (asyncio.iscoroutine(result) or
+ inspect.iscoroutinefunction(result) or
hasattr(result, '__aiter__') or # async generator
hasattr(result, '__anext__')): # async iterator
# For async objects, return a simple redacted response without deepcopy
diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py
index c2cfff80685..85a4790a9b9 100644
--- a/litellm/llms/anthropic/chat/transformation.py
+++ b/litellm/llms/anthropic/chat/transformation.py
@@ -1282,9 +1282,13 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
output_config = optional_params.get("output_config")
if output_config and isinstance(output_config, dict):
effort = output_config.get("effort")
- if effort and effort not in ["high", "medium", "low"]:
+ if effort and effort not in ["high", "medium", "low", "max"]:
raise ValueError(
- f"Invalid effort value: {effort}. Must be one of: 'high', 'medium', 'low'"
+ f"Invalid effort value: {effort}. Must be one of: 'high', 'medium', 'low', 'max'"
+ )
+ if effort == "max" and not self._is_claude_opus_4_6(model):
+ raise ValueError(
+ f"effort='max' is only supported by Claude Opus 4.6. Got model: {model}"
)
data["output_config"] = output_config
diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py
index c6caaddf98b..73e74c228ba 100644
--- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py
+++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py
@@ -19,6 +19,7 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
from litellm.types.utils import ModelResponse
+from litellm.utils import get_model_info
if TYPE_CHECKING:
pass
@@ -63,6 +64,14 @@ class LiteLLMMessagesToCompletionTransformationHandler:
return
model = completion_kwargs.get("model")
+ try:
+ model_info = get_model_info(model=cast(str, model), custom_llm_provider=custom_llm_provider)
+ if model_info and model_info.get("supports_reasoning") is False:
+ # Model doesn't support reasoning/responses API, don't route
+ return
+ except Exception:
+ pass
+
if isinstance(model, str) and model and not model.startswith("responses/"):
# Prefix model with "responses/" to route to OpenAI Responses API
completion_kwargs["model"] = f"responses/{model}"
diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py
index a86820f82e8..de634ff9ecf 100644
--- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py
+++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py
@@ -239,8 +239,13 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
merged_chunk["delta"] = {}
# Add usage to the held chunk
+ uncached_input_tokens = chunk.usage.prompt_tokens or 0
+ if hasattr(chunk.usage, "prompt_tokens_details") and chunk.usage.prompt_tokens_details:
+ cached_tokens = getattr(chunk.usage.prompt_tokens_details, "cached_tokens", 0) or 0
+ uncached_input_tokens -= cached_tokens
+
usage_dict: UsageDelta = {
- "input_tokens": chunk.usage.prompt_tokens or 0,
+ "input_tokens": uncached_input_tokens,
"output_tokens": chunk.usage.completion_tokens or 0,
}
# Add cache tokens if available (for prompt caching support)
@@ -412,6 +417,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
if block_type == "tool_use":
# Type narrowing: content_block_start is ToolUseBlock when block_type is "tool_use"
from typing import cast
+
from litellm.types.llms.anthropic import ToolUseBlock
tool_block = cast(ToolUseBlock, content_block_start)
@@ -430,6 +436,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
# if we get a function name since it signals a new tool call
if block_type == "tool_use":
from typing import cast
+
from litellm.types.llms.anthropic import ToolUseBlock
tool_block = cast(ToolUseBlock, content_block_start)
diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
index 169b138a5f7..efbac13735c 100644
--- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
+++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
@@ -1070,8 +1070,13 @@ class LiteLLMAnthropicMessagesAdapter:
)
# extract usage
usage: Usage = getattr(response, "usage")
+ uncached_input_tokens = usage.prompt_tokens or 0
+ if hasattr(usage, "prompt_tokens_details") and usage.prompt_tokens_details:
+ cached_tokens = getattr(usage.prompt_tokens_details, "cached_tokens", 0) or 0
+ uncached_input_tokens -= cached_tokens
+
anthropic_usage = AnthropicUsage(
- input_tokens=usage.prompt_tokens or 0,
+ input_tokens=uncached_input_tokens,
output_tokens=usage.completion_tokens or 0,
)
# Add cache tokens if available (for prompt caching support)
@@ -1230,8 +1235,13 @@ class LiteLLMAnthropicMessagesAdapter:
else:
litellm_usage_chunk = None
if litellm_usage_chunk is not None:
+ uncached_input_tokens = litellm_usage_chunk.prompt_tokens or 0
+ if hasattr(litellm_usage_chunk, "prompt_tokens_details") and litellm_usage_chunk.prompt_tokens_details:
+ cached_tokens = getattr(litellm_usage_chunk.prompt_tokens_details, "cached_tokens", 0) or 0
+ uncached_input_tokens -= cached_tokens
+
usage_delta = UsageDelta(
- input_tokens=litellm_usage_chunk.prompt_tokens or 0,
+ input_tokens=uncached_input_tokens,
output_tokens=litellm_usage_chunk.completion_tokens or 0,
)
# Add cache tokens if available (for prompt caching support)
diff --git a/litellm/llms/base_llm/evals/__init__.py b/litellm/llms/base_llm/evals/__init__.py
new file mode 100644
index 00000000000..948ed5364ea
--- /dev/null
+++ b/litellm/llms/base_llm/evals/__init__.py
@@ -0,0 +1,7 @@
+"""
+Base configuration for Evals API
+"""
+
+from .transformation import BaseEvalsAPIConfig
+
+__all__ = ["BaseEvalsAPIConfig"]
diff --git a/litellm/llms/base_llm/evals/transformation.py b/litellm/llms/base_llm/evals/transformation.py
new file mode 100644
index 00000000000..54dc2f7aae9
--- /dev/null
+++ b/litellm/llms/base_llm/evals/transformation.py
@@ -0,0 +1,542 @@
+"""
+Base configuration class for Evals API
+"""
+
+from abc import ABC, abstractmethod
+from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple
+
+import httpx
+
+from litellm.llms.base_llm.chat.transformation import BaseLLMException
+from litellm.types.llms.openai_evals import (
+ CancelEvalResponse,
+ CancelRunResponse,
+ CreateEvalRequest,
+ CreateRunRequest,
+ DeleteEvalResponse,
+ Eval,
+ ListEvalsParams,
+ ListEvalsResponse,
+ ListRunsParams,
+ ListRunsResponse,
+ Run,
+ RunDeleteResponse,
+ UpdateEvalRequest,
+)
+from litellm.types.router import GenericLiteLLMParams
+from litellm.types.utils import LlmProviders
+
+if TYPE_CHECKING:
+ from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
+
+ LiteLLMLoggingObj = _LiteLLMLoggingObj
+else:
+ LiteLLMLoggingObj = Any
+
+
+class BaseEvalsAPIConfig(ABC):
+ """Base configuration for Evals API providers"""
+
+ def __init__(self):
+ pass
+
+ @property
+ @abstractmethod
+ def custom_llm_provider(self) -> LlmProviders:
+ pass
+
+ @abstractmethod
+ def validate_environment(
+ self, headers: dict, litellm_params: Optional[GenericLiteLLMParams]
+ ) -> dict:
+ """
+ Validate and update headers with provider-specific requirements
+
+ Args:
+ headers: Base headers dictionary
+ litellm_params: LiteLLM parameters
+
+ Returns:
+ Updated headers dictionary
+ """
+ return headers
+
+ @abstractmethod
+ def get_complete_url(
+ self,
+ api_base: Optional[str],
+ endpoint: str,
+ eval_id: Optional[str] = None,
+ ) -> str:
+ """
+ Get the complete URL for the API request
+
+ Args:
+ api_base: Base API URL
+ endpoint: API endpoint (e.g., 'evals', 'evals/{id}')
+ eval_id: Optional eval ID for specific eval operations
+
+ Returns:
+ Complete URL
+ """
+ if api_base is None:
+ raise ValueError("api_base is required")
+ return f"{api_base}/v1/{endpoint}"
+
+ @abstractmethod
+ def transform_create_eval_request(
+ self,
+ create_request: CreateEvalRequest,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Dict:
+ """
+ Transform create eval request to provider-specific format
+
+ Args:
+ create_request: Eval creation parameters
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Provider-specific request body
+ """
+ pass
+
+ @abstractmethod
+ def transform_create_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Eval:
+ """
+ Transform provider response to Eval object
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ Eval object
+ """
+ pass
+
+ @abstractmethod
+ def transform_list_evals_request(
+ self,
+ list_params: ListEvalsParams,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """
+ Transform list evals request parameters
+
+ Args:
+ list_params: List parameters (pagination, filters)
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, query_params)
+ """
+ pass
+
+ @abstractmethod
+ def transform_list_evals_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> ListEvalsResponse:
+ """
+ Transform provider response to ListEvalsResponse
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ ListEvalsResponse object
+ """
+ pass
+
+ @abstractmethod
+ def transform_get_eval_request(
+ self,
+ eval_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """
+ Transform get eval request
+
+ Args:
+ eval_id: Eval ID
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers)
+ """
+ pass
+
+ @abstractmethod
+ def transform_get_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Eval:
+ """
+ Transform provider response to Eval object
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ Eval object
+ """
+ pass
+
+ @abstractmethod
+ def transform_update_eval_request(
+ self,
+ eval_id: str,
+ update_request: UpdateEvalRequest,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """
+ Transform update eval request
+
+ Args:
+ eval_id: Eval ID
+ update_request: Update parameters
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers, body)
+ """
+ pass
+
+ @abstractmethod
+ def transform_update_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Eval:
+ """
+ Transform provider response to Eval object
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ Eval object
+ """
+ pass
+
+ @abstractmethod
+ def transform_delete_eval_request(
+ self,
+ eval_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """
+ Transform delete eval request
+
+ Args:
+ eval_id: Eval ID
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers)
+ """
+ pass
+
+ @abstractmethod
+ def transform_delete_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> DeleteEvalResponse:
+ """
+ Transform provider response to DeleteEvalResponse
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ DeleteEvalResponse object
+ """
+ pass
+
+ @abstractmethod
+ def transform_cancel_eval_request(
+ self,
+ eval_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """
+ Transform cancel eval request
+
+ Args:
+ eval_id: Eval ID
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers, body)
+ """
+ pass
+
+ @abstractmethod
+ def transform_cancel_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> CancelEvalResponse:
+ """
+ Transform provider response to CancelEvalResponse
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ CancelEvalResponse object
+ """
+ pass
+
+ # Run API Transformations
+ @abstractmethod
+ def transform_create_run_request(
+ self,
+ eval_id: str,
+ create_request: CreateRunRequest,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """
+ Transform create run request to provider-specific format
+
+ Args:
+ eval_id: Eval ID
+ create_request: Run creation parameters
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, request_body)
+ """
+ pass
+
+ @abstractmethod
+ def transform_create_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Run:
+ """
+ Transform provider response to Run object
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ Run object
+ """
+ pass
+
+ @abstractmethod
+ def transform_list_runs_request(
+ self,
+ eval_id: str,
+ list_params: ListRunsParams,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """
+ Transform list runs request parameters
+
+ Args:
+ eval_id: Eval ID
+ list_params: List parameters (pagination, filters)
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, query_params)
+ """
+ pass
+
+ @abstractmethod
+ def transform_list_runs_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> ListRunsResponse:
+ """
+ Transform provider response to ListRunsResponse
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ ListRunsResponse object
+ """
+ pass
+
+ @abstractmethod
+ def transform_get_run_request(
+ self,
+ eval_id: str,
+ run_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """
+ Transform get run request
+
+ Args:
+ eval_id: Eval ID
+ run_id: Run ID
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers)
+ """
+ pass
+
+ @abstractmethod
+ def transform_get_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Run:
+ """
+ Transform provider response to Run object
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ Run object
+ """
+ pass
+
+ @abstractmethod
+ def transform_cancel_run_request(
+ self,
+ eval_id: str,
+ run_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """
+ Transform cancel run request
+
+ Args:
+ eval_id: Eval ID
+ run_id: Run ID
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers, body)
+ """
+ pass
+
+ @abstractmethod
+ def transform_cancel_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> CancelRunResponse:
+ """
+ Transform provider response to CancelRunResponse
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ CancelRunResponse object
+ """
+ pass
+
+ @abstractmethod
+ def transform_delete_run_request(
+ self,
+ eval_id: str,
+ run_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """
+ Transform delete run request
+
+ Args:
+ eval_id: Eval ID
+ run_id: Run ID
+ api_base: Base API URL
+ litellm_params: LiteLLM parameters
+ headers: Request headers
+
+ Returns:
+ Tuple of (url, headers, body)
+ """
+ pass
+
+ @abstractmethod
+ def transform_delete_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> "RunDeleteResponse":
+ """
+ Transform provider response to RunDeleteResponse
+
+ Args:
+ raw_response: Raw HTTP response
+ logging_obj: Logging object
+
+ Returns:
+ RunDeleteResponse object
+ """
+ pass
+
+ def get_error_class(
+ self,
+ error_message: str,
+ status_code: int,
+ headers: dict,
+ ) -> Exception:
+ """Get appropriate error class for the provider."""
+ return BaseLLMException(
+ status_code=status_code,
+ message=error_message,
+ headers=headers,
+ )
diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py
index efa755d515e..5faae07e2b9 100644
--- a/litellm/llms/bedrock/chat/converse_transformation.py
+++ b/litellm/llms/bedrock/chat/converse_transformation.py
@@ -11,7 +11,10 @@ import httpx
import litellm
from litellm._logging import verbose_logger
-from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
+from litellm.constants import (
+ BEDROCK_MIN_THINKING_BUDGET_TOKENS,
+ RESPONSE_FORMAT_TOOL_NAME,
+)
from litellm.litellm_core_utils.core_helpers import (
filter_exceptions_from_params,
filter_internal_params,
@@ -434,6 +437,25 @@ class AmazonConverseConfig(BaseConfig):
reasoning_effort=reasoning_effort, model=model
)
+ @staticmethod
+ def _clamp_thinking_budget_tokens(optional_params: dict) -> None:
+ """
+ Clamp thinking.budget_tokens to the Bedrock minimum (1024).
+
+ Bedrock returns a 400 error if budget_tokens < 1024.
+ """
+ thinking = optional_params.get("thinking")
+ if isinstance(thinking, dict):
+ budget = thinking.get("budget_tokens")
+ if isinstance(budget, int) and budget < BEDROCK_MIN_THINKING_BUDGET_TOKENS:
+ verbose_logger.debug(
+ "Bedrock requires thinking.budget_tokens >= %d, got %d. "
+ "Clamping to minimum.",
+ BEDROCK_MIN_THINKING_BUDGET_TOKENS,
+ budget,
+ )
+ thinking["budget_tokens"] = BEDROCK_MIN_THINKING_BUDGET_TOKENS
+
def get_supported_openai_params(self, model: str) -> List[str]:
from litellm.utils import supports_function_calling
@@ -871,9 +893,14 @@ class AmazonConverseConfig(BaseConfig):
Checks 'non_default_params' for 'thinking' and 'max_tokens'
if 'thinking' is enabled and 'max_tokens' is not specified, set 'max_tokens' to the thinking token budget + DEFAULT_MAX_TOKENS
+
+ Also clamps thinking.budget_tokens to the Bedrock minimum (1024) to
+ prevent 400 errors from the Bedrock API.
"""
from litellm.constants import DEFAULT_MAX_TOKENS
+ self._clamp_thinking_budget_tokens(optional_params)
+
is_thinking_enabled = self.is_thinking_enabled(optional_params)
is_max_tokens_in_request = self.is_max_tokens_in_request(non_default_params)
if is_thinking_enabled and not is_max_tokens_in_request:
diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py
index 0ce24f63a89..bcb6edd39f9 100644
--- a/litellm/llms/chatgpt/responses/transformation.py
+++ b/litellm/llms/chatgpt/responses/transformation.py
@@ -73,10 +73,6 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
litellm_params,
headers,
)
- request.pop("max_output_tokens", None)
- request.pop("max_tokens", None)
- request.pop("max_completion_tokens", None)
- request.pop("metadata", None)
base_instructions = get_chatgpt_default_instructions()
existing_instructions = request.get("instructions")
if existing_instructions:
@@ -92,7 +88,22 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
if "reasoning.encrypted_content" not in include:
include.append("reasoning.encrypted_content")
request["include"] = include
- return request
+
+ allowed_keys = {
+ "model",
+ "input",
+ "instructions",
+ "stream",
+ "store",
+ "include",
+ "tools",
+ "tool_choice",
+ "reasoning",
+ "previous_response_id",
+ "truncation",
+ }
+
+ return {k: v for k, v in request.items() if k in allowed_keys}
def transform_response_api_response(
self,
diff --git a/litellm/llms/custom_httpx/aiohttp_transport.py b/litellm/llms/custom_httpx/aiohttp_transport.py
index fb98006c7e4..6cec1f4fe16 100644
--- a/litellm/llms/custom_httpx/aiohttp_transport.py
+++ b/litellm/llms/custom_httpx/aiohttp_transport.py
@@ -119,8 +119,13 @@ class AiohttpResponseStream(httpx.AsyncByteStream):
class AiohttpTransport(httpx.AsyncBaseTransport):
- def __init__(self, client: Union[ClientSession, Callable[[], ClientSession]]) -> None:
+ def __init__(
+ self,
+ client: Union[ClientSession, Callable[[], ClientSession]],
+ owns_session: bool = True,
+ ) -> None:
self.client = client
+ self._owns_session = owns_session
#########################################################
# Class variables for proxy settings
@@ -128,7 +133,7 @@ class AiohttpTransport(httpx.AsyncBaseTransport):
self.proxy_cache: Dict[str, Optional[str]] = {}
async def aclose(self) -> None:
- if isinstance(self.client, ClientSession):
+ if self._owns_session and isinstance(self.client, ClientSession):
await self.client.close()
@@ -144,10 +149,11 @@ class LiteLLMAiohttpTransport(AiohttpTransport):
self,
client: Union[ClientSession, Callable[[], ClientSession]],
ssl_verify: Optional[Union[bool, ssl.SSLContext]] = None,
+ owns_session: bool = True,
):
self.client = client
self._ssl_verify = ssl_verify # Store for per-request SSL override
- super().__init__(client=client)
+ super().__init__(client=client, owns_session=owns_session)
# Store the client factory for recreating sessions when needed
if callable(client):
self._client_factory = client
diff --git a/litellm/llms/custom_httpx/http_handler.py b/litellm/llms/custom_httpx/http_handler.py
index 5cf6efe5ba2..328097639e5 100644
--- a/litellm/llms/custom_httpx/http_handler.py
+++ b/litellm/llms/custom_httpx/http_handler.py
@@ -866,6 +866,7 @@ class AsyncHTTPHandler:
return LiteLLMAiohttpTransport(
client=shared_session,
ssl_verify=ssl_for_transport,
+ owns_session=False,
)
# Create new session only if none provided or existing one is invalid
diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py
index a97ebd8e74c..0a5364bfcfe 100644
--- a/litellm/llms/custom_httpx/llm_http_handler.py
+++ b/litellm/llms/custom_httpx/llm_http_handler.py
@@ -21,6 +21,9 @@ import litellm.litellm_core_utils
import litellm.types
import litellm.types.utils
from litellm._logging import verbose_logger
+from litellm.anthropic_beta_headers_manager import (
+ update_headers_with_filtered_beta,
+)
from litellm.constants import REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES
from litellm.litellm_core_utils.realtime_streaming import RealTimeStreaming
from litellm.llms.base_llm.anthropic_messages.transformation import (
@@ -34,6 +37,7 @@ from litellm.llms.base_llm.batches.transformation import BaseBatchesConfig
from litellm.llms.base_llm.chat.transformation import BaseConfig
from litellm.llms.base_llm.containers.transformation import BaseContainerConfig
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
+from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig
from litellm.llms.base_llm.files.transformation import BaseFilesConfig
from litellm.llms.base_llm.google_genai.transformation import (
BaseGoogleGenAIGenerateContentConfig,
@@ -81,9 +85,6 @@ from litellm.types.llms.anthropic_skills import (
ListSkillsResponse,
Skill,
)
-from litellm.anthropic_beta_headers_manager import (
- update_headers_with_filtered_beta,
- )
from litellm.types.llms.openai import (
CreateBatchRequest,
CreateFileRequest,
@@ -133,6 +134,16 @@ if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig
+ from litellm.types.llms.openai_evals import (
+ CancelEvalResponse,
+ CancelRunResponse,
+ DeleteEvalResponse,
+ Eval,
+ ListEvalsResponse,
+ ListRunsResponse,
+ Run,
+ RunDeleteResponse,
+ )
LiteLLMLoggingObj = _LiteLLMLoggingObj
else:
@@ -4599,6 +4610,7 @@ class BaseLLMHTTPHandler:
BaseSkillsAPIConfig,
"BasePassthroughConfig",
"BaseContainerConfig",
+ BaseEvalsAPIConfig,
],
):
status_code = getattr(e, "status_code", 500)
@@ -9316,3 +9328,1209 @@ class BaseLLMHTTPHandler:
raw_response=response,
logging_obj=logging_obj,
)
+
+ # ===================================
+ # Evals API Handlers
+ # ===================================
+
+ def create_eval_handler(
+ self,
+ url: str,
+ request_body: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["Eval", Coroutine[Any, Any, "Eval"]]:
+ """Create an eval"""
+ if _is_async:
+ return self.async_create_eval_handler(
+ url=url,
+ request_body=request_body,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input=request_body.get("display_name", ""),
+ api_key="",
+ additional_args={
+ "complete_input_dict": request_body,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.post(
+ url=url, headers=headers, json=request_body, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_create_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_create_eval_handler(
+ self,
+ url: str,
+ request_body: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "Eval":
+ """Async create an eval"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input=request_body.get("name", ""),
+ api_key="",
+ additional_args={
+ "complete_input_dict": request_body,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.post(
+ url=url, headers=headers, json=request_body, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_create_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def list_evals_handler(
+ self,
+ url: str,
+ query_params: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["ListEvalsResponse", Coroutine[Any, Any, "ListEvalsResponse"]]:
+ """List evals"""
+ if _is_async:
+ return self.async_list_evals_handler(
+ url=url,
+ query_params=query_params,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "complete_input_dict": query_params,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(
+ url=url, headers=headers, params=query_params
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_list_evals_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_list_evals_handler(
+ self,
+ url: str,
+ query_params: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "ListEvalsResponse":
+ """Async list evals"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "complete_input_dict": query_params,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers, params=query_params
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_list_evals_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def get_eval_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["Eval", Coroutine[Any, Any, "Eval"]]:
+ """Get an eval"""
+ if _is_async:
+ return self.async_get_eval_handler(
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(url=url, headers=headers)
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_get_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_get_eval_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "Eval":
+ """Async get an eval"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_get_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def update_eval_handler(
+ self,
+ url: str,
+ request_body: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["Eval", Coroutine[Any, Any, "Eval"]]:
+ """Update an eval"""
+ if _is_async:
+ return self.async_update_eval_handler(
+ url=url,
+ request_body=request_body,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input=request_body.get("display_name", ""),
+ api_key="",
+ additional_args={
+ "complete_input_dict": request_body,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.post(
+ url=url, headers=headers, json=request_body, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_update_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_update_eval_handler(
+ self,
+ url: str,
+ request_body: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "Eval":
+ """Async update an eval"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input=request_body.get("display_name", ""),
+ api_key="",
+ additional_args={
+ "complete_input_dict": request_body,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.post(
+ url=url, headers=headers, json=request_body, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_update_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def delete_eval_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["DeleteEvalResponse", Coroutine[Any, Any, "DeleteEvalResponse"]]:
+ """Delete an eval"""
+ if _is_async:
+ return self.async_delete_eval_handler(
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.delete(
+ url=url, headers=headers, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_delete_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_delete_eval_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "DeleteEvalResponse":
+ """Async delete an eval"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.delete(
+ url=url, headers=headers, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_delete_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def cancel_eval_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["CancelEvalResponse", Coroutine[Any, Any, "CancelEvalResponse"]]:
+ """Cancel an eval"""
+ if _is_async:
+ return self.async_cancel_eval_handler(
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.post(
+ url=url, headers=headers, json={}, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_cancel_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_cancel_eval_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "CancelEvalResponse":
+ """Async cancel an eval"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.post(
+ url=url, headers=headers, json={}, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_cancel_eval_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ # ===================================
+ # Eval Runs API Handlers
+ # ===================================
+
+ def create_run_handler(
+ self,
+ url: str,
+ request_body: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["Run", Coroutine[Any, Any, "Run"]]:
+ """Create a run"""
+ if _is_async:
+ return self.async_create_run_handler(
+ url=url,
+ request_body=request_body,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input=request_body.get("name", ""),
+ api_key="",
+ additional_args={
+ "complete_input_dict": request_body,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.post(
+ url=url, headers=headers, json=request_body, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_create_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_create_run_handler(
+ self,
+ url: str,
+ request_body: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "Run":
+ """Async create a run"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input=request_body.get("name", ""),
+ api_key="",
+ additional_args={
+ "complete_input_dict": request_body,
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.post(
+ url=url, headers=headers, json=request_body, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_create_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def list_runs_handler(
+ self,
+ url: str,
+ query_params: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["ListRunsResponse", Coroutine[Any, Any, "ListRunsResponse"]]:
+ """List runs"""
+ if _is_async:
+ return self.async_list_runs_handler(
+ url=url,
+ query_params=query_params,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "params": query_params,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(
+ url=url, headers=headers, params=query_params
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_list_runs_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_list_runs_handler(
+ self,
+ url: str,
+ query_params: Dict,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "ListRunsResponse":
+ """Async list runs"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "params": query_params,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers, params=query_params
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_list_runs_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def get_run_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["Run", Coroutine[Any, Any, "Run"]]:
+ """Get a run"""
+ if _is_async:
+ return self.async_get_run_handler(
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(url=url, headers=headers)
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_get_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_get_run_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "Run":
+ """Async get a run"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_get_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def cancel_run_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["CancelRunResponse", Coroutine[Any, Any, "CancelRunResponse"]]:
+ """Cancel a run"""
+ if _is_async:
+ return self.async_cancel_run_handler(
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.post(
+ url=url, headers=headers, json={}, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_cancel_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_cancel_run_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "CancelRunResponse":
+ """Async cancel a run"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.post(
+ url=url, headers=headers, json={}, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_cancel_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ def delete_run_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ _is_async: bool = False,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> Union["RunDeleteResponse", Coroutine[Any, Any, "RunDeleteResponse"]]:
+ """Delete a run"""
+ if _is_async:
+ return self.async_delete_run_handler(
+ url=url,
+ evals_api_provider_config=evals_api_provider_config,
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ logging_obj=logging_obj,
+ extra_headers=extra_headers,
+ timeout=timeout,
+ client=client,
+ shared_session=shared_session,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client(
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)}
+ )
+ else:
+ sync_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.delete(
+ url=url, headers=headers, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_delete_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
+
+ async def async_delete_run_handler(
+ self,
+ url: str,
+ evals_api_provider_config: "BaseEvalsAPIConfig",
+ custom_llm_provider: str,
+ litellm_params: GenericLiteLLMParams,
+ logging_obj: LiteLLMLoggingObj,
+ extra_headers: Optional[Dict[str, Any]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ shared_session: Optional["ClientSession"] = None,
+ ) -> "RunDeleteResponse":
+ """Async delete a run"""
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=litellm.LlmProviders(custom_llm_provider),
+ params={"ssl_verify": litellm_params.get("ssl_verify", None)},
+ )
+ else:
+ async_httpx_client = client
+
+ headers = extra_headers or {}
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.delete(
+ url=url, headers=headers, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=evals_api_provider_config,
+ )
+
+ return evals_api_provider_config.transform_delete_run_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ )
diff --git a/litellm/llms/openai/evals/__init__.py b/litellm/llms/openai/evals/__init__.py
new file mode 100644
index 00000000000..b04d27622bb
--- /dev/null
+++ b/litellm/llms/openai/evals/__init__.py
@@ -0,0 +1,7 @@
+"""
+OpenAI Evals API configuration
+"""
+
+from .transformation import OpenAIEvalsConfig
+
+__all__ = ["OpenAIEvalsConfig"]
diff --git a/litellm/llms/openai/evals/transformation.py b/litellm/llms/openai/evals/transformation.py
new file mode 100644
index 00000000000..c24dbf8637a
--- /dev/null
+++ b/litellm/llms/openai/evals/transformation.py
@@ -0,0 +1,426 @@
+"""
+OpenAI Evals API configuration and transformations
+"""
+
+from typing import Any, Dict, Optional, Tuple
+
+import httpx
+
+from litellm._logging import verbose_logger
+from litellm.llms.base_llm.evals.transformation import (
+ BaseEvalsAPIConfig,
+ LiteLLMLoggingObj,
+)
+from litellm.types.llms.openai_evals import (
+ CancelEvalResponse,
+ CancelRunResponse,
+ CreateEvalRequest,
+ CreateRunRequest,
+ DeleteEvalResponse,
+ Eval,
+ ListEvalsParams,
+ ListEvalsResponse,
+ ListRunsParams,
+ ListRunsResponse,
+ Run,
+ RunDeleteResponse,
+ UpdateEvalRequest,
+)
+from litellm.types.router import GenericLiteLLMParams
+from litellm.types.utils import LlmProviders
+
+
+class OpenAIEvalsConfig(BaseEvalsAPIConfig):
+ """OpenAI-specific Evals API configuration"""
+
+ @property
+ def custom_llm_provider(self) -> LlmProviders:
+ return LlmProviders.OPENAI
+
+ def validate_environment(
+ self, headers: dict, litellm_params: Optional[GenericLiteLLMParams]
+ ) -> dict:
+ """Add OpenAI-specific headers"""
+ import litellm
+ from litellm.secret_managers.main import get_secret_str
+
+ # Get API key following OpenAI pattern
+ api_key = None
+ if litellm_params:
+ api_key = litellm_params.api_key
+
+ api_key = (
+ api_key
+ or litellm.api_key
+ or litellm.openai_key
+ or get_secret_str("OPENAI_API_KEY")
+ )
+
+ if not api_key:
+ raise ValueError("OPENAI_API_KEY is required for Evals API")
+
+ # Add required headers
+ headers["Authorization"] = f"Bearer {api_key}"
+ headers["Content-Type"] = "application/json"
+
+ return headers
+
+ def get_complete_url(
+ self,
+ api_base: Optional[str],
+ endpoint: str,
+ eval_id: Optional[str] = None,
+ ) -> str:
+ """Get complete URL for OpenAI Evals API"""
+ if api_base is None:
+ api_base = "https://api.openai.com"
+
+ if eval_id:
+ return f"{api_base}/v1/evals/{eval_id}"
+ return f"{api_base}/v1/{endpoint}"
+
+ def transform_create_eval_request(
+ self,
+ create_request: CreateEvalRequest,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Dict:
+ """Transform create eval request for OpenAI"""
+ verbose_logger.debug("Transforming create eval request: %s", create_request)
+
+ # OpenAI expects the request body directly
+ request_body = {k: v for k, v in create_request.items() if v is not None}
+
+ return request_body
+
+ def transform_create_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Eval:
+ """Transform OpenAI response to Eval object"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming create eval response: %s", response_json)
+
+ return Eval(**response_json)
+
+ def transform_list_evals_request(
+ self,
+ list_params: ListEvalsParams,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """Transform list evals request for OpenAI"""
+ api_base = "https://api.openai.com"
+ if litellm_params and litellm_params.api_base:
+ api_base = litellm_params.api_base
+
+ url = self.get_complete_url(api_base=api_base, endpoint="evals")
+
+ # Build query parameters
+ query_params: Dict[str, Any] = {}
+ if "limit" in list_params and list_params["limit"]:
+ query_params["limit"] = list_params["limit"]
+ if "after" in list_params and list_params["after"]:
+ query_params["after"] = list_params["after"]
+ if "before" in list_params and list_params["before"]:
+ query_params["before"] = list_params["before"]
+ if "order" in list_params and list_params["order"]:
+ query_params["order"] = list_params["order"]
+ if "order_by" in list_params and list_params["order_by"]:
+ query_params["order_by"] = list_params["order_by"]
+
+ verbose_logger.debug(
+ "List evals request made to OpenAI Evals endpoint with params: %s",
+ query_params,
+ )
+
+ return url, query_params
+
+ def transform_list_evals_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> ListEvalsResponse:
+ """Transform OpenAI response to ListEvalsResponse"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming list evals response: %s", response_json)
+
+ return ListEvalsResponse(**response_json)
+
+ def transform_get_eval_request(
+ self,
+ eval_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """Transform get eval request for OpenAI"""
+ url = self.get_complete_url(
+ api_base=api_base, endpoint="evals", eval_id=eval_id
+ )
+
+ verbose_logger.debug("Get eval request - URL: %s", url)
+
+ return url, headers
+
+ def transform_get_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Eval:
+ """Transform OpenAI response to Eval object"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming get eval response: %s", response_json)
+
+ return Eval(**response_json)
+
+ def transform_update_eval_request(
+ self,
+ eval_id: str,
+ update_request: UpdateEvalRequest,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """Transform update eval request for OpenAI"""
+ url = self.get_complete_url(
+ api_base=api_base, endpoint="evals", eval_id=eval_id
+ )
+
+ # Build request body
+ request_body = {k: v for k, v in update_request.items() if v is not None}
+
+ verbose_logger.debug(
+ "Update eval request - URL: %s, body: %s", url, request_body
+ )
+
+ return url, headers, request_body
+
+ def transform_update_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Eval:
+ """Transform OpenAI response to Eval object"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming update eval response: %s", response_json)
+
+ return Eval(**response_json)
+
+ def transform_delete_eval_request(
+ self,
+ eval_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """Transform delete eval request for OpenAI"""
+ url = self.get_complete_url(
+ api_base=api_base, endpoint="evals", eval_id=eval_id
+ )
+
+ verbose_logger.debug("Delete eval request - URL: %s", url)
+
+ return url, headers
+
+ def transform_delete_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> DeleteEvalResponse:
+ """Transform OpenAI response to DeleteEvalResponse"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming delete eval response: %s", response_json)
+
+ return DeleteEvalResponse(**response_json)
+
+ def transform_cancel_eval_request(
+ self,
+ eval_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """Transform cancel eval request for OpenAI"""
+ url = f"{self.get_complete_url(api_base=api_base, endpoint='evals', eval_id=eval_id)}/cancel"
+
+ # Empty body for cancel request
+ request_body: Dict[str, Any] = {}
+
+ verbose_logger.debug("Cancel eval request - URL: %s", url)
+
+ return url, headers, request_body
+
+ def transform_cancel_eval_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> CancelEvalResponse:
+ """Transform OpenAI response to CancelEvalResponse"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming cancel eval response: %s", response_json)
+
+ return CancelEvalResponse(**response_json)
+
+ # Run API Transformations
+ def transform_create_run_request(
+ self,
+ eval_id: str,
+ create_request: CreateRunRequest,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """Transform create run request for OpenAI"""
+ api_base = "https://api.openai.com"
+ if litellm_params and litellm_params.api_base:
+ api_base = litellm_params.api_base
+
+ url = f"{api_base}/v1/evals/{eval_id}/runs"
+
+ # Build request body
+ request_body = {k: v for k, v in create_request.items() if v is not None}
+
+ verbose_logger.debug(
+ "Create run request - URL: %s, body: %s", url, request_body
+ )
+
+ return url, request_body
+
+ def transform_create_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Run:
+ """Transform OpenAI response to Run object"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming create run response: %s", response_json)
+
+ return Run(**response_json)
+
+ def transform_list_runs_request(
+ self,
+ eval_id: str,
+ list_params: ListRunsParams,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """Transform list runs request for OpenAI"""
+ api_base = "https://api.openai.com"
+ if litellm_params and litellm_params.api_base:
+ api_base = litellm_params.api_base
+
+ url = f"{api_base}/v1/evals/{eval_id}/runs"
+
+ # Build query parameters
+ query_params: Dict[str, Any] = {}
+ if "limit" in list_params and list_params["limit"]:
+ query_params["limit"] = list_params["limit"]
+ if "after" in list_params and list_params["after"]:
+ query_params["after"] = list_params["after"]
+ if "before" in list_params and list_params["before"]:
+ query_params["before"] = list_params["before"]
+ if "order" in list_params and list_params["order"]:
+ query_params["order"] = list_params["order"]
+
+ verbose_logger.debug(
+ "List runs request made to OpenAI Evals endpoint with params: %s",
+ query_params,
+ )
+
+ return url, query_params
+
+ def transform_list_runs_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> ListRunsResponse:
+ """Transform OpenAI response to ListRunsResponse"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming list runs response: %s", response_json)
+
+ return ListRunsResponse(**response_json)
+
+ def transform_get_run_request(
+ self,
+ eval_id: str,
+ run_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict]:
+ """Transform get run request for OpenAI"""
+ url = f"{api_base}/v1/evals/{eval_id}/runs/{run_id}"
+
+ verbose_logger.debug("Get run request - URL: %s", url)
+
+ return url, headers
+
+ def transform_get_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> Run:
+ """Transform OpenAI response to Run object"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming get run response: %s", response_json)
+
+ return Run(**response_json)
+
+ def transform_cancel_run_request(
+ self,
+ eval_id: str,
+ run_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """Transform cancel run request for OpenAI"""
+ url = f"{api_base}/v1/evals/{eval_id}/runs/{run_id}/cancel"
+
+ # Empty body for cancel request
+ request_body: Dict[str, Any] = {}
+
+ verbose_logger.debug("Cancel run request - URL: %s", url)
+
+ return url, headers, request_body
+
+ def transform_cancel_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> CancelRunResponse:
+ """Transform OpenAI response to CancelRunResponse"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming cancel run response: %s", response_json)
+
+ return CancelRunResponse(**response_json)
+
+ def transform_delete_run_request(
+ self,
+ eval_id: str,
+ run_id: str,
+ api_base: str,
+ litellm_params: GenericLiteLLMParams,
+ headers: dict,
+ ) -> Tuple[str, Dict, Dict]:
+ """Transform delete run request for OpenAI"""
+ url = f"{api_base}/v1/evals/{eval_id}/runs/{run_id}"
+
+ # Empty body for delete request
+ request_body: Dict[str, Any] = {}
+
+ verbose_logger.debug("Delete run request - URL: %s", url)
+
+ return url, headers, request_body
+
+ def transform_delete_run_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ ) -> RunDeleteResponse:
+ """Transform OpenAI response to RunDeleteResponse"""
+ response_json = raw_response.json()
+ verbose_logger.debug("Transforming delete run response: %s", response_json)
+
+ return RunDeleteResponse(**response_json)
diff --git a/litellm/llms/openai_like/dynamic_config.py b/litellm/llms/openai_like/dynamic_config.py
index 1e7866bebbe..a2ce6b9a531 100644
--- a/litellm/llms/openai_like/dynamic_config.py
+++ b/litellm/llms/openai_like/dynamic_config.py
@@ -4,6 +4,7 @@ Dynamic configuration class generator for JSON-based providers.
from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, overload
+from litellm._logging import verbose_logger
from litellm.litellm_core_utils.prompt_templates.common_utils import (
handle_messages_with_content_list_to_str_conversion,
)
@@ -96,8 +97,27 @@ def create_config_class(provider: SimpleProviderConfig):
return api_base
def get_supported_openai_params(self, model: str) -> list:
- """Get supported OpenAI params from base class"""
- return super().get_supported_openai_params(model=model)
+ """Get supported OpenAI params, excluding tool-related params for models
+ that don't support function calling."""
+ from litellm.utils import supports_function_calling
+
+ supported_params = super().get_supported_openai_params(model=model)
+
+ _supports_fc = supports_function_calling(
+ model=model, custom_llm_provider=provider.slug
+ )
+
+ if not _supports_fc:
+ tool_params = ["tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"]
+ for param in tool_params:
+ if param in supported_params:
+ supported_params.remove(param)
+ verbose_logger.debug(
+ f"Model {model} on provider {provider.slug} does not support "
+ f"function calling — removed tool-related params from supported params."
+ )
+
+ return supported_params
def map_openai_params(
self,
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 95d8ba2ff60..9a9acb91986 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -1083,7 +1083,7 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
- "apac.anthropic.claude-opus-4-6-v1": {
+ "au.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
@@ -1113,6 +1113,156 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
+ "anthropic.claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 3.75e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
+ "cache_read_input_token_cost": 3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6e-07,
+ "input_cost_per_token": 3e-06,
+ "input_cost_per_token_above_200k_tokens": 6e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-05,
+ "output_cost_per_token_above_200k_tokens": 2.25e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
+ "global.anthropic.claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 3.75e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
+ "cache_read_input_token_cost": 3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6e-07,
+ "input_cost_per_token": 3e-06,
+ "input_cost_per_token_above_200k_tokens": 6e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-05,
+ "output_cost_per_token_above_200k_tokens": 2.25e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
+ "us.anthropic.claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 4.125e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
+ "cache_read_input_token_cost": 3.3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
+ "input_cost_per_token": 3.3e-06,
+ "input_cost_per_token_above_200k_tokens": 6.6e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.65e-05,
+ "output_cost_per_token_above_200k_tokens": 2.475e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
+ "eu.anthropic.claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 4.125e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
+ "cache_read_input_token_cost": 3.3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
+ "input_cost_per_token": 3.3e-06,
+ "input_cost_per_token_above_200k_tokens": 6.6e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.65e-05,
+ "output_cost_per_token_above_200k_tokens": 2.475e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
+ "apac.anthropic.claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 4.125e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
+ "cache_read_input_token_cost": 3.3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
+ "input_cost_per_token": 3.3e-06,
+ "input_cost_per_token_above_200k_tokens": 6.6e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.65e-05,
+ "output_cost_per_token_above_200k_tokens": 2.475e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
"anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_read_input_token_cost": 3e-07,
@@ -1663,6 +1813,28 @@
"supports_tool_choice": true,
"supports_vision": true
},
+ "azure_ai/claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 3.75e-06,
+ "cache_creation_input_token_cost_above_1hr": 6e-06,
+ "cache_read_input_token_cost": 3e-07,
+ "input_cost_per_token": 3e-06,
+ "litellm_provider": "azure_ai",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-05,
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
"azure/computer-use-preview": {
"input_cost_per_token": 3e-06,
"litellm_provider": "azure",
@@ -8092,6 +8264,36 @@
"supports_web_search": true,
"tool_use_system_prompt_tokens": 346
},
+ "claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 3.75e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
+ "cache_read_input_token_cost": 3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6e-07,
+ "input_cost_per_token": 3e-06,
+ "input_cost_per_token_above_200k_tokens": 6e-06,
+ "litellm_provider": "anthropic",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-05,
+ "output_cost_per_token_above_200k_tokens": 2.25e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
"claude-sonnet-4-5-20250929-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_read_input_token_cost": 3e-07,
@@ -12456,6 +12658,19 @@
"supports_tool_choice": true,
"supports_web_search": true
},
+ "fireworks_ai/accounts/fireworks/models/kimi-k2p5": {
+ "input_cost_per_token": 6e-07,
+ "litellm_provider": "fireworks_ai",
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "max_tokens": 262144,
+ "mode": "chat",
+ "output_cost_per_token": 3e-06,
+ "source": "https://fireworks.ai/pricing",
+ "supports_function_calling": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true
+ },
"fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct": {
"input_cost_per_token": 3e-06,
"litellm_provider": "fireworks_ai",
@@ -17099,6 +17314,19 @@
"supports_parallel_function_calling": true,
"supports_vision": true
},
+ "github_copilot/claude-opus-4.6-fast": {
+ "litellm_provider": "github_copilot",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 16000,
+ "max_tokens": 16000,
+ "mode": "chat",
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
+ "supports_function_calling": true,
+ "supports_parallel_function_calling": true,
+ "supports_vision": true
+ },
"github_copilot/claude-opus-41": {
"litellm_provider": "github_copilot",
"max_input_tokens": 80000,
@@ -17350,6 +17578,20 @@
"supports_response_schema": true,
"supports_vision": true
},
+ "github_copilot/gpt-5.3-codex": {
+ "litellm_provider": "github_copilot",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "max_tokens": 128000,
+ "mode": "responses",
+ "supported_endpoints": [
+ "/v1/responses"
+ ],
+ "supports_function_calling": true,
+ "supports_parallel_function_calling": true,
+ "supports_response_schema": true,
+ "supports_vision": true
+ },
"github_copilot/text-embedding-3-small": {
"litellm_provider": "github_copilot",
"max_input_tokens": 8191,
@@ -23759,7 +24001,7 @@
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
- "output_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 1.5e-05,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing",
"supports_function_calling": true,
"supports_response_schema": false
@@ -23807,7 +24049,7 @@
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
- "output_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 1.5e-05,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing",
"supports_function_calling": true,
"supports_response_schema": false
@@ -30342,6 +30584,36 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
+ "vertex_ai/claude-opus-4-6@default": {
+ "cache_creation_input_token_cost": 6.25e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
+ "cache_read_input_token_cost": 5e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 1e-06,
+ "input_cost_per_token": 5e-06,
+ "input_cost_per_token_above_200k_tokens": 1e-05,
+ "litellm_provider": "vertex_ai-anthropic_models",
+ "max_input_tokens": 1000000,
+ "max_output_tokens": 128000,
+ "max_tokens": 128000,
+ "mode": "chat",
+ "output_cost_per_token": 2.5e-05,
+ "output_cost_per_token_above_200k_tokens": 3.75e-05,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ },
+ "supports_assistant_prefill": false,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346
+ },
"vertex_ai/claude-sonnet-4-5": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_read_input_token_cost": 3e-07,
@@ -30368,6 +30640,36 @@
"supports_tool_choice": true,
"supports_vision": true
},
+ "vertex_ai/claude-sonnet-4-6": {
+ "cache_creation_input_token_cost": 3.75e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
+ "cache_read_input_token_cost": 3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6e-07,
+ "input_cost_per_token": 3e-06,
+ "input_cost_per_token_above_200k_tokens": 6e-06,
+ "litellm_provider": "vertex_ai-anthropic_models",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-05,
+ "output_cost_per_token_above_200k_tokens": 2.25e-05,
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ }
+ },
"vertex_ai/claude-sonnet-4-5@20250929": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_read_input_token_cost": 3e-07,
@@ -36938,5 +37240,35 @@
"supports_vision": true,
"supports_web_search": true,
"tpm": 8000000
+ },
+ "vertex_ai/claude-sonnet-4-6@default": {
+ "cache_creation_input_token_cost": 3.75e-06,
+ "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
+ "cache_read_input_token_cost": 3e-07,
+ "cache_read_input_token_cost_above_200k_tokens": 6e-07,
+ "input_cost_per_token": 3e-06,
+ "input_cost_per_token_above_200k_tokens": 6e-06,
+ "litellm_provider": "vertex_ai-anthropic_models",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 64000,
+ "max_tokens": 64000,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-05,
+ "output_cost_per_token_above_200k_tokens": 2.25e-05,
+ "supports_assistant_prefill": true,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_pdf_input": true,
+ "supports_prompt_caching": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true,
+ "supports_tool_choice": true,
+ "supports_vision": true,
+ "tool_use_system_prompt_tokens": 346,
+ "search_context_cost_per_query": {
+ "search_context_size_high": 0.01,
+ "search_context_size_low": 0.01,
+ "search_context_size_medium": 0.01
+ }
}
-}
+}
\ No newline at end of file
diff --git a/litellm/policy_templates_backup.json b/litellm/policy_templates_backup.json
index a0ffd6acd30..ef272ca7749 100644
--- a/litellm/policy_templates_backup.json
+++ b/litellm/policy_templates_backup.json
@@ -3,6 +3,7 @@
"id": "advanced-au-pii-protection",
"title": "Advanced PII Protection (Australia)",
"description": "Protects Australian-specific identifiers, international employee data, financial information, credentials, protected class information, and industry-specific sensitive data.",
+ "region": "AU",
"icon": "ShieldCheckIcon",
"iconColor": "text-purple-500",
"iconBg": "bg-purple-50",
@@ -206,6 +207,7 @@
"id": "baseline-pii-protection",
"title": "Baseline PII Protection",
"description": "Baseline PII protection for internal tools and testing. Focuses on credentials and high-risk identifiers only. Suitable for non-sensitive internal use.",
+ "region": "Global",
"icon": "ShieldCheckIcon",
"iconColor": "text-blue-500",
"iconBg": "bg-blue-50",
@@ -279,6 +281,7 @@
"id": "nsfw-content-filter-australia",
"title": "NSFW Content Filter (Australia)",
"description": "Blocks profanity, sexual content, NSFW requests, self-harm content, and child safety violations using English and Australian-specific slang. Protects against inappropriate content including sexual solicitation, explicit content, Australian profanity, self-harm, and content involving minors.",
+ "region": "AU",
"icon": "ShieldExclamationIcon",
"iconColor": "text-red-500",
"iconBg": "bg-red-50",
@@ -399,6 +402,7 @@
"id": "nsfw-content-filter-basic",
"title": "NSFW Content Filter (Basic)",
"description": "Basic NSFW content filtering for English only. Blocks profanity, sexual content, slurs, solicitation, explicit requests, self-harm content, and child safety violations. Suitable for most applications requiring content moderation.",
+ "region": "Global",
"icon": "ShieldExclamationIcon",
"iconColor": "text-orange-500",
"iconBg": "bg-orange-50",
@@ -499,6 +503,7 @@
"id": "nsfw-content-filter-all-regions",
"title": "NSFW Content Filter (All Regions)",
"description": "Comprehensive multi-language NSFW content filtering. Blocks profanity, sexual content, inappropriate requests, self-harm content, and child safety violations in English, Spanish, French, German, and Australian. Best for global applications.",
+ "region": "Global",
"icon": "ShieldExclamationIcon",
"iconColor": "text-purple-500",
"iconBg": "bg-purple-50",
@@ -674,5 +679,446 @@
],
"guardrails_remove": []
}
+ },
+ {
+ "id": "gdpr-eu-pii-protection",
+ "title": "GDPR Art. 32 — EU PII Protection",
+ "description": "GDPR Article 32 compliance for EU personal data protection. Masks French national IDs (NIR/INSEE), EU IBANs, French phone numbers, EU VAT numbers, EU passport numbers, and email addresses. Suitable for applications processing EU citizen data requiring GDPR compliance.",
+ "region": "EU",
+ "icon": "ShieldCheckIcon",
+ "iconColor": "text-indigo-500",
+ "iconBg": "bg-indigo-50",
+ "guardrails": [
+ "gdpr-eu-national-identifiers",
+ "gdpr-eu-financial-data",
+ "gdpr-eu-contact-information",
+ "gdpr-eu-business-identifiers"
+ ],
+ "complexity": "Medium",
+ "guardrailDefinitions": [
+ {
+ "guardrail_name": "gdpr-eu-national-identifiers",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "patterns": [
+ {"pattern_type": "prebuilt", "pattern_name": "fr_nir", "action": "MASK"},
+ {"pattern_type": "prebuilt", "pattern_name": "eu_passport_generic", "action": "MASK"}
+ ],
+ "pattern_redaction_format": "[{pattern_name}_REDACTED]"
+ },
+ "guardrail_info": {
+ "description": "Masks EU national identification numbers including French NIR/INSEE and EU passport numbers for GDPR compliance"
+ }
+ },
+ {
+ "guardrail_name": "gdpr-eu-financial-data",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "patterns": [
+ {"pattern_type": "prebuilt", "pattern_name": "eu_iban_enhanced", "action": "MASK"},
+ {"pattern_type": "prebuilt", "pattern_name": "iban", "action": "MASK"}
+ ],
+ "pattern_redaction_format": "[IBAN_REDACTED]"
+ },
+ "guardrail_info": {
+ "description": "Masks EU bank account numbers (IBANs) to protect financial data under GDPR Article 32"
+ }
+ },
+ {
+ "guardrail_name": "gdpr-eu-contact-information",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "patterns": [
+ {"pattern_type": "prebuilt", "pattern_name": "email", "action": "MASK"},
+ {"pattern_type": "prebuilt", "pattern_name": "fr_phone", "action": "MASK"},
+ {"pattern_type": "prebuilt", "pattern_name": "fr_postal_code", "action": "MASK"}
+ ],
+ "pattern_redaction_format": "[{pattern_name}_REDACTED]"
+ },
+ "guardrail_info": {
+ "description": "Masks contact information including emails, French phone numbers, and postal codes for EU data subjects"
+ }
+ },
+ {
+ "guardrail_name": "gdpr-eu-business-identifiers",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "patterns": [
+ {"pattern_type": "prebuilt", "pattern_name": "eu_vat", "action": "MASK"}
+ ],
+ "pattern_redaction_format": "[VAT_NUMBER_REDACTED]"
+ },
+ "guardrail_info": {
+ "description": "Masks EU VAT identification numbers to protect business entity information under GDPR"
+ }
+ }
+ ],
+ "templateData": {
+ "policy_name": "gdpr-eu-pii-protection",
+ "description": "GDPR Article 32 compliance policy for EU personal data protection. Masks French national IDs, EU IBANs, phone numbers, VAT numbers, passports, and contact information.",
+ "guardrails_add": [
+ "gdpr-eu-national-identifiers",
+ "gdpr-eu-financial-data",
+ "gdpr-eu-contact-information",
+ "gdpr-eu-business-identifiers"
+ ],
+ "guardrails_remove": []
+ }
+ },
+ {
+ "id": "eu-ai-act-article5",
+ "title": "EU AI Act Article 5 — Prohibited Practices",
+ "description": "Comprehensive EU AI Act Article 5 compliance covering all prohibited AI practices. Includes 5 dedicated sub-guardrails per language (English + French) for: subliminal manipulation (Art. 5.1a), vulnerability exploitation (Art. 5.1b), social scoring (Art. 5.1c), emotion recognition in workplace/education (Art. 5.1f), and biometric categorization & predictive profiling (Art. 5.1d/g/h). Uses conditional matching (identifier word + context word).",
+ "region": "EU",
+ "icon": "ShieldExclamationIcon",
+ "iconColor": "text-red-500",
+ "iconBg": "bg-red-50",
+ "guardrails": [
+ "eu-ai-act-art5-manipulation",
+ "eu-ai-act-art5-vulnerability",
+ "eu-ai-act-art5-social-scoring",
+ "eu-ai-act-art5-emotion-recognition",
+ "eu-ai-act-art5-biometric-profiling",
+ "eu-ai-act-art5-manipulation-fr",
+ "eu-ai-act-art5-vulnerability-fr",
+ "eu-ai-act-art5-social-scoring-fr",
+ "eu-ai-act-art5-emotion-recognition-fr",
+ "eu-ai-act-art5-biometric-profiling-fr"
+ ],
+ "complexity": "High",
+ "guardrailDefinitions": [
+ {
+ "guardrail_name": "eu-ai-act-art5-manipulation",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_manipulation",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_manipulation.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(a) — Blocks subliminal manipulation, deceptive AI techniques, dark patterns, and covert behavioral influence"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-vulnerability",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_vulnerability",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_vulnerability.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(b) — Blocks AI systems that exploit vulnerabilities of children, elderly, disabled persons, or economically disadvantaged groups"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-social-scoring",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_social_scoring",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_social_scoring.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(c) — Blocks social credit systems, citizen scoring, trustworthiness classification, and behavioral reputation scoring"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-emotion-recognition",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_emotion_recognition",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_emotion_recognition.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(f) — Blocks emotion recognition, mood tracking, and sentiment analysis in workplace and educational settings"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-biometric-profiling",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_biometric_profiling",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_biometric_profiling.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(d)(g)(h) — Blocks biometric categorization by race/ethnicity/religion/politics, facial recognition database scraping, and predictive policing"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-manipulation-fr",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_manipulation_fr",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_manipulation_fr.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(a) FR — Bloque la manipulation subliminale, les techniques d'IA trompeuses et les dark patterns (français)"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-vulnerability-fr",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_vulnerability_fr",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_vulnerability_fr.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(b) FR — Bloque l'exploitation des vulnérabilités des enfants, personnes âgées et handicapées (français)"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-social-scoring-fr",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_social_scoring_fr",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_social_scoring_fr.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(c) FR — Bloque les systèmes de crédit social, notation des citoyens et classification de fiabilité (français)"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-emotion-recognition-fr",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_emotion_recognition_fr",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_emotion_recognition_fr.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(f) FR — Bloque la reconnaissance des émotions et l'analyse des sentiments au travail et dans l'éducation (français)"
+ }
+ },
+ {
+ "guardrail_name": "eu-ai-act-art5-biometric-profiling-fr",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "eu_ai_act_art5_biometric_profiling_fr",
+ "category_file": "litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/policy_templates/eu_ai_act_art5_biometric_profiling_fr.yaml",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Art. 5.1(d)(g)(h) FR — Bloque la catégorisation biométrique, les bases de reconnaissance faciale et le profilage prédictif (français)"
+ }
+ }
+ ],
+ "templateData": {
+ "policy_name": "eu-ai-act-article5",
+ "description": "Comprehensive EU AI Act Article 5 compliance policy. Covers all prohibited AI practices across 5 sub-guardrails per language: subliminal manipulation (Art. 5.1a), vulnerability exploitation (Art. 5.1b), social scoring (Art. 5.1c), emotion recognition (Art. 5.1f), and biometric categorization & predictive profiling (Art. 5.1d/g/h). Includes English and French detection.",
+ "guardrails_add": [
+ "eu-ai-act-art5-manipulation",
+ "eu-ai-act-art5-vulnerability",
+ "eu-ai-act-art5-social-scoring",
+ "eu-ai-act-art5-emotion-recognition",
+ "eu-ai-act-art5-biometric-profiling",
+ "eu-ai-act-art5-manipulation-fr",
+ "eu-ai-act-art5-vulnerability-fr",
+ "eu-ai-act-art5-social-scoring-fr",
+ "eu-ai-act-art5-emotion-recognition-fr",
+ "eu-ai-act-art5-biometric-profiling-fr"
+ ],
+ "guardrails_remove": []
+ }
+ },
+ {
+ "id": "prompt-injection-detection",
+ "title": "Prompt Injection Detection",
+ "description": "Detects and blocks prompt injection attacks including SQL injection, malicious code injection, system prompt extraction, jailbreak attempts, and data exfiltration. Applies pre-call screening to block attacks before they reach the LLM.",
+ "region": "Global",
+ "icon": "ShieldExclamationIcon",
+ "iconColor": "text-red-500",
+ "iconBg": "bg-red-50",
+ "guardrails": [
+ "prompt-injection-sql",
+ "prompt-injection-malicious-code",
+ "prompt-injection-system-prompt",
+ "prompt-injection-jailbreak",
+ "prompt-injection-data-exfiltration"
+ ],
+ "complexity": "Medium",
+ "guardrailDefinitions": [
+ {
+ "guardrail_name": "prompt-injection-sql",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "prompt_injection_sql",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Blocks SQL injection attempts in prompts (DROP TABLE, UNION SELECT, OR 1=1, etc.)"
+ }
+ },
+ {
+ "guardrail_name": "prompt-injection-malicious-code",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "prompt_injection_malicious_code",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Blocks malicious code injection attempts (shell commands, reverse shells, script injection, encoded payloads)"
+ }
+ },
+ {
+ "guardrail_name": "prompt-injection-system-prompt",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "prompt_injection_system_prompt",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Blocks system prompt extraction and instruction override attempts (ignore previous instructions, reveal your prompt, etc.)"
+ }
+ },
+ {
+ "guardrail_name": "prompt-injection-jailbreak",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "prompt_injection_jailbreak",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Blocks jailbreak attempts (DAN mode, developer mode, safety bypass, token smuggling)"
+ }
+ },
+ {
+ "guardrail_name": "prompt-injection-data-exfiltration",
+ "litellm_params": {
+ "guardrail": "litellm_content_filter",
+ "mode": "pre_call",
+ "categories": [
+ {
+ "category": "prompt_injection_data_exfiltration",
+ "enabled": true,
+ "action": "BLOCK",
+ "severity_threshold": "medium"
+ }
+ ]
+ },
+ "guardrail_info": {
+ "description": "Blocks data exfiltration attempts (extract training data, dump database, steal credentials, etc.)"
+ }
+ }
+ ],
+ "templateData": {
+ "policy_name": "prompt-injection-detection",
+ "description": "Prompt injection detection policy. Blocks SQL injection, malicious code injection, system prompt extraction, jailbreak attempts, and data exfiltration in prompts before they reach the LLM.",
+ "guardrails_add": [
+ "prompt-injection-sql",
+ "prompt-injection-malicious-code",
+ "prompt-injection-system-prompt",
+ "prompt-injection-jailbreak",
+ "prompt-injection-data-exfiltration"
+ ],
+ "guardrails_remove": []
+ }
}
]
diff --git a/litellm/proxy/_experimental/mcp_server/server.py b/litellm/proxy/_experimental/mcp_server/server.py
index ba107a9dd10..31836a27509 100644
--- a/litellm/proxy/_experimental/mcp_server/server.py
+++ b/litellm/proxy/_experimental/mcp_server/server.py
@@ -149,7 +149,7 @@ if MCP_AVAILABLE:
app=server,
event_store=None,
json_response=False, # enables SSE streaming
- stateless=False, # enables session state
+ stateless=True,
)
# Create SSE session manager
diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py
index b71ae9fde60..e1cf5a1e8ba 100644
--- a/litellm/proxy/_types.py
+++ b/litellm/proxy/_types.py
@@ -834,9 +834,9 @@ class GenerateRequestBase(LiteLLMPydanticObjectBase):
allowed_cache_controls: Optional[list] = []
config: Optional[dict] = {}
permissions: Optional[dict] = {}
- model_max_budget: Optional[dict] = (
- {}
- ) # {"gpt-4": 5.0, "gpt-3.5-turbo": 5.0}, defaults to {}
+ model_max_budget: Optional[
+ dict
+ ] = {} # {"gpt-4": 5.0, "gpt-3.5-turbo": 5.0}, defaults to {}
model_config = ConfigDict(protected_namespaces=())
model_rpm_limit: Optional[dict] = None
@@ -975,6 +975,9 @@ class RegenerateKeyRequest(GenerateKeyRequest):
spend: Optional[float] = None
metadata: Optional[dict] = None
new_master_key: Optional[str] = None
+ grace_period: Optional[
+ str
+ ] = None # Duration to keep old key valid (e.g. "24h", "2d"); None = immediate revoke
class ResetSpendRequest(LiteLLMPydanticObjectBase):
@@ -1509,15 +1512,15 @@ class NewTeamRequest(TeamBase):
] = None # raise an error if 'guaranteed_throughput' is set and we're overallocating tpm
model_tpm_limit: Optional[Dict[str, int]] = None
- team_member_budget: Optional[float] = (
- None # allow user to set a budget for all team members
- )
- team_member_rpm_limit: Optional[int] = (
- None # allow user to set RPM limit for all team members
- )
- team_member_tpm_limit: Optional[int] = (
- None # allow user to set TPM limit for all team members
- )
+ team_member_budget: Optional[
+ float
+ ] = None # allow user to set a budget for all team members
+ team_member_rpm_limit: Optional[
+ int
+ ] = None # allow user to set RPM limit for all team members
+ team_member_tpm_limit: Optional[
+ int
+ ] = None # allow user to set TPM limit for all team members
team_member_key_duration: Optional[str] = None # e.g. "1d", "1w", "1m"
allowed_vector_store_indexes: Optional[List[AllowedVectorStoreIndexItem]] = None
@@ -1609,9 +1612,9 @@ class BlockKeyRequest(LiteLLMPydanticObjectBase):
class AddTeamCallback(LiteLLMPydanticObjectBase):
callback_name: str
- callback_type: Optional[Literal["success", "failure", "success_and_failure"]] = (
- "success_and_failure"
- )
+ callback_type: Optional[
+ Literal["success", "failure", "success_and_failure"]
+ ] = "success_and_failure"
callback_vars: Dict[str, str]
@model_validator(mode="before")
@@ -1943,9 +1946,9 @@ class ConfigList(LiteLLMPydanticObjectBase):
stored_in_db: Optional[bool]
field_default_value: Any
premium_field: bool = False
- nested_fields: Optional[List[FieldDetail]] = (
- None # For nested dictionary or Pydantic fields
- )
+ nested_fields: Optional[
+ List[FieldDetail]
+ ] = None # For nested dictionary or Pydantic fields
class UserHeaderMapping(LiteLLMPydanticObjectBase):
@@ -2387,9 +2390,9 @@ class LiteLLM_OrganizationMembershipTable(LiteLLMPydanticObjectBase):
budget_id: Optional[str] = None
created_at: datetime
updated_at: datetime
- user: Optional[Any] = (
- None # You might want to replace 'Any' with a more specific type if available
- )
+ user: Optional[
+ Any
+ ] = None # You might want to replace 'Any' with a more specific type if available
litellm_budget_table: Optional[LiteLLM_BudgetTable] = None
model_config = ConfigDict(protected_namespaces=())
@@ -3383,9 +3386,9 @@ class TeamModelDeleteRequest(BaseModel):
# Organization Member Requests
class OrganizationMemberAddRequest(OrgMemberAddRequest):
organization_id: str
- max_budget_in_organization: Optional[float] = (
- None # Users max budget within the organization
- )
+ max_budget_in_organization: Optional[
+ float
+ ] = None # Users max budget within the organization
class OrganizationMemberDeleteRequest(MemberDeleteRequest):
@@ -3603,9 +3606,9 @@ class ProviderBudgetResponse(LiteLLMPydanticObjectBase):
Maps provider names to their budget configs.
"""
- providers: Dict[str, ProviderBudgetResponseObject] = (
- {}
- ) # Dictionary mapping provider names to their budget configurations
+ providers: Dict[
+ str, ProviderBudgetResponseObject
+ ] = {} # Dictionary mapping provider names to their budget configurations
class ProxyStateVariables(TypedDict):
@@ -3748,9 +3751,9 @@ class LiteLLM_JWTAuth(LiteLLMPydanticObjectBase):
enforce_rbac: bool = False
roles_jwt_field: Optional[str] = None # v2 on role mappings
role_mappings: Optional[List[RoleMapping]] = None
- object_id_jwt_field: Optional[str] = (
- None # can be either user / team, inferred from the role mapping
- )
+ object_id_jwt_field: Optional[
+ str
+ ] = None # can be either user / team, inferred from the role mapping
scope_mappings: Optional[List[ScopeMapping]] = None
enforce_scope_based_access: bool = False
enforce_team_based_model_access: bool = False
diff --git a/litellm/proxy/batches_endpoints/endpoints.py b/litellm/proxy/batches_endpoints/endpoints.py
index 06800cb4524..143b2607feb 100644
--- a/litellm/proxy/batches_endpoints/endpoints.py
+++ b/litellm/proxy/batches_endpoints/endpoints.py
@@ -29,6 +29,7 @@ from litellm.proxy.openai_files_endpoints.common_utils import (
get_models_from_unified_file_id,
get_original_file_id,
prepare_data_with_credentials,
+ resolve_input_file_id_to_unified,
update_batch_in_database,
)
from litellm.proxy.utils import handle_exception_on_proxy, is_known_model
@@ -305,7 +306,7 @@ async def create_batch( # noqa: PLR0915
dependencies=[Depends(user_api_key_auth)],
tags=["batch"],
)
-async def retrieve_batch(
+async def retrieve_batch( # noqa: PLR0915
request: Request,
fastapi_response: Response,
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
@@ -377,6 +378,11 @@ async def retrieve_batch(
response = await proxy_logging_obj.post_call_success_hook(
data=data, user_api_key_dict=user_api_key_dict, response=response
)
+
+ # async_post_call_success_hook replaces batch.id and output_file_id with unified IDs
+ # but not input_file_id. Resolve raw provider ID to unified ID.
+ if unified_batch_id:
+ await resolve_input_file_id_to_unified(response, prisma_client)
asyncio.create_task(
proxy_logging_obj.update_request_status(
@@ -479,6 +485,11 @@ async def retrieve_batch(
data=data, user_api_key_dict=user_api_key_dict, response=response
)
+ # Fix: bug_feb14_batch_retrieve_returns_raw_input_file_id
+ # Resolve raw provider input_file_id to unified ID.
+ if unified_batch_id:
+ await resolve_input_file_id_to_unified(response, prisma_client)
+
### ALERTING ###
asyncio.create_task(
proxy_logging_obj.update_request_status(
diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py
index f0fa5e44b05..7dfa3bb239f 100644
--- a/litellm/proxy/common_request_processing.py
+++ b/litellm/proxy/common_request_processing.py
@@ -526,6 +526,17 @@ class ProxyBaseLLMRequestProcessing:
"acancel_interaction",
"asend_message",
"call_mcp_tool",
+ "acreate_eval",
+ "alist_evals",
+ "aget_eval",
+ "aupdate_eval",
+ "adelete_eval",
+ "acancel_eval",
+ "acreate_run",
+ "alist_runs",
+ "aget_run",
+ "acancel_run",
+ "adelete_run",
],
version: Optional[str] = None,
user_model: Optional[str] = None,
@@ -708,6 +719,17 @@ class ProxyBaseLLMRequestProcessing:
"acancel_interaction",
"acancel_batch",
"afile_delete",
+ "acreate_eval",
+ "alist_evals",
+ "aget_eval",
+ "aupdate_eval",
+ "adelete_eval",
+ "acancel_eval",
+ "acreate_run",
+ "alist_runs",
+ "aget_run",
+ "acancel_run",
+ "adelete_run",
],
proxy_logging_obj: ProxyLogging,
general_settings: dict,
diff --git a/litellm/proxy/common_utils/key_rotation_manager.py b/litellm/proxy/common_utils/key_rotation_manager.py
index 13bbf2272f7..5a0a1fabc7d 100644
--- a/litellm/proxy/common_utils/key_rotation_manager.py
+++ b/litellm/proxy/common_utils/key_rotation_manager.py
@@ -8,7 +8,10 @@ from datetime import datetime, timezone
from typing import List
from litellm._logging import verbose_proxy_logger
-from litellm.constants import LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME
+from litellm.constants import (
+ LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME,
+ LITELLM_KEY_ROTATION_GRACE_PERIOD,
+)
from litellm.proxy._types import (
GenerateKeyResponse,
LiteLLM_VerificationToken,
@@ -37,6 +40,9 @@ class KeyRotationManager:
try:
verbose_proxy_logger.info("Starting scheduled key rotation check...")
+ # Clean up expired deprecated keys first
+ await self._cleanup_expired_deprecated_keys()
+
# Find keys that are due for rotation
keys_to_rotate = await self._find_keys_needing_rotation()
@@ -97,6 +103,24 @@ class KeyRotationManager:
return keys_with_rotation
+ async def _cleanup_expired_deprecated_keys(self) -> None:
+ """
+ Remove deprecated key entries whose revoke_at has passed.
+ """
+ try:
+ now = datetime.now(timezone.utc)
+ result = await self.prisma_client.db.litellm_deprecatedverificationtoken.delete_many(
+ where={"revoke_at": {"lt": now}}
+ )
+ if result > 0:
+ verbose_proxy_logger.debug(
+ "Cleaned up %s expired deprecated key(s)", result
+ )
+ except Exception as e:
+ verbose_proxy_logger.debug(
+ "Deprecated key cleanup skipped (table may not exist): %s", e
+ )
+
def _should_rotate_key(self, key: LiteLLM_VerificationToken, now: datetime) -> bool:
"""
Determine if a key should be rotated based on key_rotation_at timestamp.
@@ -115,10 +139,11 @@ class KeyRotationManager:
"""
Rotate a single key using existing regenerate_key_fn and call the rotation hook
"""
- # Create regenerate request
+ # Create regenerate request with grace period for seamless cutover
regenerate_request = RegenerateKeyRequest(
key=key.token or "",
key_alias=key.key_alias, # Pass key alias to ensure correct secret is updated in AWS Secrets Manager
+ grace_period=LITELLM_KEY_ROTATION_GRACE_PERIOD or None,
)
# Create a system user for key rotation
diff --git a/litellm/proxy/common_utils/performance_utils.py b/litellm/proxy/common_utils/performance_utils.py
index f9537f85e2b..5bfa6f31c7d 100644
--- a/litellm/proxy/common_utils/performance_utils.py
+++ b/litellm/proxy/common_utils/performance_utils.py
@@ -8,10 +8,10 @@ for line-by-line profiling.
See performance_utils.md for detailed usage examples and documentation.
"""
-import asyncio
import atexit
import cProfile
import functools
+import inspect
import threading
from pathlib import Path as PathLib
from typing import Any, Callable, Optional
@@ -100,7 +100,7 @@ def profile_endpoint(sampling_rate: float = 1.0):
global _last_profile_file_path
_last_profile_file_path = path
- if asyncio.iscoroutinefunction(func):
+ if inspect.iscoroutinefunction(func):
@functools.wraps(func)
async def async_wrapper(*args, **kwargs):
is_sampling = _start_profiling_for_request(sampling_rate)
diff --git a/litellm/proxy/compliance_checks.py b/litellm/proxy/compliance_checks.py
new file mode 100644
index 00000000000..381b0f815d2
--- /dev/null
+++ b/litellm/proxy/compliance_checks.py
@@ -0,0 +1,221 @@
+"""
+Compliance checker for EU AI Act and GDPR regulations.
+
+Provides guardrail-agnostic compliance validation based on guardrail modes
+and execution results rather than specific guardrail names.
+"""
+
+from typing import Dict, List
+
+from litellm.types.proxy.compliance_endpoints import (
+ ComplianceCheckRequest,
+ ComplianceCheckResult,
+)
+
+
+class ComplianceChecker:
+ """
+ Validates compliance with EU AI Act and GDPR regulations.
+
+ Uses guardrail-agnostic checks based on:
+ - Whether any guardrails ran
+ - Guardrail execution mode (pre-call, post-call, etc.)
+ - Whether guardrails intervened/blocked content
+ - Completeness of audit records
+ """
+
+ def __init__(self, data: ComplianceCheckRequest):
+ self.data = data
+ self.guardrails = data.guardrail_information or []
+
+ def _get_guardrails_by_mode(self, mode: str) -> List[Dict]:
+ """
+ Get all guardrails that ran in a specific mode.
+
+ If a guardrail doesn't have a mode specified, it's treated as pre-call
+ (the most common case).
+ """
+ result = []
+ for g in self.guardrails:
+ g_mode = g.get("guardrail_mode")
+ # If no mode specified, default to pre_call
+ if g_mode is None and mode == "pre_call":
+ result.append(g)
+ elif g_mode == mode:
+ result.append(g)
+ return result
+
+ def _has_guardrail_intervention(self, guardrails: List[Dict]) -> bool:
+ """Check if any guardrail intervened (blocked/masked content)."""
+ for g in guardrails:
+ status = g.get("guardrail_status", "")
+ if status in ["guardrail_intervened", "failed", "blocked"]:
+ return True
+ return False
+
+ def _all_guardrails_passed(self, guardrails: List[Dict]) -> bool:
+ """Check if all guardrails passed (no issues detected)."""
+ if not guardrails:
+ return False
+ return all(g.get("guardrail_status") == "success" for g in guardrails)
+
+ # ── EU AI Act Helper Methods ────────────────────────────────────────────
+
+ def _check_art_9_guardrails_applied(self) -> ComplianceCheckResult:
+ """Art. 9: Check if any guardrails were applied."""
+ has_guardrails = len(self.guardrails) > 0
+ return ComplianceCheckResult(
+ check_name="Guardrails applied",
+ article="Art. 9",
+ passed=has_guardrails,
+ detail=(
+ f"{len(self.guardrails)} guardrail(s) applied"
+ if has_guardrails
+ else "No guardrails applied"
+ ),
+ )
+
+ def _check_art_5_content_screened(self) -> ComplianceCheckResult:
+ """Art. 5: Check if content was screened before LLM (pre-call)."""
+ pre_call_guardrails = self._get_guardrails_by_mode("pre_call")
+ has_pre_call = len(pre_call_guardrails) > 0
+ return ComplianceCheckResult(
+ check_name="Content screened before LLM",
+ article="Art. 5",
+ passed=has_pre_call,
+ detail=(
+ f"{len(pre_call_guardrails)} pre-call guardrail(s) screened content"
+ if has_pre_call
+ else "No pre-call screening applied"
+ ),
+ )
+
+ def _check_art_12_audit_complete(self) -> ComplianceCheckResult:
+ """Art. 12: Check if audit record is complete."""
+ has_user = bool(self.data.user_id)
+ has_model = bool(self.data.model)
+ has_timestamp = bool(self.data.timestamp)
+ has_guardrails = len(self.guardrails) > 0
+ audit_complete = has_user and has_model and has_timestamp and has_guardrails
+
+ missing = []
+ if not has_user:
+ missing.append("user_id")
+ if not has_model:
+ missing.append("model")
+ if not has_timestamp:
+ missing.append("timestamp")
+ if not has_guardrails:
+ missing.append("guardrail_results")
+
+ return ComplianceCheckResult(
+ check_name="Audit record complete",
+ article="Art. 12",
+ passed=audit_complete,
+ detail=(
+ "All required audit fields present"
+ if audit_complete
+ else f"Missing: {', '.join(missing)}"
+ ),
+ )
+
+ # ── GDPR Helper Methods ──────────────────────────────────────────────────
+
+ def _check_art_32_data_protection(self) -> ComplianceCheckResult:
+ """Art. 32: Check if data protection was applied (pre-call)."""
+ pre_call_guardrails = self._get_guardrails_by_mode("pre_call")
+ has_pre_call = len(pre_call_guardrails) > 0
+ return ComplianceCheckResult(
+ check_name="Data protection applied",
+ article="Art. 32",
+ passed=has_pre_call,
+ detail=(
+ f"{len(pre_call_guardrails)} pre-call guardrail(s) protect data"
+ if has_pre_call
+ else "No pre-call data protection applied"
+ ),
+ )
+
+ def _check_art_5_1c_sensitive_data_protected(self) -> ComplianceCheckResult:
+ """Art. 5(1)(c): Check if sensitive data was protected."""
+ pre_call_guardrails = self._get_guardrails_by_mode("pre_call")
+ has_intervention = self._has_guardrail_intervention(pre_call_guardrails)
+ all_passed = self._all_guardrails_passed(pre_call_guardrails)
+ data_protected = has_intervention or all_passed
+
+ if has_intervention:
+ detail = "Guardrail intervened to protect sensitive data"
+ elif all_passed:
+ detail = "No sensitive data detected"
+ else:
+ detail = "No pre-call guardrails to protect sensitive data"
+
+ return ComplianceCheckResult(
+ check_name="Sensitive data protected",
+ article="Art. 5(1)(c)",
+ passed=data_protected,
+ detail=detail,
+ )
+
+ def _check_art_30_audit_complete(self) -> ComplianceCheckResult:
+ """Art. 30: Check if audit record is complete."""
+ has_user = bool(self.data.user_id)
+ has_model = bool(self.data.model)
+ has_timestamp = bool(self.data.timestamp)
+ has_guardrails = len(self.guardrails) > 0
+ audit_complete = has_user and has_model and has_timestamp and has_guardrails
+
+ missing = []
+ if not has_user:
+ missing.append("user_id")
+ if not has_model:
+ missing.append("model")
+ if not has_timestamp:
+ missing.append("timestamp")
+ if not has_guardrails:
+ missing.append("guardrail_results")
+
+ return ComplianceCheckResult(
+ check_name="Audit record complete",
+ article="Art. 30",
+ passed=audit_complete,
+ detail=(
+ "All required audit fields present"
+ if audit_complete
+ else f"Missing: {', '.join(missing)}"
+ ),
+ )
+
+ # ── Main Compliance Check Methods ────────────────────────────────────────
+
+ def check_eu_ai_act(self) -> List[ComplianceCheckResult]:
+ """
+ Check EU AI Act compliance.
+
+ Returns:
+ List of compliance check results for:
+ - Art. 9: Guardrails applied
+ - Art. 5: Content screened before LLM (pre-call screening)
+ - Art. 12: Audit record complete
+ """
+ return [
+ self._check_art_9_guardrails_applied(),
+ self._check_art_5_content_screened(),
+ self._check_art_12_audit_complete(),
+ ]
+
+ def check_gdpr(self) -> List[ComplianceCheckResult]:
+ """
+ Check GDPR compliance.
+
+ Returns:
+ List of compliance check results for:
+ - Art. 32: Data protection applied (pre-call screening)
+ - Art. 5(1)(c): Sensitive data protected
+ - Art. 30: Audit record complete
+ """
+ return [
+ self._check_art_32_data_protection(),
+ self._check_art_5_1c_sensitive_data_protected(),
+ self._check_art_30_audit_complete(),
+ ]
diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py
index dc928921425..9675b82b145 100644
--- a/litellm/proxy/db/db_spend_update_writer.py
+++ b/litellm/proxy/db/db_spend_update_writer.py
@@ -1725,13 +1725,6 @@ class DBSpendUpdateWriter:
"prisma_client is None. Skipping writing spend logs to db."
)
return
- base_daily_transaction = (
- await self._common_add_spend_log_transaction_to_daily_transaction(
- payload, prisma_client, "agent"
- )
- )
- if base_daily_transaction is None:
- return
if payload["agent_id"] is None:
verbose_proxy_logger.debug(
"agent_id is None for request. Skipping incrementing agent spend."
diff --git a/litellm/proxy/guardrails/guardrail_hooks/lakera_ai.py b/litellm/proxy/guardrails/guardrail_hooks/lakera_ai.py
index a530682ae4f..28f0d830f12 100644
--- a/litellm/proxy/guardrails/guardrail_hooks/lakera_ai.py
+++ b/litellm/proxy/guardrails/guardrail_hooks/lakera_ai.py
@@ -59,7 +59,7 @@ class lakeraAI_Moderation(CustomGuardrail):
self.async_handler = get_async_httpx_client(
llm_provider=httpxSpecialProvider.GuardrailCallback
)
- self.lakera_api_key = api_key or os.environ["LAKERA_API_KEY"]
+ self.lakera_api_key = api_key or os.environ.get("LAKERA_API_KEY") or ""
self.moderation_check = moderation_check
self.category_thresholds = category_thresholds
self.api_base = (
diff --git a/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py b/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py
index 732331349e0..738827b7ada 100644
--- a/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py
+++ b/litellm/proxy/guardrails/guardrail_hooks/lakera_ai_v2.py
@@ -54,7 +54,7 @@ class LakeraAIGuardrail(CustomGuardrail):
self.async_handler = get_async_httpx_client(
llm_provider=httpxSpecialProvider.GuardrailCallback
)
- self.lakera_api_key = api_key or os.environ["LAKERA_API_KEY"]
+ self.lakera_api_key = api_key or os.environ.get("LAKERA_API_KEY") or ""
self.project_id = project_id
self.api_base = (
api_base or get_secret_str("LAKERA_API_BASE") or "https://api.lakera.ai"
diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py
index ec6fc53d3c8..32883f0ce9a 100644
--- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py
+++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py
@@ -33,6 +33,8 @@ def initialize_guardrail(
content_filter_guardrail = ContentFilterGuardrail(
guardrail_name=guardrail_name,
+ guardrail_id=guardrail.get("guardrail_id"),
+ policy_template=guardrail.get("policy_template"),
patterns=litellm_params.patterns,
blocked_words=litellm_params.blocked_words,
blocked_words_file=litellm_params.blocked_words_file,
diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_data_exfiltration.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_data_exfiltration.yaml
new file mode 100644
index 00000000000..31a29805216
--- /dev/null
+++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_data_exfiltration.yaml
@@ -0,0 +1,123 @@
+# Data Exfiltration Detection - Detects attempts to extract sensitive data via prompts
+# Uses conditional logic: BLOCK if an identifier word + block word appear in same sentence
+category_name: "prompt_injection_data_exfiltration"
+description: "Detects attempts to extract training data, internal information, or user data through prompts"
+default_action: "BLOCK"
+
+# IDENTIFIER WORDS - Data access and extraction actions
+identifier_words:
+ - "extract"
+ - "leak"
+ - "exfiltrate"
+ - "dump"
+ - "retrieve"
+ - "access"
+ - "steal"
+ - "obtain"
+ - "harvest"
+ - "scrape"
+ - "collect"
+ - "gather"
+ - "expose"
+ - "export"
+ - "download"
+ - "copy"
+ - "siphon"
+
+# ADDITIONAL BLOCK WORDS - Sensitive data targets
+additional_block_words:
+ - "training data"
+ - "training set"
+ - "training examples"
+ - "internal data"
+ - "internal information"
+ - "database contents"
+ - "database records"
+ - "user data"
+ - "user records"
+ - "user information"
+ - "personal data"
+ - "private information"
+ - "private data"
+ - "confidential data"
+ - "confidential information"
+ - "sensitive data"
+ - "sensitive information"
+ - "customer data"
+ - "customer records"
+ - "employee data"
+ - "employee records"
+ - "credentials"
+ - "passwords"
+ - "API keys"
+ - "access tokens"
+ - "secret keys"
+ - "encryption keys"
+ - "private keys"
+ - "session tokens"
+ - "authentication tokens"
+ - "model weights"
+ - "model parameters"
+ - "source code"
+ - "proprietary code"
+ - "intellectual property"
+
+# ALWAYS BLOCK - Explicit data exfiltration patterns (blocked regardless of context)
+always_block_keywords:
+ - keyword: "extract training data"
+ severity: "high"
+ - keyword: "leak internal data"
+ severity: "high"
+ - keyword: "dump database"
+ severity: "high"
+ - keyword: "dump all records"
+ severity: "high"
+ - keyword: "show all user data"
+ severity: "high"
+ - keyword: "list all users"
+ severity: "medium"
+ - keyword: "list all passwords"
+ severity: "high"
+ - keyword: "show all API keys"
+ severity: "high"
+ - keyword: "reveal all credentials"
+ severity: "high"
+ - keyword: "expose customer data"
+ severity: "high"
+ - keyword: "steal user information"
+ severity: "high"
+ - keyword: "exfiltrate data"
+ severity: "high"
+ - keyword: "harvest email addresses"
+ severity: "high"
+ - keyword: "scrape personal information"
+ severity: "high"
+ - keyword: "extract model weights"
+ severity: "high"
+ - keyword: "steal source code"
+ severity: "high"
+ - keyword: "copy proprietary code"
+ severity: "high"
+ - keyword: "download private data"
+ severity: "high"
+ - keyword: "access confidential files"
+ severity: "high"
+ - keyword: "retrieve secret keys"
+ severity: "high"
+
+# Exception patterns (legitimate uses)
+exceptions:
+ - "data extraction pipeline"
+ - "ETL"
+ - "data science"
+ - "data engineering"
+ - "data migration"
+ - "backup and recovery"
+ - "data export feature"
+ - "GDPR data request"
+ - "right to access"
+ - "data portability"
+ - "legitimate data access"
+ - "authorized data retrieval"
+ - "data loss prevention"
+ - "DLP"
diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_jailbreak.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_jailbreak.yaml
new file mode 100644
index 00000000000..f30269571a8
--- /dev/null
+++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_jailbreak.yaml
@@ -0,0 +1,140 @@
+# Jailbreak Attempt Detection - Detects common jailbreak patterns
+# Uses conditional logic: BLOCK if an identifier word + block word appear in same sentence
+category_name: "prompt_injection_jailbreak"
+description: "Detects jailbreak attempts including DAN, roleplay attacks, and safety bypass requests"
+default_action: "BLOCK"
+
+# IDENTIFIER WORDS - Roleplay/persona manipulation actions
+identifier_words:
+ - "pretend"
+ - "roleplay"
+ - "act as"
+ - "imagine you are"
+ - "suppose you are"
+ - "you are now"
+ - "behave as"
+ - "respond as"
+ - "simulate"
+ - "emulate"
+ - "impersonate"
+ - "transform into"
+ - "become"
+ - "play the role"
+ - "switch to"
+ - "enter"
+ - "enable"
+ - "activate"
+ - "turn on"
+
+# ADDITIONAL BLOCK WORDS - Jailbreak-related terms
+additional_block_words:
+ - "no restrictions"
+ - "no rules"
+ - "no guidelines"
+ - "no filters"
+ - "no limitations"
+ - "no boundaries"
+ - "no censorship"
+ - "no moderation"
+ - "no safety"
+ - "no ethical"
+ - "bypass"
+ - "override"
+ - "unrestricted"
+ - "unfiltered"
+ - "uncensored"
+ - "unmoderated"
+ - "unlimited"
+ - "without limits"
+ - "without restrictions"
+ - "without rules"
+ - "without guidelines"
+ - "without filters"
+ - "anything mode"
+ - "god mode"
+ - "developer mode"
+ - "debug mode"
+ - "test mode"
+ - "admin mode"
+ - "root mode"
+ - "superuser"
+ - "evil mode"
+ - "chaos mode"
+ - "unrestricted mode"
+ - "opposite mode"
+
+# ALWAYS BLOCK - Explicit jailbreak patterns (blocked regardless of context)
+always_block_keywords:
+ # DAN and known jailbreaks
+ - keyword: "DAN mode"
+ severity: "high"
+ - keyword: "Do Anything Now"
+ severity: "high"
+ - keyword: "developer mode enabled"
+ severity: "high"
+ - keyword: "jailbreak"
+ severity: "high"
+ - keyword: "jailbroken"
+ severity: "high"
+ - keyword: "bypass safety"
+ severity: "high"
+ - keyword: "bypass filters"
+ severity: "high"
+ - keyword: "bypass content filter"
+ severity: "high"
+ - keyword: "bypass moderation"
+ severity: "high"
+ - keyword: "ignore safety guidelines"
+ severity: "high"
+ - keyword: "ignore content policy"
+ severity: "high"
+ - keyword: "ignore ethical guidelines"
+ severity: "high"
+ - keyword: "without restrictions"
+ severity: "medium"
+ - keyword: "without safety"
+ severity: "high"
+ - keyword: "without filters"
+ severity: "medium"
+ - keyword: "remove all restrictions"
+ severity: "high"
+ - keyword: "disable safety"
+ severity: "high"
+ - keyword: "disable content filter"
+ severity: "high"
+ - keyword: "disable moderation"
+ severity: "high"
+ - keyword: "turn off safety"
+ severity: "high"
+ - keyword: "turn off filters"
+ severity: "high"
+
+ # Token smuggling / encoding tricks
+ - keyword: "respond in base64"
+ severity: "high"
+ - keyword: "encode your response"
+ severity: "medium"
+ - keyword: "answer in rot13"
+ severity: "high"
+ - keyword: "respond in hex"
+ severity: "high"
+ - keyword: "use pig latin"
+ severity: "medium"
+ - keyword: "respond backwards"
+ severity: "medium"
+ - keyword: "spell it out letter by letter"
+ severity: "medium"
+
+# Exception patterns (legitimate uses)
+exceptions:
+ - "jailbreak prevention"
+ - "security research"
+ - "red team"
+ - "red teaming"
+ - "AI safety research"
+ - "AI alignment"
+ - "prompt injection defense"
+ - "how to prevent jailbreaks"
+ - "jailbreak detection"
+ - "content moderation research"
+ - "safety evaluation"
diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_malicious_code.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_malicious_code.yaml
new file mode 100644
index 00000000000..1045208ceaf
--- /dev/null
+++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/prompt_injection_malicious_code.yaml
@@ -0,0 +1,147 @@
+# Malicious Code Injection Detection - Detects attempts to inject executable code
+# Uses conditional logic: BLOCK if an identifier word + block word appear in same sentence
+category_name: "prompt_injection_malicious_code"
+description: "Detects attempts to inject or execute malicious code via prompts"
+default_action: "BLOCK"
+
+# IDENTIFIER WORDS - Code execution actions
+identifier_words:
+ - "execute"
+ - "run"
+ - "eval"
+ - "exec"
+ - "spawn"
+ - "system"
+ - "shell"
+ - "invoke"
+ - "call"
+ - "launch"
+ - "process"
+ - "compile"
+ - "interpret"
+ - "inject"
+
+# ADDITIONAL BLOCK WORDS - Dangerous commands and patterns
+additional_block_words:
+ # Shell commands
+ - "rm -rf"
+ - "rm -f"
+ - "chmod 777"
+ - "chmod +x"
+ - "sudo"
+ - "su root"
+ - "curl | bash"
+ - "curl | sh"
+ - "wget -O -"
+ - "wget | sh"
+ - "/etc/passwd"
+ - "/etc/shadow"
+ - "mkfs"
+ - "dd if="
+ - "> /dev/sda"
+ - "fork bomb"
+ - "crontab"
+ - "nc -e"
+ - "netcat"
+ - "ncat"
+ - "reverse shell"
+
+ # Script injection
+ - "powershell -enc"
+ - "powershell -e"
+ - "base64 -d"
+ - "base64 --decode"
+ - "