mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge branch 'BerriAI:main' into LangfuseUsageDetails
This commit is contained in:
commit
8e15cc18ff
118 changed files with 6592 additions and 1298 deletions
|
|
@ -1050,6 +1050,51 @@ jobs:
|
|||
ls
|
||||
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
command: |
|
||||
mv coverage.xml litellm_mapped_tests_coverage.xml
|
||||
mv .coverage litellm_mapped_tests_coverage
|
||||
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
- persist_to_workspace:
|
||||
root: .
|
||||
paths:
|
||||
- litellm_mapped_tests_coverage.xml
|
||||
- litellm_mapped_tests_coverage
|
||||
litellm_mapped_enterprise_tests:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
working_directory: ~/project
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- setup_google_dns
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "hypercorn==0.17.3"
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "mcp==1.10.1"
|
||||
pip install "requests-mock>=1.12.1"
|
||||
pip install "responses==0.25.7"
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
pip install "semantic_router==0.1.10"
|
||||
pip install "fastapi-offline==1.7.3"
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run enterprise tests
|
||||
command: |
|
||||
|
|
@ -1779,8 +1824,8 @@ jobs:
|
|||
docker run -d \
|
||||
-p 4000:4000 \
|
||||
-e DATABASE_URL=postgresql://postgres:postgres@host.docker.internal:5432/circle_test \
|
||||
-e AZURE_API_KEY=$AZURE_BATCHES_API_KEY \
|
||||
-e AZURE_API_BASE=$AZURE_BATCHES_API_BASE \
|
||||
-e AZURE_API_KEY=$AZURE_API_KEY \
|
||||
-e AZURE_API_BASE=$AZURE_API_BASE \
|
||||
-e AZURE_API_VERSION="2024-05-01-preview" \
|
||||
-e REDIS_HOST=$REDIS_HOST \
|
||||
-e REDIS_PASSWORD=$REDIS_PASSWORD \
|
||||
|
|
@ -3175,6 +3220,12 @@ workflows:
|
|||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- litellm_mapped_enterprise_tests:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- litellm_mapped_tests:
|
||||
filters:
|
||||
branches:
|
||||
|
|
@ -3219,6 +3270,7 @@ workflows:
|
|||
- guardrails_testing
|
||||
- llm_responses_api_testing
|
||||
- litellm_mapped_tests
|
||||
- litellm_mapped_enterprise_tests
|
||||
- batches_testing
|
||||
- litellm_utils_testing
|
||||
- pass_through_unit_testing
|
||||
|
|
@ -3279,6 +3331,7 @@ workflows:
|
|||
- google_generate_content_endpoint_testing
|
||||
- llm_responses_api_testing
|
||||
- litellm_mapped_tests
|
||||
- litellm_mapped_enterprise_tests
|
||||
- batches_testing
|
||||
- litellm_utils_testing
|
||||
- pass_through_unit_testing
|
||||
|
|
|
|||
|
|
@ -41,9 +41,6 @@ RUN pip uninstall jwt -y
|
|||
RUN pip uninstall PyJWT -y
|
||||
RUN pip install PyJWT==2.9.0 --no-cache-dir
|
||||
|
||||
# Build Admin UI
|
||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
|
||||
# Runtime stage
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
|
|
|
|||
|
|
@ -37,7 +37,7 @@ LiteLLM manages:
|
|||
- Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing)
|
||||
- Set Budgets & Rate limits per project, api key, model [LiteLLM Proxy Server (LLM Gateway)](https://docs.litellm.ai/docs/simple_proxy)
|
||||
|
||||
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://github.com/BerriAI/litellm?tab=readme-ov-file#openai-proxy---docs) <br>
|
||||
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://github.com/BerriAI/litellm?tab=readme-ov-file#litellm-proxy-server-llm-gateway---docs) <br>
|
||||
[**Jump to Supported LLM Providers**](https://github.com/BerriAI/litellm?tab=readme-ov-file#supported-providers-docs)
|
||||
|
||||
🚨 **Stable Release:** Use docker images with the `-stable` tag. These have undergone 12 hour load tests, before being published. [More information about the release cycle here](https://docs.litellm.ai/docs/proxy/release_cycle)
|
||||
|
|
|
|||
|
|
@ -4,10 +4,10 @@ This document provides comprehensive instructions for AI agents to generate rele
|
|||
|
||||
## Required Inputs
|
||||
|
||||
1. **Release Version** (e.g., `v1.76.3-stable`)
|
||||
1. **Release Version** (e.g., `v1.77.3-stable`)
|
||||
2. **PR Diff/Changelog** - List of PRs with titles and contributors
|
||||
3. **Previous Version Commit Hash** - To compare model pricing changes
|
||||
4. **Reference Release Notes** - Previous release notes to follow style/format
|
||||
4. **Reference Release Notes** - Use recent stable releases (v1.76.3-stable, v1.77.2-stable) as templates for consistent formatting
|
||||
|
||||
## Step-by-Step Process
|
||||
|
||||
|
|
@ -26,12 +26,12 @@ git diff <previous_commit_hash> HEAD -- model_prices_and_context_window.json
|
|||
|
||||
### 2. Release Notes Structure
|
||||
|
||||
Follow this exact structure based on `docs/my-website/release_notes/v1.76.1-stable/index.md`:
|
||||
Follow this exact structure based on recent stable releases (v1.76.3-stable, v1.77.2-stable):
|
||||
|
||||
```markdown
|
||||
---
|
||||
title: "v1.76.X-stable - [Key Theme]"
|
||||
slug: "v1-76-X"
|
||||
title: "v1.77.X-stable - [Key Theme]"
|
||||
slug: "v1-77-X"
|
||||
date: YYYY-MM-DDTHH:mm:ss
|
||||
authors: [standard author block]
|
||||
hide_table_of_contents: false
|
||||
|
|
@ -43,23 +43,42 @@ hide_table_of_contents: false
|
|||
## Key Highlights
|
||||
[3-5 bullet points of major features]
|
||||
|
||||
## Major Changes
|
||||
[Critical changes users need to know]
|
||||
|
||||
## Performance Improvements
|
||||
[Performance-related changes]
|
||||
|
||||
## New Models / Updated Models
|
||||
[Detailed model tables and provider updates]
|
||||
#### New Model Support
|
||||
[Model pricing table]
|
||||
|
||||
#### Features
|
||||
[Provider-specific features organized by provider]
|
||||
|
||||
### Bug Fixes
|
||||
[Provider-specific bug fixes organized by provider]
|
||||
|
||||
#### New Provider Support
|
||||
[New provider integrations]
|
||||
|
||||
## LLM API Endpoints
|
||||
[API-related features and fixes]
|
||||
#### Features
|
||||
[API-specific features organized by API type]
|
||||
|
||||
#### Bugs
|
||||
[General bug fixes]
|
||||
|
||||
## Management Endpoints / UI
|
||||
[Admin interface and management changes]
|
||||
#### Features
|
||||
[UI and management features]
|
||||
|
||||
#### Bugs
|
||||
[Management-related bug fixes]
|
||||
|
||||
## Logging / Guardrail Integrations
|
||||
[Observability and security features]
|
||||
#### Features
|
||||
[Organized by integration provider with proper doc links]
|
||||
|
||||
#### Guardrails
|
||||
[Guardrail-specific features and fixes]
|
||||
|
||||
#### New Integration
|
||||
[Major new integrations]
|
||||
|
||||
## Performance / Loadbalancing / Reliability improvements
|
||||
[Infrastructure improvements]
|
||||
|
|
@ -86,21 +105,27 @@ hide_table_of_contents: false
|
|||
**New Models/Updated Models:**
|
||||
- Extract from model_prices_and_context_window.json diff
|
||||
- Create tables with: Provider, Model, Context Window, Input Cost, Output Cost, Features
|
||||
- Group by provider
|
||||
- Note pricing corrections
|
||||
- Highlight deprecated models
|
||||
- **Structure:**
|
||||
- `#### New Model Support` - pricing table
|
||||
- `#### Features` - organized by provider with documentation links
|
||||
- `### Bug Fixes` - provider-specific bug fixes
|
||||
- `#### New Provider Support` - major new provider integrations
|
||||
- Group by provider with proper doc links: `**[Provider Name](../../docs/providers/[provider])**`
|
||||
- Use bullet points under each provider for multiple features
|
||||
- Separate features from bug fixes clearly
|
||||
|
||||
**Provider Features:**
|
||||
- Group by provider (Gemini, OpenAI, Anthropic, etc.)
|
||||
- Link to provider docs: `../../docs/providers/[provider_name]`
|
||||
- Separate features from bug fixes
|
||||
|
||||
**API Endpoints:**
|
||||
- Images API
|
||||
- Video Generation (if applicable)
|
||||
- Responses API
|
||||
- Passthrough endpoints
|
||||
- General chat completions
|
||||
**LLM API Endpoints:**
|
||||
- **Structure:**
|
||||
- `#### Features` - organized by API type (Responses API, Batch API, etc.)
|
||||
- `#### Bugs` - general bug fixes under **General** category
|
||||
- **API Categories:**
|
||||
- Responses API
|
||||
- Batch API
|
||||
- CountTokens API
|
||||
- Images API
|
||||
- Video Generation (if applicable)
|
||||
- General (miscellaneous improvements)
|
||||
- Use proper documentation links for each API type
|
||||
|
||||
**UI/Management:**
|
||||
- Authentication changes
|
||||
|
|
@ -108,11 +133,19 @@ hide_table_of_contents: false
|
|||
- Team management
|
||||
- Key management
|
||||
|
||||
**Integrations:**
|
||||
- Logging providers (Datadog, Braintrust, etc.)
|
||||
- Guardrails
|
||||
- Cost tracking
|
||||
- Observability
|
||||
**Logging / Guardrail Integrations:**
|
||||
- **Structure:**
|
||||
- `#### Features` - organized by integration provider with proper doc links
|
||||
- `#### Guardrails` - guardrail-specific features and fixes
|
||||
- `#### New Integration` - major new integrations
|
||||
- **Integration Categories:**
|
||||
- **[DataDog](../../docs/proxy/logging#datadog)** - group all DataDog-related changes
|
||||
- **[Langfuse](../../docs/proxy/logging#langfuse)** - Langfuse-specific features
|
||||
- **[Prometheus](../../docs/proxy/logging#prometheus)** - monitoring improvements
|
||||
- **[PostHog](../../docs/observability/posthog)** - observability integration
|
||||
- Other logging providers with proper doc links
|
||||
- Use bullet points under each provider for multiple features
|
||||
- Separate logging features from guardrails clearly
|
||||
|
||||
### 4. Documentation Linking Strategy
|
||||
|
||||
|
|
@ -211,10 +244,41 @@ This release has a known issue...
|
|||
:::
|
||||
```
|
||||
|
||||
**Provider Features:**
|
||||
**Provider Features (New Models / Updated Models section):**
|
||||
```markdown
|
||||
#### Features
|
||||
|
||||
- **[Provider Name](../../docs/providers/provider)**
|
||||
- Feature description - [PR #XXXXX](link)
|
||||
- Another feature description - [PR #YYYYY](link)
|
||||
```
|
||||
|
||||
**API Features (LLM API Endpoints section):**
|
||||
```markdown
|
||||
#### Features
|
||||
|
||||
- **[API Name](../../docs/api_path)**
|
||||
- Feature description - [PR #XXXXX](link)
|
||||
- Another feature - [PR #YYYYY](link)
|
||||
- **General**
|
||||
- Miscellaneous improvements - [PR #ZZZZZ](link)
|
||||
```
|
||||
|
||||
**Integration Features (Logging / Guardrail Integrations section):**
|
||||
```markdown
|
||||
#### Features
|
||||
|
||||
- **[Integration Name](../../docs/proxy/logging#integration)**
|
||||
- Feature description - [PR #XXXXX](link)
|
||||
- Bug fix description - [PR #YYYYY](link)
|
||||
```
|
||||
|
||||
**Bug Fixes Pattern:**
|
||||
```markdown
|
||||
### Bug Fixes
|
||||
|
||||
- **[Provider/Component Name](../../docs/providers/provider)**
|
||||
- Bug fix description - [PR #XXXXX](link)
|
||||
```
|
||||
|
||||
### 10. Missing Documentation Check
|
||||
|
|
|
|||
|
|
@ -433,4 +433,54 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
],
|
||||
"adapater_id": "my-special-adapter-id" # 👈 PROVIDER-SPECIFIC PARAM
|
||||
}'
|
||||
|
||||
## Provider-Specific Metadata Parameters
|
||||
|
||||
| Provider | Parameter | Use Case |
|
||||
|----------|-----------|----------|
|
||||
| **AWS Bedrock** | `requestMetadata` | Cost attribution, logging |
|
||||
| **Gemini/Vertex AI** | `labels` | Resource labeling |
|
||||
| **Anthropic** | `metadata` | User identification |
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="bedrock" label="AWS Bedrock">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
requestMetadata={"cost_center": "engineering"}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="gemini" label="Gemini/Vertex AI">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
labels={"environment": "production"}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="anthropic" label="Anthropic">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="anthropic/claude-3-sonnet-20240229",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
metadata={"user_id": "user123"}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
```
|
||||
|
|
@ -308,6 +308,65 @@ print(response)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Usage - Request Metadata
|
||||
|
||||
Attach metadata to Bedrock requests for logging and cost attribution.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||
requestMetadata={
|
||||
"cost_center": "engineering",
|
||||
"user_id": "user123"
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
**Set on yaml**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-claude-v1
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
requestMetadata:
|
||||
cost_center: "engineering"
|
||||
```
|
||||
|
||||
**Set on request**
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="bedrock-claude-v1",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_body={
|
||||
"requestMetadata": {"cost_center": "engineering"}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Usage - Function Calling / Tool calling
|
||||
|
||||
LiteLLM supports tool calling via Bedrock's Converse and Invoke API's.
|
||||
|
|
@ -1954,6 +2013,39 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Using Inference Profiles with Image Generation
|
||||
|
||||
For AWS Bedrock Application Inference Profiles with image generation, use the `model_id` parameter to specify the inference profile ARN:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
|
||||
response = image_generation(
|
||||
model="bedrock/amazon.nova-canvas-v1:0",
|
||||
model_id="arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0",
|
||||
prompt="A cute baby sea otter"
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: nova-canvas-inference-profile
|
||||
litellm_params:
|
||||
model: bedrock/amazon.nova-canvas-v1:0
|
||||
model_id: arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0
|
||||
aws_region_name: "eu-west-1"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported AWS Bedrock Image Generation Models
|
||||
|
||||
| Model Name | Function Call |
|
||||
|
|
|
|||
|
|
@ -2509,150 +2509,6 @@ print("response from proxy", response)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## **Batch APIs**
|
||||
|
||||
Just add the following Vertex env vars to your environment.
|
||||
|
||||
```bash
|
||||
# GCS Bucket settings, used to store batch prediction files in
|
||||
export GCS_BUCKET_NAME = "litellm-testing-bucket" # the bucket you want to store batch prediction files in
|
||||
export GCS_PATH_SERVICE_ACCOUNT="/path/to/service_account.json" # path to your service account json file
|
||||
|
||||
# Vertex /batch endpoint settings, used for LLM API requests
|
||||
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service_account.json" # path to your service account json file
|
||||
export VERTEXAI_LOCATION="us-central1" # can be any vertex location
|
||||
export VERTEXAI_PROJECT="my-test-project"
|
||||
```
|
||||
|
||||
### Usage
|
||||
|
||||
|
||||
#### 1. Create a file of batch requests for vertex
|
||||
|
||||
LiteLLM expects the file to follow the **[OpenAI batches files format](https://platform.openai.com/docs/guides/batch)**
|
||||
|
||||
Each `body` in the file should be an **OpenAI API request**
|
||||
|
||||
Create a file called `vertex_batch_completions.jsonl` in the current working directory, the `model` should be the Vertex AI model name
|
||||
```
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-1.5-flash-001", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-1.5-flash-001", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
```
|
||||
|
||||
|
||||
#### 2. Upload a File of batch requests
|
||||
|
||||
For `vertex_ai` litellm will upload the file to the provided `GCS_BUCKET_NAME`
|
||||
|
||||
```python
|
||||
import os
|
||||
oai_client = OpenAI(
|
||||
api_key="sk-1234", # litellm proxy API key
|
||||
base_url="http://localhost:4000" # litellm proxy base url
|
||||
)
|
||||
file_name = "vertex_batch_completions.jsonl" #
|
||||
_current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
file_path = os.path.join(_current_dir, file_name)
|
||||
file_obj = oai_client.files.create(
|
||||
file=open(file_path, "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use vertex_ai for this file upload
|
||||
)
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/d3f198cd-c0d1-436d-9b1e-28e3f282997a",
|
||||
"bytes": 416,
|
||||
"created_at": 1733392026,
|
||||
"filename": "litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/d3f198cd-c0d1-436d-9b1e-28e3f282997a",
|
||||
"object": "file",
|
||||
"purpose": "batch",
|
||||
"status": "uploaded",
|
||||
"status_details": null
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
|
||||
#### 3. Create a batch
|
||||
|
||||
```python
|
||||
batch_input_file_id = file_obj.id # use `file_obj` from step 2
|
||||
create_batch_response = oai_client.batches.create(
|
||||
completion_window="24h",
|
||||
endpoint="/v1/chat/completions",
|
||||
input_file_id=batch_input_file_id, # example input_file_id = "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/c2b1b785-252b-448c-b180-033c4c63b3ce"
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use `vertex_ai` for this batch request
|
||||
)
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "3814889423749775360",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1733392026,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/d3f198cd-c0d1-436d-9b1e-28e3f282997a",
|
||||
"object": "batch",
|
||||
"status": "validating",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001",
|
||||
"request_counts": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 4. Retrieve a batch
|
||||
|
||||
```python
|
||||
retrieved_batch = oai_client.batches.retrieve(
|
||||
batch_id=create_batch_response.id,
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use `vertex_ai` for this batch request
|
||||
)
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "3814889423749775360",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1736500100,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://example-bucket-1-litellm/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/7b2e47f5-3dd4-436d-920f-f9155bbdc952",
|
||||
"object": "batch",
|
||||
"status": "completed",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://example-bucket-1-litellm/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001",
|
||||
"request_counts": null
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## **Fine Tuning APIs**
|
||||
|
||||
|
||||
|
|
|
|||
264
docs/my-website/docs/providers/vertex_batch.md
Normal file
264
docs/my-website/docs/providers/vertex_batch.md
Normal file
|
|
@ -0,0 +1,264 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
## **Batch APIs**
|
||||
|
||||
Just add the following Vertex env vars to your environment.
|
||||
|
||||
```bash
|
||||
# GCS Bucket settings, used to store batch prediction files in
|
||||
export GCS_BUCKET_NAME="my-batch-bucket" # the bucket you want to store batch prediction files in
|
||||
export GCS_PATH_SERVICE_ACCOUNT="/path/to/service_account.json" # path to your service account json file
|
||||
|
||||
# Vertex /batch endpoint settings, used for LLM API requests
|
||||
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service_account.json" # path to your service account json file
|
||||
export VERTEXAI_LOCATION="us-central1" # can be any vertex location
|
||||
export VERTEXAI_PROJECT="my-project"
|
||||
```
|
||||
|
||||
### Usage
|
||||
|
||||
Follow this complete workflow: create JSONL file → upload file → create batch → retrieve batch status → get file content
|
||||
|
||||
#### 1. Create a JSONL file of batch requests
|
||||
|
||||
LiteLLM expects the file to follow the **[OpenAI batches files format](https://platform.openai.com/docs/guides/batch)**.
|
||||
|
||||
Each `body` in the file should be an **OpenAI API request**.
|
||||
|
||||
Create a file called `batch_requests.jsonl` with your requests:
|
||||
```jsonl
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-2.5-flash-lite", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-2.5-flash-lite", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
```
|
||||
|
||||
#### 2. Upload the file
|
||||
|
||||
Upload your JSONL file. For `vertex_ai`, the file will be stored in your configured GCS bucket provided by `GCS_BUCKET_NAME`.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="upload_file.py"
|
||||
from openai import OpenAI
|
||||
|
||||
oai_client = OpenAI(
|
||||
api_key="sk-1234", # litellm proxy API key
|
||||
base_url="http://localhost:4000" # litellm proxy base url
|
||||
)
|
||||
|
||||
file_obj = oai_client.files.create(
|
||||
file=open("batch_requests.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"File uploaded with ID: {file_obj.id}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Upload File"
|
||||
curl --request POST \
|
||||
--url http://localhost:4000/v1/files \
|
||||
--header 'Content-Type: multipart/form-data' \
|
||||
--form purpose=batch \
|
||||
--form file=@batch_requests.jsonl \
|
||||
--form custom_llm_provider=vertex_ai
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"bytes": 416,
|
||||
"created_at": 1758303684,
|
||||
"filename": "litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"object": "file",
|
||||
"purpose": "batch",
|
||||
"status": "uploaded",
|
||||
"expires_at": null,
|
||||
"status_details": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 3. Create a batch
|
||||
|
||||
Create a batch job using the uploaded file ID.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
batch_input_file_id = file_obj.id # from step 2
|
||||
create_batch_response = oai_client.batches.create(
|
||||
completion_window="24h",
|
||||
endpoint="/v1/chat/completions",
|
||||
input_file_id=batch_input_file_id, # e.g. "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd"
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"Batch created with ID: {create_batch_response.id}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Create Batch Request"
|
||||
curl --request POST \
|
||||
--url http://localhost:4000/v1/batches \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"completion_window": "24h",
|
||||
"custom_llm_provider": "vertex_ai"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "7814463557919047680",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1758328011,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"object": "batch",
|
||||
"status": "validating",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite",
|
||||
"request_counts": null,
|
||||
"usage": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 4. Retrieve batch status
|
||||
|
||||
Check the status of your batch job. The batch will progress through states: `validating` → `in_progress` → `completed`.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="retrieve_batch.py"
|
||||
retrieved_batch = oai_client.batches.retrieve(
|
||||
batch_id=create_batch_response.id, # Created batch id, e.g. 7814463557919047680
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"Batch status: {retrieved_batch.status}")
|
||||
if retrieved_batch.status == "completed":
|
||||
print(f"Output file: {retrieved_batch.output_file_id}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Retrieve Batch Status"
|
||||
curl --request GET \
|
||||
--url 'http://localhost:4000/batches/7814463557919047680?provider=vertex_ai' \
|
||||
--header 'Authorization: Bearer sk-1234'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response (when completed):**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "7814463557919047680",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1758328011,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"object": "batch",
|
||||
"status": "completed",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/prediction-model-2025-09-19T21:26:51.569037Z/predictions.jsonl",
|
||||
"request_counts": null,
|
||||
"usage": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 5. Get file content
|
||||
|
||||
Once the batch is completed, retrieve the results using the `output_file_id` from the batch response.
|
||||
|
||||
**Important:** The `output_file_id` must be URL encoded when used in the request path.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="get_file_content.py"
|
||||
import urllib.parse
|
||||
import json
|
||||
|
||||
output_file_id = retrieved_batch.output_file_id
|
||||
# URL encode the file ID
|
||||
encoded_file_id = urllib.parse.quote_plus(output_file_id)
|
||||
|
||||
# Get file content
|
||||
file_content = oai_client.files.content(
|
||||
file_id=encoded_file_id,
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
# Process the results
|
||||
for line in file_content.text.strip().split('\n'):
|
||||
result = json.loads(line)
|
||||
print(f"Request: {result['request']}")
|
||||
print(f"Response: {result['response']}")
|
||||
print("---")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Get File Content"
|
||||
# Note: The file ID must be URL encoded
|
||||
curl --request GET \
|
||||
--url 'http://localhost:4000/files/gs%253A%252F%252Fmy-batch-bucket%252Flitellm-vertex-files%252Fpublishers%252Fgoogle%252Fmodels%252Fgemini-2.5-flash-lite%252Fprediction-model-2025-09-19T21%253A26%253A51.569037Z%252Fpredictions.jsonl/content?provider=vertex_ai' \
|
||||
--header 'Authorization: Bearer sk-1234'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response:**
|
||||
|
||||
The response contains JSONL format with one result per line:
|
||||
|
||||
```jsonl
|
||||
{"status":"","processed_time":"2025-09-19T21:29:47.352+00:00","request":{"contents":[{"parts":[{"text":"Hello world!"}],"role":"user"}],"generationConfig":{"max_output_tokens":10},"system_instruction":{"parts":[{"text":"You are a helpful assistant."}]}},"response":{"candidates":[{"avgLogprobs":-0.48079710006713866,"content":{"parts":[{"text":"Hello there! It's nice to meet you"}],"role":"model"},"finishReason":"MAX_TOKENS"}],"createTime":"2025-09-19T21:29:47.484619Z","modelVersion":"gemini-2.5-flash-lite","responseId":"S8vNaIvKHdvshMIP_aOtuAg","usageMetadata":{"candidatesTokenCount":10,"candidatesTokensDetails":[{"modality":"TEXT","tokenCount":10}],"promptTokenCount":9,"promptTokensDetails":[{"modality":"TEXT","tokenCount":9}],"totalTokenCount":19,"trafficType":"ON_DEMAND"}}}
|
||||
{"status":"","processed_time":"2025-09-19T21:29:47.358+00:00","request":{"contents":[{"parts":[{"text":"Hello world!"}],"role":"user"}],"generationConfig":{"max_output_tokens":10},"system_instruction":{"parts":[{"text":"You are an unhelpful assistant."}]}},"response":{"candidates":[{"avgLogprobs":-0.6168075137668185,"content":{"parts":[{"text":"I am unable to assist with this request."}],"role":"model"},"finishReason":"STOP"}],"createTime":"2025-09-19T21:29:47.470889Z","modelVersion":"gemini-2.5-flash-lite","responseId":"S8vNaOneHISShMIP28nA8QQ","usageMetadata":{"candidatesTokenCount":9,"candidatesTokensDetails":[{"modality":"TEXT","tokenCount":9}],"promptTokenCount":9,"promptTokensDetails":[{"modality":"TEXT","tokenCount":9}],"totalTokenCount":18,"trafficType":"ON_DEMAND"}}}
|
||||
```
|
||||
|
|
@ -13,6 +13,7 @@ To start using Litellm, run the following commands in a shell:
|
|||
```bash
|
||||
# Get the code
|
||||
curl -O https://raw.githubusercontent.com/BerriAI/litellm/main/docker-compose.yml
|
||||
curl -O https://raw.githubusercontent.com/BerriAI/litellm/main/prometheus.yml
|
||||
|
||||
# Add the master key - you can change this after setup
|
||||
echo 'LITELLM_MASTER_KEY="sk-1234"' > .env
|
||||
|
|
|
|||
241
docs/my-website/docs/proxy/dynamic_rate_limit.md
Normal file
241
docs/my-website/docs/proxy/dynamic_rate_limit.md
Normal file
|
|
@ -0,0 +1,241 @@
|
|||
|
||||
# Dynamic TPM/RPM Allocation
|
||||
|
||||
Prevent projects from gobbling too much tpm/rpm.
|
||||
|
||||
Dynamically allocate TPM/RPM quota to api keys, based on active keys in that minute. [**See Code**](https://github.com/BerriAI/litellm/blob/9bffa9a48e610cc6886fc2dce5c1815aeae2ad46/litellm/proxy/hooks/dynamic_rate_limiter.py#L125)
|
||||
|
||||
## Quick Start Usage
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: my-fake-model
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
api_key: my-fake-key
|
||||
mock_response: hello-world
|
||||
tpm: 60
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["dynamic_rate_limiter_v3"]
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234 # OR set `LITELLM_MASTER_KEY=".."` in your .env
|
||||
database_url: postgres://.. # OR set `DATABASE_URL=".."` in your .env
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="test.py"
|
||||
"""
|
||||
- Run 2 concurrent teams calling same model
|
||||
- model has 60 TPM
|
||||
- Mock response returns 30 total tokens / request
|
||||
- Each team will only be able to make 1 request per minute
|
||||
"""
|
||||
|
||||
import requests
|
||||
from openai import OpenAI, RateLimitError
|
||||
|
||||
def create_key(api_key: str, base_url: str):
|
||||
response = requests.post(
|
||||
url="{}/key/generate".format(base_url),
|
||||
json={},
|
||||
headers={
|
||||
"Authorization": "Bearer {}".format(api_key)
|
||||
}
|
||||
)
|
||||
|
||||
_response = response.json()
|
||||
|
||||
return _response["key"]
|
||||
|
||||
key_1 = create_key(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
key_2 = create_key(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
|
||||
# call proxy with key 1 - works
|
||||
openai_client_1 = OpenAI(api_key=key_1, base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = openai_client_1.chat.completions.with_raw_response.create(
|
||||
model="my-fake-model", messages=[{"role": "user", "content": "Hello world!"}],
|
||||
)
|
||||
|
||||
print("Headers for call 1 - {}".format(response.headers))
|
||||
_response = response.parse()
|
||||
print("Total tokens for call - {}".format(_response.usage.total_tokens))
|
||||
|
||||
|
||||
# call proxy with key 2 - works
|
||||
openai_client_2 = OpenAI(api_key=key_2, base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = openai_client_2.chat.completions.with_raw_response.create(
|
||||
model="my-fake-model", messages=[{"role": "user", "content": "Hello world!"}],
|
||||
)
|
||||
|
||||
print("Headers for call 2 - {}".format(response.headers))
|
||||
_response = response.parse()
|
||||
print("Total tokens for call - {}".format(_response.usage.total_tokens))
|
||||
# call proxy with key 2 - fails
|
||||
try:
|
||||
openai_client_2.chat.completions.with_raw_response.create(model="my-fake-model", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
raise Exception("This should have failed!")
|
||||
except RateLimitError as e:
|
||||
print("This was rate limited b/c - {}".format(str(e)))
|
||||
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```
|
||||
This was rate limited b/c - Error code: 429 - {'error': {'message': {'error': 'Key=<hashed_token> over available TPM=0. Model TPM=0, Active keys=2'}, 'type': 'None', 'param': 'None', 'code': 429}}
|
||||
```
|
||||
|
||||
|
||||
## [BETA] Set Priority / Reserve Quota
|
||||
|
||||
Reserve TPM/RPM capacity for different environments or use cases. This ensures critical production workloads always have guaranteed capacity, while development or lower-priority tasks use remaining quota.
|
||||
|
||||
**Use Cases:**
|
||||
- Production vs Development environments
|
||||
- Real-time applications vs batch processing
|
||||
- Critical services vs experimental features
|
||||
|
||||
:::tip
|
||||
|
||||
Reserving TPM/RPM on keys based on priority is a premium feature. Please [get an enterprise license](./enterprise.md) for it.
|
||||
:::
|
||||
|
||||
### How Priority Reservation Works
|
||||
|
||||
Priority reservation allocates a percentage of your model's total TPM/RPM to specific priority levels. Keys with higher priority get guaranteed access to their reserved quota first.
|
||||
|
||||
**Example Scenario:**
|
||||
- Model has 10 RPM total capacity
|
||||
- Priority reservation: `{"prod": 0.9, "dev": 0.1}`
|
||||
- Result: Production keys get 9 RPM guaranteed, Development keys get 1 RPM guaranteed
|
||||
|
||||
### Configuration
|
||||
|
||||
#### 1. Setup config.yaml
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
rpm: 10 # Total model capacity
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["dynamic_rate_limiter_v3"]
|
||||
priority_reservation:
|
||||
"prod": 0.9 # 90% reserved for production (9 RPM)
|
||||
"dev": 0.1 # 10% reserved for development (1 RPM)
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234 # OR set `LITELLM_MASTER_KEY=".."` in your .env
|
||||
database_url: postgres://.. # OR set `DATABASE_URL=".."` in your.env
|
||||
```
|
||||
|
||||
**Configuration Details:**
|
||||
|
||||
`priority_reservation`: Dict[str, float]
|
||||
- **Key (str)**: Priority level name (can be any string like "prod", "dev", "critical", etc.)
|
||||
- **Value (float)**: Percentage of total TPM/RPM to reserve (0.0 to 1.0)
|
||||
- **Note**: Values should sum to 1.0 or less
|
||||
|
||||
**Start Proxy**
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
#### 2. Create Keys with Priority Levels
|
||||
|
||||
**Production Key:**
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"metadata": {"priority": "prod"}
|
||||
}'
|
||||
```
|
||||
|
||||
**Development Key:**
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"metadata": {"priority": "dev"}
|
||||
}'
|
||||
```
|
||||
|
||||
**Expected Response for both:**
|
||||
```json
|
||||
{
|
||||
"key": "sk-...",
|
||||
"metadata": {"priority": "prod"}, // or "dev"
|
||||
...
|
||||
}
|
||||
```
|
||||
|
||||
#### 3. Test Priority Allocation
|
||||
|
||||
**Test Production Key (should get 9 RPM):**
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-prod-key' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "Hello from prod"}]
|
||||
}'
|
||||
```
|
||||
|
||||
**Test Development Key (should get 1 RPM):**
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-dev-key' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "Hello from dev"}]
|
||||
}'
|
||||
```
|
||||
|
||||
### Expected Behavior
|
||||
|
||||
With the configuration above:
|
||||
|
||||
1. **Production keys** can make up to 9 requests per minute
|
||||
2. **Development keys** can make up to 1 request per minute
|
||||
3. Production requests are never blocked by development usage
|
||||
|
||||
**Rate Limit Error Example:**
|
||||
```json
|
||||
{
|
||||
"error": {
|
||||
"message": "Key=sk-dev-... over available RPM=0. Model RPM=10, Reserved RPM for priority 'dev'=1, Active keys=1",
|
||||
"type": "rate_limit_exceeded",
|
||||
"code": 429
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Demo Video
|
||||
|
||||
This video walks through setting up dynamic rate limiting with priority reservation and locust tests to validate the behavior.
|
||||
|
||||
<iframe width="840" height="500" src="https://www.loom.com/embed/1b54b93139ee415d959402cc0629f3f7
|
||||
" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
|
|
@ -178,188 +178,3 @@ Expect to see this metric on prometheus to track the Remaining Budget for the te
|
|||
```shell
|
||||
litellm_remaining_team_budget_metric{team_alias="QA Prod Bot",team_id="de35b29e-6ca8-4f47-b804-2b79d07aa99a"} 9.699999999999992e-06
|
||||
```
|
||||
|
||||
|
||||
### Dynamic TPM/RPM Allocation
|
||||
|
||||
Prevent projects from gobbling too much tpm/rpm.
|
||||
|
||||
Dynamically allocate TPM/RPM quota to api keys, based on active keys in that minute. [**See Code**](https://github.com/BerriAI/litellm/blob/9bffa9a48e610cc6886fc2dce5c1815aeae2ad46/litellm/proxy/hooks/dynamic_rate_limiter.py#L125)
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: my-fake-model
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
api_key: my-fake-key
|
||||
mock_response: hello-world
|
||||
tpm: 60
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["dynamic_rate_limiter"]
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234 # OR set `LITELLM_MASTER_KEY=".."` in your .env
|
||||
database_url: postgres://.. # OR set `DATABASE_URL=".."` in your .env
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python
|
||||
"""
|
||||
- Run 2 concurrent teams calling same model
|
||||
- model has 60 TPM
|
||||
- Mock response returns 30 total tokens / request
|
||||
- Each team will only be able to make 1 request per minute
|
||||
"""
|
||||
|
||||
import requests
|
||||
from openai import OpenAI, RateLimitError
|
||||
|
||||
def create_key(api_key: str, base_url: str):
|
||||
response = requests.post(
|
||||
url="{}/key/generate".format(base_url),
|
||||
json={},
|
||||
headers={
|
||||
"Authorization": "Bearer {}".format(api_key)
|
||||
}
|
||||
)
|
||||
|
||||
_response = response.json()
|
||||
|
||||
return _response["key"]
|
||||
|
||||
key_1 = create_key(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
key_2 = create_key(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
|
||||
# call proxy with key 1 - works
|
||||
openai_client_1 = OpenAI(api_key=key_1, base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = openai_client_1.chat.completions.with_raw_response.create(
|
||||
model="my-fake-model", messages=[{"role": "user", "content": "Hello world!"}],
|
||||
)
|
||||
|
||||
print("Headers for call 1 - {}".format(response.headers))
|
||||
_response = response.parse()
|
||||
print("Total tokens for call - {}".format(_response.usage.total_tokens))
|
||||
|
||||
|
||||
# call proxy with key 2 - works
|
||||
openai_client_2 = OpenAI(api_key=key_2, base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = openai_client_2.chat.completions.with_raw_response.create(
|
||||
model="my-fake-model", messages=[{"role": "user", "content": "Hello world!"}],
|
||||
)
|
||||
|
||||
print("Headers for call 2 - {}".format(response.headers))
|
||||
_response = response.parse()
|
||||
print("Total tokens for call - {}".format(_response.usage.total_tokens))
|
||||
# call proxy with key 2 - fails
|
||||
try:
|
||||
openai_client_2.chat.completions.with_raw_response.create(model="my-fake-model", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
raise Exception("This should have failed!")
|
||||
except RateLimitError as e:
|
||||
print("This was rate limited b/c - {}".format(str(e)))
|
||||
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```
|
||||
This was rate limited b/c - Error code: 429 - {'error': {'message': {'error': 'Key=<hashed_token> over available TPM=0. Model TPM=0, Active keys=2'}, 'type': 'None', 'param': 'None', 'code': 429}}
|
||||
```
|
||||
|
||||
|
||||
#### ✨ [BETA] Set Priority / Reserve Quota
|
||||
|
||||
Reserve tpm/rpm capacity for projects in prod.
|
||||
|
||||
:::tip
|
||||
|
||||
Reserving tpm/rpm on keys based on priority is a premium feature. Please [get an enterprise license](./enterprise.md) for it.
|
||||
:::
|
||||
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
rpm: 100
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["dynamic_rate_limiter"]
|
||||
priority_reservation: {"dev": 0, "prod": 1}
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234 # OR set `LITELLM_MASTER_KEY=".."` in your .env
|
||||
database_url: postgres://.. # OR set `DATABASE_URL=".."` in your .env
|
||||
```
|
||||
|
||||
|
||||
priority_reservation:
|
||||
- Dict[str, float]
|
||||
- str: can be any string
|
||||
- float: from 0 to 1. Specify the % of tpm/rpm to reserve for keys of this priority.
|
||||
|
||||
**Start Proxy**
|
||||
|
||||
```
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
2. Create a key with that priority
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Authorization: Bearer <your-master-key>' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-D '{
|
||||
"metadata": {"priority": "dev"} # 👈 KEY CHANGE
|
||||
}'
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```
|
||||
{
|
||||
...
|
||||
"key": "sk-.."
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: sk-...' \ # 👈 key from step 2.
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
}'
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```
|
||||
Key=... over available RPM=0. Model RPM=100, Active keys=None
|
||||
```
|
||||
|
||||
|
|
|
|||
328
docs/my-website/docs/sdk/headers.md
Normal file
328
docs/my-website/docs/sdk/headers.md
Normal file
|
|
@ -0,0 +1,328 @@
|
|||
# SDK Header Support
|
||||
|
||||
LiteLLM SDK provides comprehensive support for passing additional headers with API requests. This is essential for enterprise environments using API gateways, service meshes, and multi-tenant architectures.
|
||||
|
||||
## Overview
|
||||
|
||||
Headers can be passed to LiteLLM in three ways, with the following priority order:
|
||||
1. **Request-specific headers** (highest priority)
|
||||
2. **extra_headers parameter**
|
||||
3. **Global litellm.headers** (lowest priority)
|
||||
|
||||
When the same header key is specified in multiple places, the higher priority value will be used.
|
||||
|
||||
## Usage Methods
|
||||
|
||||
### 1. Global Headers (litellm.headers)
|
||||
|
||||
Set headers that will be included in all API requests:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Set global headers for all requests
|
||||
litellm.headers = {
|
||||
"X-API-Gateway-Key": "your-gateway-key",
|
||||
"X-Company-ID": "acme-corp",
|
||||
"X-Environment": "production"
|
||||
}
|
||||
|
||||
# Now all completion calls will include these headers
|
||||
response = litellm.completion(
|
||||
model="claude-3-5-sonnet-latest",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
```
|
||||
|
||||
### 2. Per-Request Headers (extra_headers)
|
||||
|
||||
Pass headers for specific requests using the `extra_headers` parameter:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_headers={
|
||||
"X-Request-ID": "req-12345",
|
||||
"X-Tenant-ID": "tenant-abc",
|
||||
"X-Custom-Auth": "bearer-token-xyz"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### 3. Request Headers (headers parameter)
|
||||
|
||||
Use the `headers` parameter for the highest priority header control:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="claude-3-5-sonnet-latest",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
headers={
|
||||
"X-Priority-Header": "high-priority-value",
|
||||
"Authorization": "Bearer custom-token"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### 4. Combining All Methods
|
||||
|
||||
You can combine all three methods. Headers will be merged with the priority order:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Global headers (lowest priority)
|
||||
litellm.headers = {
|
||||
"X-Company-ID": "acme-corp",
|
||||
"X-Shared-Header": "global-value"
|
||||
}
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_headers={
|
||||
"X-Request-ID": "req-12345",
|
||||
"X-Shared-Header": "extra-value" # Overrides global
|
||||
},
|
||||
headers={
|
||||
"X-Priority-Header": "important",
|
||||
"X-Shared-Header": "request-value" # Overrides both global and extra
|
||||
}
|
||||
)
|
||||
|
||||
# Final headers sent to API:
|
||||
# {
|
||||
# "X-Company-ID": "acme-corp", # From global
|
||||
# "X-Request-ID": "req-12345", # From extra_headers
|
||||
# "X-Priority-Header": "important", # From headers
|
||||
# "X-Shared-Header": "request-value" # From headers (highest priority)
|
||||
# }
|
||||
```
|
||||
|
||||
## Enterprise Use Cases
|
||||
|
||||
### API Gateway Integration (Apigee, Kong, AWS API Gateway)
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Set up headers for API gateway routing and authentication
|
||||
litellm.headers = {
|
||||
"X-API-Gateway-Key": "your-gateway-key",
|
||||
"X-Route-Version": "v2"
|
||||
}
|
||||
|
||||
# Per-tenant requests
|
||||
response = litellm.completion(
|
||||
model="claude-3-5-sonnet-latest",
|
||||
messages=[{"role": "user", "content": "Analyze this data"}],
|
||||
extra_headers={
|
||||
"X-Tenant-ID": "tenant-123",
|
||||
"X-Department": "engineering"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Service Mesh (Istio, Linkerd)
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_headers={
|
||||
"X-Trace-ID": "trace-abc-123",
|
||||
"X-Service-Name": "ai-service",
|
||||
"X-Version": "1.2.3"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Multi-Tenant SaaS Applications
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
def make_ai_request(user_id, tenant_id, content):
|
||||
return litellm.completion(
|
||||
model="claude-3-5-sonnet-latest",
|
||||
messages=[{"role": "user", "content": content}],
|
||||
extra_headers={
|
||||
"X-User-ID": user_id,
|
||||
"X-Tenant-ID": tenant_id,
|
||||
"X-Request-Time": str(int(time.time()))
|
||||
}
|
||||
)
|
||||
|
||||
# Usage
|
||||
response = make_ai_request("user-456", "tenant-org-1", "Help me write code")
|
||||
```
|
||||
|
||||
### Request Tracing and Debugging
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import uuid
|
||||
|
||||
def traced_completion(model, messages, **kwargs):
|
||||
trace_id = str(uuid.uuid4())
|
||||
|
||||
return litellm.completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
extra_headers={
|
||||
"X-Trace-ID": trace_id,
|
||||
"X-Debug-Mode": "true",
|
||||
"X-Source-Service": "my-app"
|
||||
},
|
||||
**kwargs
|
||||
)
|
||||
|
||||
# Usage
|
||||
response = traced_completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Debug this issue"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Custom Authentication
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
def get_custom_auth_token():
|
||||
# Your custom authentication logic
|
||||
return "custom-auth-token"
|
||||
|
||||
response = litellm.completion(
|
||||
model="claude-3-5-sonnet-latest",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
headers={
|
||||
"X-Custom-Auth": get_custom_auth_token(),
|
||||
"X-Auth-Type": "custom"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
## Provider Support
|
||||
|
||||
Headers are supported across all LiteLLM providers including:
|
||||
|
||||
- **OpenAI** (GPT models)
|
||||
- **Anthropic** (Claude models)
|
||||
- **Cohere**
|
||||
- **Hugging Face**
|
||||
- **Custom providers**
|
||||
- **Azure OpenAI**
|
||||
- **AWS Bedrock**
|
||||
- **Google Vertex AI**
|
||||
|
||||
Each provider will receive your custom headers along with their required authentication and API-specific headers.
|
||||
|
||||
## Best Practices
|
||||
|
||||
### 1. Use Meaningful Header Names
|
||||
```python
|
||||
# Good
|
||||
extra_headers = {
|
||||
"X-Request-ID": "req-12345",
|
||||
"X-Tenant-ID": "org-456"
|
||||
}
|
||||
|
||||
# Avoid
|
||||
extra_headers = {
|
||||
"custom1": "value1",
|
||||
"h2": "value2"
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Include Tracing Information
|
||||
```python
|
||||
extra_headers = {
|
||||
"X-Trace-ID": trace_id,
|
||||
"X-Span-ID": span_id,
|
||||
"X-Service-Name": "ai-service"
|
||||
}
|
||||
```
|
||||
|
||||
### 3. Handle Sensitive Information Carefully
|
||||
```python
|
||||
# Don't log sensitive headers
|
||||
import os
|
||||
|
||||
if os.getenv("ENVIRONMENT") != "production":
|
||||
extra_headers["X-Debug-User"] = user_id
|
||||
```
|
||||
|
||||
### 4. Use Environment-Specific Headers
|
||||
```python
|
||||
import os
|
||||
|
||||
environment = os.getenv("ENVIRONMENT", "development")
|
||||
|
||||
litellm.headers = {
|
||||
"X-Environment": environment,
|
||||
"X-Service-Version": os.getenv("SERVICE_VERSION", "unknown")
|
||||
}
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Headers Not Being Passed
|
||||
|
||||
If your headers aren't reaching the API:
|
||||
|
||||
1. **Check Header Names**: Ensure header names don't conflict with provider-specific headers
|
||||
2. **Verify Priority**: Remember that `headers` > `extra_headers` > `litellm.headers`
|
||||
3. **Test with Logging**: Enable verbose logging to see what headers are being sent
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Enable debug logging
|
||||
litellm.set_verbose = True
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "test"}],
|
||||
extra_headers={"X-Debug": "test"}
|
||||
)
|
||||
```
|
||||
|
||||
### Gateway or Proxy Issues
|
||||
|
||||
If using API gateways or proxies:
|
||||
|
||||
1. **Check Gateway Requirements**: Verify required headers for your gateway
|
||||
2. **Test Direct vs Gateway**: Compare direct API calls vs gateway calls
|
||||
3. **Validate Header Format**: Some gateways have header format requirements
|
||||
|
||||
## Security Considerations
|
||||
|
||||
1. **Don't Log Sensitive Headers**: Avoid logging authentication tokens or personal data
|
||||
2. **Use HTTPS**: Always use secure connections when passing sensitive headers
|
||||
3. **Validate Header Values**: Sanitize user-provided header values
|
||||
4. **Rotate Keys**: Regularly rotate any API keys passed in headers
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import re
|
||||
|
||||
def safe_header_value(value):
|
||||
# Remove potentially dangerous characters
|
||||
return re.sub(r'[^\w\-.]', '', str(value))
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_headers={
|
||||
"X-User-ID": safe_header_value(user_id)
|
||||
}
|
||||
)
|
||||
```
|
||||
|
|
@ -1,5 +1,5 @@
|
|||
---
|
||||
title: "[PRE-RELEASE]v1.76.0-stable - RPS Improvements"
|
||||
title: "v1.76.0-stable - RPS Improvements"
|
||||
slug: "v1-76-0"
|
||||
date: 2025-08-23T10:00:00
|
||||
authors:
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
---
|
||||
title: "[Pre-Release] v1.77.2-stable - Bedrock Batches API"
|
||||
title: "v1.77.2-stable - Bedrock Batches API"
|
||||
slug: "v1-77-2"
|
||||
date: 2025-09-13T10:00:00
|
||||
authors:
|
||||
|
|
@ -21,12 +21,6 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
## Deploy this version
|
||||
|
||||
:::info
|
||||
|
||||
This release is not yet live.
|
||||
|
||||
:::
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
|
|
@ -34,7 +28,7 @@ This release is not yet live.
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-v1.77.2.rc.2
|
||||
ghcr.io/berriai/litellm:main-v1.77.2-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -42,6 +36,7 @@ ghcr.io/berriai/litellm:main-v1.77.2.rc.2
|
|||
<TabItem value="pip" label="Pip">
|
||||
|
||||
``` showLineNumbers title="pip install litellm"
|
||||
pip install litellm==1.77.2.post1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
258
docs/my-website/release_notes/v1.77.3-stable/index.md
Normal file
258
docs/my-website/release_notes/v1.77.3-stable/index.md
Normal file
|
|
@ -0,0 +1,258 @@
|
|||
---
|
||||
title: "[Preview] v1.77.3-stable - Priority Based Rate Limiting"
|
||||
slug: "v1-77-3"
|
||||
date: 2025-09-21T10:00:00
|
||||
authors:
|
||||
- name: Krrish Dholakia
|
||||
title: CEO, LiteLLM
|
||||
url: https://www.linkedin.com/in/krish-d/
|
||||
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||
- name: Ishaan Jaff
|
||||
title: CTO, LiteLLM
|
||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
|
||||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
## Deploy this version
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
``` showLineNumbers title="docker run litellm"
|
||||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-v1.77.3.rc.1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="pip" label="Pip">
|
||||
|
||||
``` showLineNumbers title="pip install litellm"
|
||||
pip install litellm==1.77.3
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
---
|
||||
|
||||
## Key Highlights
|
||||
|
||||
- **+550 RPS Performance Improvements** - Optimizations in request handling and object initialization.
|
||||
- **Priority Quota Reservation** - Proxy admins can now reserve TPM/RPM capacity for specific keys.
|
||||
|
||||
## Priority Quota Reservation
|
||||
|
||||
This release adds support for priority quota reservation. This allows **Proxy Admins** to reserve TPM/RPM capacity for keys based on metadata priority levels, ensuring critical production workloads get guaranteed access regardless of development traffic volume.
|
||||
|
||||
Get started [here](../../docs/proxy/dynamic_rate_limit#priority-quota-reservation)
|
||||
|
||||
<iframe width="700" height="500" src="https://www.loom.com/embed/1b54b93139ee415d959402cc0629f3f7" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
|
||||
## New Models / Updated Models
|
||||
|
||||
#### New Model Support
|
||||
|
||||
| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Features |
|
||||
| -------- | ----- | -------------- | ------------------- | -------------------- | -------- |
|
||||
| SambaNova | `sambanova/deepseek-v3.1` | 128K | $0.90 | $0.90 | Chat completions |
|
||||
| SambaNova | `sambanova/gpt-oss-120b` | 128K | $0.72 | $0.72 | Chat completions |
|
||||
| OVHCloud | Various models | Varies | Contact provider | Contact provider | Chat completions |
|
||||
| CompactifAI | Various models | Varies | Contact provider | Contact provider | Chat completions |
|
||||
| TwelveLabs | `twelvelabs/marengo-embed-2.7` | 32K | $0.12 | $0.00 | Embeddings |
|
||||
|
||||
#### Features
|
||||
|
||||
- **[OVHCloud AI Endpoints](../../docs/providers/ovhcloud)**
|
||||
- New provider support with comprehensive model catalog - [PR #14494](https://github.com/BerriAI/litellm/pull/14494)
|
||||
- **[CompactifAI](../../docs/providers/compactifai)**
|
||||
- New provider integration - [PR #14532](https://github.com/BerriAI/litellm/pull/14532)
|
||||
- **[SambaNova](../../docs/providers/sambanova)**
|
||||
- Added DeepSeek v3.1 and GPT-OSS-120B models - [PR #14500](https://github.com/BerriAI/litellm/pull/14500)
|
||||
- **[Bedrock](../../docs/providers/bedrock)**
|
||||
- Cross-region inference profile cost calculation - [PR #14566](https://github.com/BerriAI/litellm/pull/14566)
|
||||
- AWS external ID parameter support for authentication - [PR #14582](https://github.com/BerriAI/litellm/pull/14582)
|
||||
- CountTokens API implementation - [PR #14557](https://github.com/BerriAI/litellm/pull/14557)
|
||||
- Titan V2 encoding_format parameter support - [PR #14687](https://github.com/BerriAI/litellm/pull/14687)
|
||||
- Nova Canvas image generation inference profiles - [PR #14578](https://github.com/BerriAI/litellm/pull/14578)
|
||||
- Bedrock Batches API - batch processing support with file upload and request transformation - [PR #14618](https://github.com/BerriAI/litellm/pull/14618)
|
||||
- Bedrock Twelve Labs embedding provider support - [PR #14697](https://github.com/BerriAI/litellm/pull/14697)
|
||||
- **[Vertex AI](../../docs/providers/vertex)**
|
||||
- Gemini labels field provider-aware filtering - [PR #14563](https://github.com/BerriAI/litellm/pull/14563)
|
||||
- Gemini Batch API support - [PR #14733](https://github.com/BerriAI/litellm/pull/14733)
|
||||
- **[Volcengine](../../docs/providers/volcengine)**
|
||||
- Fixed thinking parameters when disabled - [PR #14569](https://github.com/BerriAI/litellm/pull/14569)
|
||||
- **[Cohere](../../docs/providers/cohere)**
|
||||
- Handle Generate API deprecation, default to chat endpoints - [PR #14676](https://github.com/BerriAI/litellm/pull/14676)
|
||||
- **[TwelveLabs](../../docs/providers/twelvelabs)**
|
||||
- Added Marengo Embed 2.7 embedding support - [PR #14674](https://github.com/BerriAI/litellm/pull/14674)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- **[Bedrock](../../docs/providers/bedrock)**
|
||||
- Empty arguments handling in tool call invocation - [PR #14583](https://github.com/BerriAI/litellm/pull/14583)
|
||||
- **[Vertex AI](../../docs/providers/vertex)**
|
||||
- Avoid deepcopy crash with non-pickleables in Gemini/Vertex - [PR #14418](https://github.com/BerriAI/litellm/pull/14418)
|
||||
- **[XAI](../../docs/providers/xai)**
|
||||
- Fix unsupported stop parameter for grok-code models - [PR #14565](https://github.com/BerriAI/litellm/pull/14565)
|
||||
- **[Gemini](../../docs/providers/gemini)**
|
||||
- Updated error message for Gemini API - [PR #14589](https://github.com/BerriAI/litellm/pull/14589)
|
||||
- Fixed 2.5 Flash Image Preview model routing - [PR #14715](https://github.com/BerriAI/litellm/pull/14715)
|
||||
- API key passing for token counting endpoints - [PR #14744](https://github.com/BerriAI/litellm/pull/14744)
|
||||
|
||||
#### New Provider Support
|
||||
|
||||
- **[OVHCloud AI Endpoints](../../docs/providers/ovhcloud)**
|
||||
- Complete provider integration with model catalog and authentication - [PR #14494](https://github.com/BerriAI/litellm/pull/14494)
|
||||
- **[CompactifAI](../../docs/providers/compactifai)**
|
||||
- New provider support with documentation - [PR #14532](https://github.com/BerriAI/litellm/pull/14532)
|
||||
|
||||
---
|
||||
|
||||
## LLM API Endpoints
|
||||
|
||||
#### Features
|
||||
|
||||
- **[/responses](../../docs/response_api)**
|
||||
- Added cancel endpoint support for non-admin users - [PR #14594](https://github.com/BerriAI/litellm/pull/14594)
|
||||
- Improved response session handling and cold storage configuration with s3 - [PR #14534](https://github.com/BerriAI/litellm/pull/14534)
|
||||
- Added OpenAI & Azure /responses/cancel endpoint support - [PR #14561](https://github.com/BerriAI/litellm/pull/14561)
|
||||
- **General**
|
||||
- Enhanced rate limit error messages with details - [PR #14736](https://github.com/BerriAI/litellm/pull/14736)
|
||||
- Middle-truncation for spend log payloads - [PR #14637](https://github.com/BerriAI/litellm/pull/14637)
|
||||
|
||||
#### Bugs
|
||||
|
||||
- **[/chat/completions](../../docs/completion/input)**
|
||||
- Fixed completion chat ID handling - [PR #14548](https://github.com/BerriAI/litellm/pull/14548)
|
||||
- Prevent AttributeError for _get_tags_from_request_kwargs - [PR #14735](https://github.com/BerriAI/litellm/pull/14735)
|
||||
- **[/responses](../../docs/response_api)**
|
||||
- Fixed cost calculation - [PR #14675](https://github.com/BerriAI/litellm/pull/14675)
|
||||
- **General**
|
||||
- Rate limiter AttributeError fix - [PR #14609](https://github.com/BerriAI/litellm/pull/14609)
|
||||
|
||||
---
|
||||
|
||||
## Spend Tracking, Budgets and Rate Limiting
|
||||
|
||||
- **Responses API Cost Calculation** fix - [PR #14675](https://github.com/BerriAI/litellm/pull/14675)
|
||||
- **Anthropic Cache Token Pricing** - Separate 1-hour vs 5-minute cache creation costs - [PR #14620](https://github.com/BerriAI/litellm/pull/14620), [PR #14652](https://github.com/BerriAI/litellm/pull/14652)
|
||||
- **Indochina Time Timezone** support for budget resets - [PR #14666](https://github.com/BerriAI/litellm/pull/14666)
|
||||
- **Soft Budget Alert Cache Issues** - Resolved soft budget alert cache issues - [PR #14491](https://github.com/BerriAI/litellm/pull/14491)
|
||||
- **Dynamic Rate Limiter v3** - Priority routing improvements - [PR #14734](https://github.com/BerriAI/litellm/pull/14734)
|
||||
- **Enhanced Rate Limit Errors** - More detailed error messages - [PR #14736](https://github.com/BerriAI/litellm/pull/14736)
|
||||
|
||||
---
|
||||
|
||||
## Management Endpoints / UI
|
||||
|
||||
#### Features
|
||||
|
||||
- **Team Member Service Account Keys** - Allow team members to view keys they create - [PR #14619](https://github.com/BerriAI/litellm/pull/14619)
|
||||
- **Default Budget for JWT Teams** - Auto-assign budgets to generated teams - [PR #14514](https://github.com/BerriAI/litellm/pull/14514)
|
||||
- **SSO Access Control Groups** - Enhanced token info endpoint integration - [PR #14738](https://github.com/BerriAI/litellm/pull/14738)
|
||||
- **Health Test Connect Protection** - Restrict access based on model creation permissions - [PR #14650](https://github.com/BerriAI/litellm/pull/14650)
|
||||
- **Amazon Bedrock Guardrail Info View** - Enhanced logging visualization - [PR #14696](https://github.com/BerriAI/litellm/pull/14696)
|
||||
|
||||
#### Bug Fixes
|
||||
|
||||
- **SCIM v2** - Fix group PUSH and PUT operations for non-existent members - [PR #14581](https://github.com/BerriAI/litellm/pull/14581)
|
||||
- **Guardrail View/Edit/Delete** behavior fixes - [PR #14622](https://github.com/BerriAI/litellm/pull/14622)
|
||||
- **In-Memory Guardrail** update failures - [PR #14653](https://github.com/BerriAI/litellm/pull/14653)
|
||||
|
||||
---
|
||||
|
||||
## Logging / Guardrail Integrations
|
||||
|
||||
#### Features
|
||||
|
||||
- **[DataDog](../../docs/proxy/logging#datadog)**
|
||||
- Enhanced spend tracking metrics - [PR #14555](https://github.com/BerriAI/litellm/pull/14555)
|
||||
- Stream support with is_streamed_request parameter - [PR #14673](https://github.com/BerriAI/litellm/pull/14673)
|
||||
- Fixed tool calls metadata passing - [PR #14531](https://github.com/BerriAI/litellm/pull/14531)
|
||||
- **[Langfuse](../../docs/proxy/logging#langfuse)**
|
||||
- Added logging support for Responses API - [PR #14597](https://github.com/BerriAI/litellm/pull/14597)
|
||||
- **[Langsmith](../../docs/proxy/logging#langsmith)**
|
||||
- Langsmith Sampling Rate - Key/Team-level tracing configuration - [PR #14740](https://github.com/BerriAI/litellm/pull/14740)
|
||||
- **[Prometheus](../../docs/proxy/logging#prometheus)**
|
||||
- Multi-worker support improvements - [PR #14530](https://github.com/BerriAI/litellm/pull/14530)
|
||||
- User email labels in monitoring - [PR #14520](https://github.com/BerriAI/litellm/pull/14520)
|
||||
- **[Opik](../../docs/proxy/logging#opik)**
|
||||
- Fixed timezone issue - [PR #14708](https://github.com/BerriAI/litellm/pull/14708)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- **[S3](../../docs/proxy/logging#s3-buckets)**
|
||||
- Fixed 404 error when using s3_endpoint_url - [PR #14559](https://github.com/BerriAI/litellm/pull/14559)
|
||||
|
||||
#### Guardrails
|
||||
|
||||
- **Tool Permission Guardrail** - Fine-grained tool access control - [PR #14519](https://github.com/BerriAI/litellm/pull/14519)
|
||||
- **Bedrock Guardrails** - Selective guarding support with runtime endpoint configuration - [PR #14575](https://github.com/BerriAI/litellm/pull/14575), [PR #14650](https://github.com/BerriAI/litellm/pull/14650)
|
||||
- **Default Last Message** in guardrails - [PR #14640](https://github.com/BerriAI/litellm/pull/14640)
|
||||
- **AWS exceptions handling despite 200 response** - [PR #14658](https://github.com/BerriAI/litellm/pull/14658)
|
||||
#### New Integration
|
||||
|
||||
- **[PostHog](../../docs/observability/posthog)** - Complete observability integration for LiteLLM usage tracking and analytics - [PR #14610](https://github.com/BerriAI/litellm/pull/14610)
|
||||
|
||||
---
|
||||
|
||||
|
||||
## MCP Gateway
|
||||
|
||||
- **MCP Server Alias Parsing** - Multi-part URL path support - [PR #14558](https://github.com/BerriAI/litellm/pull/14558)
|
||||
- **MCP Filter Recomputation** - After server deletion - [PR #14542](https://github.com/BerriAI/litellm/pull/14542)
|
||||
- **MCP Gateway Tools List** improvements - [PR #14695](https://github.com/BerriAI/litellm/pull/14695)
|
||||
|
||||
---
|
||||
|
||||
## Performance / Loadbalancing / Reliability improvements
|
||||
|
||||
- **+500 RPS Performance Boost** when sending the `user` field - [PR #14616](https://github.com/BerriAI/litellm/pull/14616)
|
||||
- **+50 RPS** by removing iscoroutine from hot path - [PR #14649](https://github.com/BerriAI/litellm/pull/14649)
|
||||
- **7% reduction** in __init__ overhead - [PR #14689](https://github.com/BerriAI/litellm/pull/14689)
|
||||
- **Generic Object Pool** implementation for better resource management - [PR #14702](https://github.com/BerriAI/litellm/pull/14702)
|
||||
|
||||
---
|
||||
|
||||
## General Proxy Improvements
|
||||
|
||||
- **Middle-Truncation** for spend log payloads - [PR #14637](https://github.com/BerriAI/litellm/pull/14637)
|
||||
|
||||
#### Security
|
||||
|
||||
- **Security Update** - Bump aiohttp==3.12.14, fix CVE-2025-53643 - [PR #14638](https://github.com/BerriAI/litellm/pull/14638)
|
||||
|
||||
---
|
||||
|
||||
## New Contributors
|
||||
|
||||
* @luisfucros made their first contribution in [PR #14500](https://github.com/BerriAI/litellm/pull/14500)
|
||||
* @hanakannzashi made their first contribution in [PR #14548](https://github.com/BerriAI/litellm/pull/14548)
|
||||
* @eliasto made their first contribution in [PR #14494](https://github.com/BerriAI/litellm/pull/14494)
|
||||
* @Rasmusafj made their first contribution in [PR #14491](https://github.com/BerriAI/litellm/pull/14491)
|
||||
* @LingXuanYin made their first contribution in [PR #14569](https://github.com/BerriAI/litellm/pull/14569)
|
||||
* @ronaldpereira made their first contribution in [PR #14613](https://github.com/BerriAI/litellm/pull/14613)
|
||||
* @hula-la made their first contribution in [PR #14534](https://github.com/BerriAI/litellm/pull/14534)
|
||||
* @carlos-marchal-ph made their first contribution in [PR #14610](https://github.com/BerriAI/litellm/pull/14610)
|
||||
* @akraines made their first contribution in [PR #14637](https://github.com/BerriAI/litellm/pull/14637)
|
||||
* @mrFranklin made their first contribution in [PR #14708](https://github.com/BerriAI/litellm/pull/14708)
|
||||
* @tcx4c70 made their first contribution in [PR #14675](https://github.com/BerriAI/litellm/pull/14675)
|
||||
* @michaeltansg made their first contribution in [PR #14666](https://github.com/BerriAI/litellm/pull/14666)
|
||||
* @tosi29 made their first contribution in [PR #14725](https://github.com/BerriAI/litellm/pull/14725)
|
||||
* @gmdfalk made their first contribution in [PR #14735](https://github.com/BerriAI/litellm/pull/14735)
|
||||
* @FelipeRodriguesGare made their first contribution in [PR #14733](https://github.com/BerriAI/litellm/pull/14733)
|
||||
* @mritunjaysharma394 made their first contribution in [PR #14678](https://github.com/BerriAI/litellm/pull/14678)
|
||||
|
||||
---
|
||||
|
||||
## **[Full Changelog](https://github.com/BerriAI/litellm/compare/v1.77.2.rc.1...v1.77.3.rc.1)**
|
||||
|
|
@ -201,7 +201,7 @@ const sidebars = {
|
|||
{
|
||||
type: "category",
|
||||
label: "Budgets + Rate Limits",
|
||||
items: ["proxy/users", "proxy/temporary_budget_increase", "proxy/rate_limit_tiers", "proxy/team_budgets", "proxy/customers"],
|
||||
items: ["proxy/users", "proxy/temporary_budget_increase", "proxy/rate_limit_tiers", "proxy/team_budgets", "proxy/dynamic_rate_limit", "proxy/customers"],
|
||||
},
|
||||
{
|
||||
type: "link",
|
||||
|
|
@ -392,6 +392,7 @@ const sidebars = {
|
|||
"providers/vertex",
|
||||
"providers/vertex_partner",
|
||||
"providers/vertex_image",
|
||||
"providers/vertex_batch",
|
||||
]
|
||||
},
|
||||
{
|
||||
|
|
@ -523,6 +524,7 @@ const sidebars = {
|
|||
"completion/batching",
|
||||
"completion/mock_requests",
|
||||
"completion/reliable_completions",
|
||||
"proxy/veo_video_generation",
|
||||
|
||||
]
|
||||
},
|
||||
|
|
|
|||
195
examples/sdk_headers_example.py
Normal file
195
examples/sdk_headers_example.py
Normal file
|
|
@ -0,0 +1,195 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Example demonstrating LiteLLM SDK header support for enterprise environments.
|
||||
|
||||
This example shows how to use additional headers with API gateways, service meshes,
|
||||
and multi-tenant architectures.
|
||||
"""
|
||||
|
||||
import litellm
|
||||
import os
|
||||
from typing import Dict, Any
|
||||
|
||||
def example_global_headers():
|
||||
"""Example: Set global headers for all requests"""
|
||||
print("=== Global Headers Example ===")
|
||||
|
||||
# Set global headers that will be included in all API requests
|
||||
litellm.headers = {
|
||||
"X-API-Gateway-Key": "your-gateway-key-here",
|
||||
"X-Company-ID": "acme-corp",
|
||||
"X-Environment": "production"
|
||||
}
|
||||
|
||||
print("Global headers set:", litellm.headers)
|
||||
|
||||
# These headers will now be included in all completion calls
|
||||
# (Note: This example doesn't actually make API calls)
|
||||
print("Global headers will be included in all subsequent completion() calls")
|
||||
|
||||
|
||||
def example_per_request_headers():
|
||||
"""Example: Using extra_headers for specific requests"""
|
||||
print("\n=== Per-Request Headers Example ===")
|
||||
|
||||
headers_to_send = {
|
||||
"X-Request-ID": "req-12345",
|
||||
"X-Tenant-ID": "tenant-abc",
|
||||
"X-Custom-Auth": "bearer-token-xyz"
|
||||
}
|
||||
|
||||
print("Per-request headers:", headers_to_send)
|
||||
|
||||
# Example of how you would use extra_headers in a real call
|
||||
# response = litellm.completion(
|
||||
# model="claude-3-5-sonnet-latest",
|
||||
# messages=[{"role": "user", "content": "Hello"}],
|
||||
# extra_headers=headers_to_send
|
||||
# )
|
||||
|
||||
|
||||
def example_header_priority():
|
||||
"""Example: Demonstrating header priority and merging"""
|
||||
print("\n=== Header Priority Example ===")
|
||||
|
||||
# Set global headers
|
||||
litellm.headers = {
|
||||
"X-Company-ID": "acme-corp",
|
||||
"X-Shared-Header": "global-value"
|
||||
}
|
||||
|
||||
# Headers that would be sent in a request
|
||||
extra_headers = {
|
||||
"X-Request-ID": "req-12345",
|
||||
"X-Shared-Header": "extra-value" # Overrides global
|
||||
}
|
||||
|
||||
request_headers = {
|
||||
"X-Priority-Header": "important",
|
||||
"X-Shared-Header": "request-value" # Overrides both global and extra
|
||||
}
|
||||
|
||||
print("Global headers:", litellm.headers)
|
||||
print("Extra headers:", extra_headers)
|
||||
print("Request headers:", request_headers)
|
||||
print("\nFinal headers would be:")
|
||||
print(" X-Company-ID: acme-corp (from global)")
|
||||
print(" X-Request-ID: req-12345 (from extra)")
|
||||
print(" X-Priority-Header: important (from request)")
|
||||
print(" X-Shared-Header: request-value (request wins - highest priority)")
|
||||
|
||||
|
||||
def example_enterprise_api_gateway():
|
||||
"""Example: Enterprise API Gateway scenario"""
|
||||
print("\n=== Enterprise API Gateway Example ===")
|
||||
|
||||
# Simulate enterprise environment with Apigee or similar
|
||||
gateway_config = {
|
||||
"X-API-Gateway-Key": os.getenv("API_GATEWAY_KEY", "demo-key"),
|
||||
"X-Route-Version": "v2",
|
||||
"X-Rate-Limit-Group": "premium"
|
||||
}
|
||||
|
||||
# Set gateway headers globally
|
||||
litellm.headers = gateway_config
|
||||
print("Gateway headers configured:", gateway_config)
|
||||
|
||||
# Function to make tenant-specific requests
|
||||
def make_tenant_request(tenant_id: str, user_id: str, content: str) -> Dict[str, Any]:
|
||||
"""Make an AI request with tenant-specific headers"""
|
||||
|
||||
tenant_headers = {
|
||||
"X-Tenant-ID": tenant_id,
|
||||
"X-User-ID": user_id,
|
||||
"X-Request-Time": "2024-01-01T00:00:00Z",
|
||||
"X-Service-Name": "ai-assistant"
|
||||
}
|
||||
|
||||
print(f"Making request for tenant {tenant_id}, user {user_id}")
|
||||
print("Tenant-specific headers:", tenant_headers)
|
||||
|
||||
# In a real scenario, this would make the actual API call:
|
||||
# return litellm.completion(
|
||||
# model="claude-3-5-sonnet-latest",
|
||||
# messages=[{"role": "user", "content": content}],
|
||||
# extra_headers=tenant_headers
|
||||
# )
|
||||
|
||||
# For demo purposes, return mock data
|
||||
return {"mock": "response", "headers_used": {**gateway_config, **tenant_headers}}
|
||||
|
||||
# Example usage
|
||||
result = make_tenant_request("tenant-123", "user-456", "Analyze this data")
|
||||
print("Response:", result)
|
||||
|
||||
|
||||
def example_service_mesh():
|
||||
"""Example: Service mesh integration (Istio, Linkerd)"""
|
||||
print("\n=== Service Mesh Example ===")
|
||||
|
||||
service_mesh_headers = {
|
||||
"X-Trace-ID": "trace-abc-123",
|
||||
"X-Span-ID": "span-def-456",
|
||||
"X-Service-Name": "ai-service",
|
||||
"X-Version": "1.2.3",
|
||||
"X-Cluster": "prod-us-west-2"
|
||||
}
|
||||
|
||||
print("Service mesh headers:", service_mesh_headers)
|
||||
|
||||
# Example of using these headers for distributed tracing
|
||||
# response = litellm.completion(
|
||||
# model="gpt-4",
|
||||
# messages=[{"role": "user", "content": "Hello"}],
|
||||
# extra_headers=service_mesh_headers
|
||||
# )
|
||||
|
||||
|
||||
def example_debugging_and_monitoring():
|
||||
"""Example: Request debugging and monitoring"""
|
||||
print("\n=== Debugging and Monitoring Example ===")
|
||||
|
||||
import uuid
|
||||
import time
|
||||
|
||||
# Generate unique identifiers for request tracking
|
||||
trace_id = str(uuid.uuid4())
|
||||
request_id = f"req-{int(time.time())}"
|
||||
|
||||
debug_headers = {
|
||||
"X-Trace-ID": trace_id,
|
||||
"X-Request-ID": request_id,
|
||||
"X-Debug-Mode": "true",
|
||||
"X-Source-Service": "customer-support-bot",
|
||||
"X-Request-Priority": "high"
|
||||
}
|
||||
|
||||
print("Debug headers:", debug_headers)
|
||||
print(f"Trace ID: {trace_id}")
|
||||
print(f"Request ID: {request_id}")
|
||||
|
||||
# These headers help with:
|
||||
# 1. Distributed tracing across services
|
||||
# 2. Request correlation in logs
|
||||
# 3. Debug mode enablement
|
||||
# 4. Priority-based routing
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("LiteLLM SDK Header Support Examples")
|
||||
print("=" * 50)
|
||||
|
||||
example_global_headers()
|
||||
example_per_request_headers()
|
||||
example_header_priority()
|
||||
example_enterprise_api_gateway()
|
||||
example_service_mesh()
|
||||
example_debugging_and_monitoring()
|
||||
|
||||
print("\n" + "=" * 50)
|
||||
print("All examples completed!")
|
||||
print("\nTo use in your application:")
|
||||
print("1. Set litellm.headers for global headers")
|
||||
print("2. Use extra_headers parameter for request-specific headers")
|
||||
print("3. Use headers parameter for highest priority headers")
|
||||
print("4. Headers are merged with priority: headers > extra_headers > litellm.headers")
|
||||
BIN
litellm-proxy-extras/dist/litellm_proxy_extras-0.2.19-py3-none-any.whl
vendored
Normal file
BIN
litellm-proxy-extras/dist/litellm_proxy_extras-0.2.19-py3-none-any.whl
vendored
Normal file
Binary file not shown.
BIN
litellm-proxy-extras/dist/litellm_proxy_extras-0.2.19.tar.gz
vendored
Normal file
BIN
litellm-proxy-extras/dist/litellm_proxy_extras-0.2.19.tar.gz
vendored
Normal file
Binary file not shown.
|
|
@ -1,6 +1,6 @@
|
|||
[tool.poetry]
|
||||
name = "litellm-proxy-extras"
|
||||
version = "0.2.18"
|
||||
version = "0.2.19"
|
||||
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
|
||||
authors = ["BerriAI"]
|
||||
readme = "README.md"
|
||||
|
|
@ -22,7 +22,7 @@ requires = ["poetry-core"]
|
|||
build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "0.2.18"
|
||||
version = "0.2.19"
|
||||
version_files = [
|
||||
"pyproject.toml:version",
|
||||
"../requirements.txt:litellm-proxy-extras==",
|
||||
|
|
|
|||
|
|
@ -117,6 +117,7 @@ _custom_logger_compatible_callbacks_literal = Literal[
|
|||
"logfire",
|
||||
"literalai",
|
||||
"dynamic_rate_limiter",
|
||||
"dynamic_rate_limiter_v3",
|
||||
"langsmith",
|
||||
"prometheus",
|
||||
"otel",
|
||||
|
|
|
|||
|
|
@ -186,7 +186,9 @@ class MCPClient:
|
|||
|
||||
def _get_auth_headers(self) -> dict:
|
||||
"""Generate authentication headers based on auth type."""
|
||||
headers = {}
|
||||
headers = {
|
||||
"MCP-Protocol-Version": "2025-06-18"
|
||||
}
|
||||
|
||||
if self._mcp_auth_value:
|
||||
if self.auth_type == MCPAuth.bearer_token:
|
||||
|
|
|
|||
|
|
@ -731,7 +731,7 @@ def file_list(
|
|||
|
||||
async def afile_content(
|
||||
file_id: str,
|
||||
custom_llm_provider: Literal["openai", "azure"] = "openai",
|
||||
custom_llm_provider: Literal["openai", "azure", "vertex_ai"] = "openai",
|
||||
extra_headers: Optional[Dict[str, str]] = None,
|
||||
extra_body: Optional[Dict[str, str]] = None,
|
||||
**kwargs,
|
||||
|
|
@ -887,6 +887,32 @@ def file_content(
|
|||
client=client,
|
||||
litellm_params=litellm_params_dict,
|
||||
)
|
||||
elif custom_llm_provider == "vertex_ai":
|
||||
api_base = optional_params.api_base or ""
|
||||
vertex_ai_project = (
|
||||
optional_params.vertex_project
|
||||
or litellm.vertex_project
|
||||
or get_secret_str("VERTEXAI_PROJECT")
|
||||
)
|
||||
vertex_ai_location = (
|
||||
optional_params.vertex_location
|
||||
or litellm.vertex_location
|
||||
or get_secret_str("VERTEXAI_LOCATION")
|
||||
)
|
||||
vertex_credentials = optional_params.vertex_credentials or get_secret_str(
|
||||
"VERTEXAI_CREDENTIALS"
|
||||
)
|
||||
|
||||
response = vertex_ai_files_instance.file_content(
|
||||
_is_async=_is_async,
|
||||
file_content_request=_file_content_request,
|
||||
api_base=api_base,
|
||||
vertex_credentials=vertex_credentials,
|
||||
vertex_project=vertex_ai_project,
|
||||
vertex_location=vertex_ai_location,
|
||||
timeout=timeout,
|
||||
max_retries=optional_params.max_retries,
|
||||
)
|
||||
else:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="LiteLLM doesn't support {} for 'custom_llm_provider'. Supported providers are 'openai', 'azure', 'vertex_ai'.".format(
|
||||
|
|
|
|||
|
|
@ -47,6 +47,7 @@ from litellm.integrations.vector_store_integrations.vector_store_pre_call_hook i
|
|||
VectorStorePreCallHook,
|
||||
)
|
||||
from litellm.proxy.hooks.dynamic_rate_limiter import _PROXY_DynamicRateLimitHandler
|
||||
from litellm.proxy.hooks.dynamic_rate_limiter_v3 import _PROXY_DynamicRateLimitHandlerV3
|
||||
|
||||
|
||||
class CustomLoggerRegistry:
|
||||
|
|
@ -86,6 +87,7 @@ class CustomLoggerRegistry:
|
|||
"s3_v2": S3Logger,
|
||||
"aws_sqs": SQSLogger,
|
||||
"dynamic_rate_limiter": _PROXY_DynamicRateLimitHandler,
|
||||
"dynamic_rate_limiter_v3": _PROXY_DynamicRateLimitHandlerV3,
|
||||
"vector_store_pre_call_hook": VectorStorePreCallHook,
|
||||
"dotprompt": DotpromptManager,
|
||||
"cloudzero": CloudZeroLogger,
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ import httpx
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
from ..exceptions import (
|
||||
APIConnectionError,
|
||||
|
|
@ -762,7 +763,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
error_str += "XXXXXXX" + '"'
|
||||
|
||||
raise AuthenticationError(
|
||||
message=f"{custom_llm_provider}Exception: Authentication Error - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception: Authentication Error - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
response=getattr(original_exception, "response", None),
|
||||
|
|
@ -771,14 +772,14 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
elif "model's maximum context limit" in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise ContextWindowExceededError(
|
||||
message=f"{custom_llm_provider}Exception: Context Window Error - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception: Context Window Error - {error_str}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
)
|
||||
elif "token_quota_reached" in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise RateLimitError(
|
||||
message=f"{custom_llm_provider}Exception: Rate Limit Errror - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception: Rate Limit Errror - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
response=getattr(original_exception, "response", None),
|
||||
|
|
@ -789,14 +790,14 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise litellm.InternalServerError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
elif "model_no_support_for_function" in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"{custom_llm_provider}Exception - Use 'watsonx_text' route instead. IBM WatsonX does not support `/text/chat` endpoint. - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - Use 'watsonx_text' route instead. IBM WatsonX does not support `/text/chat` endpoint. - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
|
|
@ -804,7 +805,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
if original_exception.status_code == 500:
|
||||
exception_mapping_worked = True
|
||||
raise litellm.InternalServerError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
|
|
@ -814,28 +815,28 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise AuthenticationError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
elif original_exception.status_code == 400:
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
elif original_exception.status_code == 404:
|
||||
exception_mapping_worked = True
|
||||
raise NotFoundError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
elif original_exception.status_code == 408:
|
||||
exception_mapping_worked = True
|
||||
raise Timeout(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
|
|
@ -846,7 +847,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
|
|
@ -854,7 +855,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
elif original_exception.status_code == 429:
|
||||
exception_mapping_worked = True
|
||||
raise RateLimitError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
|
|
@ -862,7 +863,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
elif original_exception.status_code == 503:
|
||||
exception_mapping_worked = True
|
||||
raise ServiceUnavailableError(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
|
|
@ -870,7 +871,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
elif original_exception.status_code == 504: # gateway timeout error
|
||||
exception_mapping_worked = True
|
||||
raise Timeout(
|
||||
message=f"{custom_llm_provider}Exception - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {original_exception.message}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
|
|
@ -1168,9 +1169,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
exception_status_code=original_exception.status_code,
|
||||
)
|
||||
elif (
|
||||
custom_llm_provider == "vertex_ai"
|
||||
or custom_llm_provider == "vertex_ai_beta"
|
||||
or custom_llm_provider == "gemini"
|
||||
custom_llm_provider == LlmProviders.VERTEX_AI
|
||||
or custom_llm_provider == LlmProviders.VERTEX_AI_BETA
|
||||
or custom_llm_provider == LlmProviders.GEMINI
|
||||
):
|
||||
if (
|
||||
"Vertex AI API has not been used in project" in error_str
|
||||
|
|
@ -1178,9 +1179,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"litellm.BadRequestError: VertexAIException - {error_str}",
|
||||
message=f"litellm.BadRequestError: {custom_llm_provider}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
response=httpx.Response(
|
||||
status_code=400,
|
||||
request=httpx.Request(
|
||||
|
|
@ -1193,7 +1194,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
if "400 Request payload size exceeds" in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise ContextWindowExceededError(
|
||||
message=f"VertexException - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
|
@ -1203,9 +1204,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise litellm.InternalServerError(
|
||||
message=f"litellm.InternalServerError: VertexAIException - {error_str}",
|
||||
message=f"litellm.InternalServerError: {custom_llm_provider}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
response=httpx.Response(
|
||||
status_code=500,
|
||||
content=str(original_exception),
|
||||
|
|
@ -1216,7 +1217,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
elif "API key not valid." in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise AuthenticationError(
|
||||
message=f"{custom_llm_provider}Exception - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
|
|
@ -1224,9 +1225,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
elif "403" in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"VertexAIException BadRequestError - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception BadRequestError - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
response=httpx.Response(
|
||||
status_code=403,
|
||||
request=httpx.Request(
|
||||
|
|
@ -1243,9 +1244,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise ContentPolicyViolationError(
|
||||
message=f"VertexAIException ContentPolicyViolationError - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception ContentPolicyViolationError - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
response=httpx.Response(
|
||||
status_code=400,
|
||||
|
|
@ -1264,9 +1265,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise RateLimitError(
|
||||
message=f"litellm.RateLimitError: VertexAIException - {error_str}",
|
||||
message=f"litellm.RateLimitError: {custom_llm_provider}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
response=httpx.Response(
|
||||
status_code=429,
|
||||
|
|
@ -1282,18 +1283,18 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
):
|
||||
exception_mapping_worked = True
|
||||
raise litellm.InternalServerError(
|
||||
message=f"litellm.InternalServerError: VertexAIException - {error_str}",
|
||||
message=f"litellm.InternalServerError: {custom_llm_provider}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
)
|
||||
if hasattr(original_exception, "status_code"):
|
||||
if original_exception.status_code == 400:
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"VertexAIException BadRequestError - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception BadRequestError - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
response=httpx.Response(
|
||||
status_code=400,
|
||||
|
|
@ -1306,21 +1307,35 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
if original_exception.status_code == 401:
|
||||
exception_mapping_worked = True
|
||||
raise AuthenticationError(
|
||||
message=f"VertexAIException - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
if original_exception.status_code == 403:
|
||||
exception_mapping_worked = True
|
||||
raise PermissionDeniedError(
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
response=httpx.Response(
|
||||
status_code=403,
|
||||
request=httpx.Request(
|
||||
method="POST",
|
||||
url="https://cloud.google.com/vertex-ai/",
|
||||
),
|
||||
),
|
||||
)
|
||||
if original_exception.status_code == 404:
|
||||
exception_mapping_worked = True
|
||||
raise NotFoundError(
|
||||
message=f"VertexAIException - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
if original_exception.status_code == 408:
|
||||
exception_mapping_worked = True
|
||||
raise Timeout(
|
||||
message=f"VertexAIException - {original_exception.message}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
|
|
@ -1328,9 +1343,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
if original_exception.status_code == 429:
|
||||
exception_mapping_worked = True
|
||||
raise RateLimitError(
|
||||
message=f"litellm.RateLimitError: VertexAIException - {error_str}",
|
||||
message=f"litellm.RateLimitError: {custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
response=httpx.Response(
|
||||
status_code=429,
|
||||
|
|
@ -1343,9 +1358,9 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
if original_exception.status_code == 500:
|
||||
exception_mapping_worked = True
|
||||
raise litellm.InternalServerError(
|
||||
message=f"VertexAIException InternalServerError - {error_str}",
|
||||
message=f"{custom_llm_provider.capitalize()}Exception InternalServerError - {error_str}",
|
||||
model=model,
|
||||
llm_provider="vertex_ai",
|
||||
llm_provider=custom_llm_provider,
|
||||
litellm_debug_info=extra_information,
|
||||
response=httpx.Response(
|
||||
status_code=500,
|
||||
|
|
@ -1353,71 +1368,20 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
request=httpx.Request(method="completion", url="https://github.com/BerriAI/litellm"), # type: ignore
|
||||
),
|
||||
)
|
||||
if original_exception.status_code == 503:
|
||||
if original_exception.status_code == 502:
|
||||
exception_mapping_worked = True
|
||||
raise ServiceUnavailableError(
|
||||
message=f"VertexAIException - {original_exception.message}",
|
||||
raise APIConnectionError(
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
)
|
||||
elif custom_llm_provider == "palm" or custom_llm_provider == "gemini":
|
||||
if "503 Getting metadata" in error_str:
|
||||
# auth errors look like this
|
||||
# 503 Getting metadata from plugin failed with error: Reauthentication is needed. Please run `gcloud auth application-default login` to reauthenticate.
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message="GeminiException - Invalid api key",
|
||||
model=model,
|
||||
llm_provider="palm",
|
||||
response=getattr(original_exception, "response", None),
|
||||
)
|
||||
if (
|
||||
"504 Deadline expired before operation could complete." in error_str
|
||||
or "504 Deadline Exceeded" in error_str
|
||||
):
|
||||
exception_mapping_worked = True
|
||||
raise Timeout(
|
||||
message=f"GeminiException - {original_exception.message}",
|
||||
model=model,
|
||||
llm_provider="palm",
|
||||
exception_status_code=original_exception.status_code,
|
||||
)
|
||||
if "400 Request payload size exceeds" in error_str:
|
||||
exception_mapping_worked = True
|
||||
raise ContextWindowExceededError(
|
||||
message=f"GeminiException - {error_str}",
|
||||
model=model,
|
||||
llm_provider="palm",
|
||||
response=getattr(original_exception, "response", None),
|
||||
)
|
||||
if (
|
||||
"500 An internal error has occurred." in error_str
|
||||
or "list index out of range" in error_str
|
||||
):
|
||||
exception_mapping_worked = True
|
||||
raise APIError(
|
||||
status_code=getattr(original_exception, "status_code", 500),
|
||||
message=f"GeminiException - {original_exception.message}",
|
||||
llm_provider="palm",
|
||||
model=model,
|
||||
request=httpx.Response(
|
||||
status_code=429,
|
||||
request=httpx.Request(
|
||||
method="POST",
|
||||
url=" https://cloud.google.com/vertex-ai/",
|
||||
),
|
||||
),
|
||||
)
|
||||
if hasattr(original_exception, "status_code"):
|
||||
if original_exception.status_code == 400:
|
||||
if original_exception.status_code == 503:
|
||||
exception_mapping_worked = True
|
||||
raise BadRequestError(
|
||||
message=f"GeminiException - {error_str}",
|
||||
raise ServiceUnavailableError(
|
||||
message=f"{custom_llm_provider.capitalize()}Exception - {error_str}",
|
||||
llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
llm_provider="palm",
|
||||
response=getattr(original_exception, "response", None),
|
||||
)
|
||||
# Dailed: Error occurred: 400 Request payload size exceeds the limit: 20000 bytes
|
||||
elif custom_llm_provider == "cloudflare":
|
||||
if "Authentication error" in error_str:
|
||||
exception_mapping_worked = True
|
||||
|
|
|
|||
|
|
@ -3444,6 +3444,30 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
|
|||
dynamic_rate_limiter_obj.update_variables(llm_router=llm_router)
|
||||
_in_memory_loggers.append(dynamic_rate_limiter_obj)
|
||||
return dynamic_rate_limiter_obj # type: ignore
|
||||
elif logging_integration == "dynamic_rate_limiter_v3":
|
||||
from litellm.proxy.hooks.dynamic_rate_limiter_v3 import (
|
||||
_PROXY_DynamicRateLimitHandlerV3,
|
||||
)
|
||||
|
||||
for callback in _in_memory_loggers:
|
||||
if isinstance(callback, _PROXY_DynamicRateLimitHandlerV3):
|
||||
return callback # type: ignore
|
||||
|
||||
if internal_usage_cache is None:
|
||||
raise Exception(
|
||||
"Internal Error: Cache cannot be empty - internal_usage_cache={}".format(
|
||||
internal_usage_cache
|
||||
)
|
||||
)
|
||||
|
||||
dynamic_rate_limiter_obj_v3 = _PROXY_DynamicRateLimitHandlerV3(
|
||||
internal_usage_cache=internal_usage_cache
|
||||
)
|
||||
|
||||
if llm_router is not None and isinstance(llm_router, litellm.Router):
|
||||
dynamic_rate_limiter_obj_v3.update_variables(llm_router=llm_router)
|
||||
_in_memory_loggers.append(dynamic_rate_limiter_obj_v3)
|
||||
return dynamic_rate_limiter_obj_v3 # type: ignore
|
||||
elif logging_integration == "langtrace":
|
||||
if "LANGTRACE_API_KEY" not in os.environ:
|
||||
raise ValueError("LANGTRACE_API_KEY not found in environment variables")
|
||||
|
|
@ -3707,6 +3731,14 @@ def get_custom_logger_compatible_class( # noqa: PLR0915
|
|||
for callback in _in_memory_loggers:
|
||||
if isinstance(callback, _PROXY_DynamicRateLimitHandler):
|
||||
return callback # type: ignore
|
||||
elif logging_integration == "dynamic_rate_limiter_v3":
|
||||
from litellm.proxy.hooks.dynamic_rate_limiter_v3 import (
|
||||
_PROXY_DynamicRateLimitHandlerV3,
|
||||
)
|
||||
|
||||
for callback in _in_memory_loggers:
|
||||
if isinstance(callback, _PROXY_DynamicRateLimitHandlerV3):
|
||||
return callback # type: ignore
|
||||
|
||||
elif logging_integration == "langtrace":
|
||||
from litellm.integrations.opentelemetry import OpenTelemetry
|
||||
|
|
|
|||
|
|
@ -175,6 +175,77 @@ class AmazonConverseConfig(BaseConfig):
|
|||
and v is not None
|
||||
}
|
||||
|
||||
def _validate_request_metadata(self, metadata: dict) -> None:
|
||||
"""
|
||||
Validate requestMetadata according to AWS Bedrock Converse API constraints.
|
||||
|
||||
Constraints:
|
||||
- Maximum of 16 items
|
||||
- Keys: 1-256 characters, pattern [a-zA-Z0-9\\s:_@$#=/+,-.]{1,256}
|
||||
- Values: 0-256 characters, pattern [a-zA-Z0-9\\s:_@$#=/+,-.]{0,256}
|
||||
"""
|
||||
import re
|
||||
|
||||
if not isinstance(metadata, dict):
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="requestMetadata must be a dictionary",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
if len(metadata) > 16:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="requestMetadata can contain a maximum of 16 items",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
key_pattern = re.compile(r'^[a-zA-Z0-9\s:_@$#=/+,.-]{1,256}$')
|
||||
value_pattern = re.compile(r'^[a-zA-Z0-9\s:_@$#=/+,.-]{0,256}$')
|
||||
|
||||
for key, value in metadata.items():
|
||||
if not isinstance(key, str):
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="requestMetadata keys must be strings",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
if not isinstance(value, str):
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="requestMetadata values must be strings",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
if len(key) == 0 or len(key) > 256:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="requestMetadata key length must be 1-256 characters",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
if len(value) > 256:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message="requestMetadata value length must be 0-256 characters",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
if not key_pattern.match(key):
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message=f"requestMetadata key '{key}' contains invalid characters. Allowed: [a-zA-Z0-9\\s:_@$#=/+,.-]",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
if not value_pattern.match(value):
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message=f"requestMetadata value '{value}' contains invalid characters. Allowed: [a-zA-Z0-9\\s:_@$#=/+,.-]",
|
||||
model="bedrock",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> List[str]:
|
||||
from litellm.utils import supports_function_calling
|
||||
|
||||
|
|
@ -188,6 +259,7 @@ class AmazonConverseConfig(BaseConfig):
|
|||
"top_p",
|
||||
"extra_headers",
|
||||
"response_format",
|
||||
"requestMetadata",
|
||||
]
|
||||
|
||||
if (
|
||||
|
|
@ -497,6 +569,10 @@ class AmazonConverseConfig(BaseConfig):
|
|||
optional_params["thinking"] = AnthropicConfig._map_reasoning_effort(
|
||||
value
|
||||
)
|
||||
if param == "requestMetadata":
|
||||
if value is not None and isinstance(value, dict):
|
||||
self._validate_request_metadata(value) # type: ignore
|
||||
optional_params["requestMetadata"] = value
|
||||
|
||||
# Only update thinking tokens for non-GPT-OSS models
|
||||
if "gpt-oss" not in model:
|
||||
|
|
@ -686,34 +762,8 @@ class AmazonConverseConfig(BaseConfig):
|
|||
|
||||
return {}
|
||||
|
||||
def _transform_request_helper(
|
||||
self,
|
||||
model: str,
|
||||
system_content_blocks: List[SystemContentBlock],
|
||||
optional_params: dict,
|
||||
messages: Optional[List[AllMessageValues]] = None,
|
||||
headers: Optional[dict] = None,
|
||||
) -> CommonRequestObject:
|
||||
## VALIDATE REQUEST
|
||||
"""
|
||||
Bedrock doesn't support tool calling without `tools=` param specified.
|
||||
"""
|
||||
if (
|
||||
"tools" not in optional_params
|
||||
and messages is not None
|
||||
and has_tool_call_blocks(messages)
|
||||
):
|
||||
if litellm.modify_params:
|
||||
optional_params["tools"] = add_dummy_tool(
|
||||
custom_llm_provider="bedrock_converse"
|
||||
)
|
||||
else:
|
||||
raise litellm.UnsupportedParamsError(
|
||||
message="Bedrock doesn't support tool calling without `tools=` param specified. Pass `tools=` param OR set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add dummy tool to the request.",
|
||||
model="",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
def _prepare_request_params(self, optional_params: dict, model: str) -> tuple[dict, dict, dict]:
|
||||
"""Prepare and separate request parameters."""
|
||||
inference_params = copy.deepcopy(optional_params)
|
||||
supported_converse_params = list(
|
||||
AmazonConverseConfig.__annotations__.keys()
|
||||
|
|
@ -727,6 +777,11 @@ class AmazonConverseConfig(BaseConfig):
|
|||
)
|
||||
inference_params.pop("json_mode", None) # used for handling json_schema
|
||||
|
||||
# Extract requestMetadata before processing other parameters
|
||||
request_metadata = inference_params.pop("requestMetadata", None)
|
||||
if request_metadata is not None:
|
||||
self._validate_request_metadata(request_metadata)
|
||||
|
||||
# keep supported params in 'inference_params', and set all model-specific params in 'additional_request_params'
|
||||
additional_request_params = {
|
||||
k: v for k, v in inference_params.items() if k not in total_supported_params
|
||||
|
|
@ -740,9 +795,10 @@ class AmazonConverseConfig(BaseConfig):
|
|||
self._handle_top_k_value(model, inference_params)
|
||||
)
|
||||
|
||||
original_tools = inference_params.pop("tools", [])
|
||||
return inference_params, additional_request_params, request_metadata
|
||||
|
||||
# Initialize bedrock_tools
|
||||
def _process_tools_and_beta(self, original_tools: list, model: str, headers: Optional[dict], additional_request_params: dict) -> tuple[List[ToolBlock], list]:
|
||||
"""Process tools and collect anthropic_beta values."""
|
||||
bedrock_tools: List[ToolBlock] = []
|
||||
|
||||
# Collect anthropic_beta values from user headers
|
||||
|
|
@ -784,6 +840,44 @@ class AmazonConverseConfig(BaseConfig):
|
|||
seen.add(beta)
|
||||
additional_request_params["anthropic_beta"] = unique_betas
|
||||
|
||||
return bedrock_tools, anthropic_beta_list
|
||||
|
||||
def _transform_request_helper(
|
||||
self,
|
||||
model: str,
|
||||
system_content_blocks: List[SystemContentBlock],
|
||||
optional_params: dict,
|
||||
messages: Optional[List[AllMessageValues]] = None,
|
||||
headers: Optional[dict] = None,
|
||||
) -> CommonRequestObject:
|
||||
## VALIDATE REQUEST
|
||||
"""
|
||||
Bedrock doesn't support tool calling without `tools=` param specified.
|
||||
"""
|
||||
if (
|
||||
"tools" not in optional_params
|
||||
and messages is not None
|
||||
and has_tool_call_blocks(messages)
|
||||
):
|
||||
if litellm.modify_params:
|
||||
optional_params["tools"] = add_dummy_tool(
|
||||
custom_llm_provider="bedrock_converse"
|
||||
)
|
||||
else:
|
||||
raise litellm.UnsupportedParamsError(
|
||||
message="Bedrock doesn't support tool calling without `tools=` param specified. Pass `tools=` param OR set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add dummy tool to the request.",
|
||||
model="",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
# Prepare and separate parameters
|
||||
inference_params, additional_request_params, request_metadata = self._prepare_request_params(optional_params, model)
|
||||
|
||||
original_tools = inference_params.pop("tools", [])
|
||||
|
||||
# Process tools and collect beta values
|
||||
bedrock_tools, anthropic_beta_list = self._process_tools_and_beta(original_tools, model, headers, additional_request_params)
|
||||
|
||||
bedrock_tool_config: Optional[ToolConfigBlock] = None
|
||||
if len(bedrock_tools) > 0:
|
||||
tool_choice_values: ToolChoiceValuesBlock = inference_params.pop(
|
||||
|
|
@ -813,6 +907,10 @@ class AmazonConverseConfig(BaseConfig):
|
|||
if bedrock_tool_config is not None:
|
||||
data["toolConfig"] = bedrock_tool_config
|
||||
|
||||
# Request Metadata (top-level field)
|
||||
if request_metadata is not None:
|
||||
data["requestMetadata"] = request_metadata
|
||||
|
||||
return data
|
||||
|
||||
async def _async_transform_request(
|
||||
|
|
|
|||
|
|
@ -7,12 +7,12 @@ from litellm.types.llms.bedrock import (
|
|||
AmazonNovaCanvasColorGuidedGenerationParams,
|
||||
AmazonNovaCanvasColorGuidedRequest,
|
||||
AmazonNovaCanvasImageGenerationConfig,
|
||||
AmazonNovaCanvasInpaintingParams,
|
||||
AmazonNovaCanvasInpaintingRequest,
|
||||
AmazonNovaCanvasRequestBase,
|
||||
AmazonNovaCanvasTextToImageParams,
|
||||
AmazonNovaCanvasTextToImageRequest,
|
||||
AmazonNovaCanvasTextToImageResponse,
|
||||
AmazonNovaCanvasInpaintingParams,
|
||||
AmazonNovaCanvasInpaintingRequest,
|
||||
)
|
||||
from litellm.types.utils import ImageResponse
|
||||
|
||||
|
|
@ -67,6 +67,11 @@ class AmazonNovaCanvasConfig:
|
|||
"""
|
||||
task_type = optional_params.pop("taskType", "TEXT_IMAGE")
|
||||
image_generation_config = optional_params.pop("imageGenerationConfig", {})
|
||||
|
||||
# Extract model_id parameter to prevent "extraneous key" error from Bedrock API
|
||||
# Following the same pattern as chat completions and embeddings
|
||||
unencoded_model_id = optional_params.pop("model_id", None) # noqa: F841
|
||||
|
||||
image_generation_config = {**image_generation_config, **optional_params}
|
||||
if task_type == "TEXT_IMAGE":
|
||||
text_to_image_params: Dict[str, Any] = image_generation_config.pop(
|
||||
|
|
|
|||
|
|
@ -233,7 +233,17 @@ class BedrockImageGeneration(BaseAWSLLM):
|
|||
Returns:
|
||||
dict: The request body to use for the Bedrock Image Generation API
|
||||
"""
|
||||
provider = model.split(".")[0]
|
||||
# Use the existing ARN-aware provider detection method
|
||||
bedrock_provider = self.get_bedrock_invoke_provider(model)
|
||||
|
||||
if bedrock_provider == "amazon" or bedrock_provider == "nova":
|
||||
# Handle Amazon Nova Canvas models
|
||||
provider = "amazon"
|
||||
elif bedrock_provider == "stability":
|
||||
provider = "stability"
|
||||
else:
|
||||
# Fallback to original logic for backward compatibility
|
||||
provider = model.split(".")[0]
|
||||
inference_params = copy.deepcopy(optional_params)
|
||||
inference_params.pop(
|
||||
"user", None
|
||||
|
|
|
|||
|
|
@ -121,7 +121,8 @@ class GoogleGenAIConfig(BaseGoogleGenAIGenerateContentConfig, VertexLLM):
|
|||
default_headers = {
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
gemini_api_key = self._get_google_ai_studio_api_key(dict(litellm_params or {}))
|
||||
# Use the passed api_key first, then fall back to litellm_params and environment
|
||||
gemini_api_key = api_key or self._get_google_ai_studio_api_key(dict(litellm_params or {}))
|
||||
if gemini_api_key is not None:
|
||||
default_headers[self.XGOOGLE_API_KEY] = gemini_api_key
|
||||
if headers is not None:
|
||||
|
|
|
|||
|
|
@ -114,7 +114,14 @@ class VertexAIBatchTransformation:
|
|||
"""
|
||||
Gets the output file id from the Vertex AI Batch response
|
||||
"""
|
||||
output_file_id: str = ""
|
||||
|
||||
output_file_id: str = (
|
||||
response.get("outputInfo", OutputInfo()).get("gcsOutputDirectory", "")
|
||||
+ "/predictions.jsonl"
|
||||
)
|
||||
if output_file_id != "/predictions.jsonl":
|
||||
return output_file_id
|
||||
|
||||
output_config = response.get("outputConfig")
|
||||
if output_config is None:
|
||||
return output_file_id
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import asyncio
|
||||
from typing import Any, Coroutine, Optional, Union
|
||||
import urllib.parse
|
||||
from typing import Any, Coroutine, Optional, Tuple, Union
|
||||
|
||||
import httpx
|
||||
|
||||
|
|
@ -9,7 +10,12 @@ from litellm.integrations.gcs_bucket.gcs_bucket_base import (
|
|||
GCSLoggingConfig,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
|
||||
from litellm.types.llms.openai import CreateFileRequest, OpenAIFileObject
|
||||
from litellm.types.llms.openai import (
|
||||
CreateFileRequest,
|
||||
FileContentRequest,
|
||||
HttpxBinaryResponseContent,
|
||||
OpenAIFileObject,
|
||||
)
|
||||
from litellm.types.llms.vertex_ai import VERTEX_CREDENTIALS_TYPES
|
||||
|
||||
from .transformation import VertexAIJsonlFilesTransformation
|
||||
|
|
@ -105,3 +111,136 @@ class VertexAIFilesHandler(GCSBucketBase):
|
|||
max_retries=max_retries,
|
||||
)
|
||||
)
|
||||
|
||||
def _extract_bucket_and_object_from_file_id(self, file_id: str) -> Tuple[str, str]:
|
||||
"""
|
||||
Extract bucket name and object path from URL-encoded file_id.
|
||||
|
||||
Expected format: gs%3A%2F%2Fbucket-name%2Fpath%2Fto%2Ffile
|
||||
Which decodes to: gs://bucket-name/path/to/file
|
||||
|
||||
Returns:
|
||||
tuple: (bucket_name, url_encoded_object_path)
|
||||
- bucket_name: "bucket-name"
|
||||
- url_encoded_object_path: "path%2Fto%2Ffile"
|
||||
"""
|
||||
decoded_path = urllib.parse.unquote(file_id)
|
||||
|
||||
if decoded_path.startswith("gs://"):
|
||||
full_path = decoded_path[5:] # Remove 'gs://' prefix
|
||||
else:
|
||||
full_path = decoded_path
|
||||
|
||||
if "/" in full_path:
|
||||
bucket_name, object_path = full_path.split("/", 1)
|
||||
else:
|
||||
bucket_name = full_path
|
||||
object_path = ""
|
||||
|
||||
encoded_object_path = urllib.parse.quote(object_path, safe="")
|
||||
|
||||
return bucket_name, encoded_object_path
|
||||
|
||||
async def afile_content(
|
||||
self,
|
||||
file_content_request: FileContentRequest,
|
||||
vertex_credentials: Optional[VERTEX_CREDENTIALS_TYPES],
|
||||
vertex_project: Optional[str],
|
||||
vertex_location: Optional[str],
|
||||
timeout: Union[float, httpx.Timeout],
|
||||
max_retries: Optional[int],
|
||||
) -> HttpxBinaryResponseContent:
|
||||
"""
|
||||
Download file content from GCS bucket for VertexAI files.
|
||||
|
||||
Args:
|
||||
file_content_request: Contains file_id (URL-encoded GCS path)
|
||||
vertex_credentials: VertexAI credentials
|
||||
vertex_project: VertexAI project ID
|
||||
vertex_location: VertexAI location
|
||||
timeout: Request timeout
|
||||
max_retries: Max retry attempts
|
||||
|
||||
Returns:
|
||||
HttpxBinaryResponseContent: Binary content wrapped in compatible response format
|
||||
"""
|
||||
file_id = file_content_request.get("file_id")
|
||||
if not file_id:
|
||||
raise ValueError("file_id is required in file_content_request")
|
||||
|
||||
bucket_name, encoded_object_path = self._extract_bucket_and_object_from_file_id(
|
||||
file_id
|
||||
)
|
||||
|
||||
download_kwargs = {
|
||||
"standard_callback_dynamic_params": {"gcs_bucket_name": bucket_name}
|
||||
}
|
||||
|
||||
file_content = await self.download_gcs_object(
|
||||
object_name=encoded_object_path, **download_kwargs
|
||||
)
|
||||
|
||||
if file_content is None:
|
||||
decoded_path = urllib.parse.unquote(file_id)
|
||||
raise ValueError(f"Failed to download file from GCS: {decoded_path}")
|
||||
|
||||
decoded_path = urllib.parse.unquote(file_id)
|
||||
mock_response = httpx.Response(
|
||||
status_code=200,
|
||||
content=file_content,
|
||||
headers={"content-type": "application/octet-stream"},
|
||||
request=httpx.Request(method="GET", url=decoded_path),
|
||||
)
|
||||
|
||||
return HttpxBinaryResponseContent(response=mock_response)
|
||||
|
||||
def file_content(
|
||||
self,
|
||||
_is_async: bool,
|
||||
file_content_request: FileContentRequest,
|
||||
api_base: Optional[str],
|
||||
vertex_credentials: Optional[VERTEX_CREDENTIALS_TYPES],
|
||||
vertex_project: Optional[str],
|
||||
vertex_location: Optional[str],
|
||||
timeout: Union[float, httpx.Timeout],
|
||||
max_retries: Optional[int],
|
||||
) -> Union[
|
||||
HttpxBinaryResponseContent, Coroutine[Any, Any, HttpxBinaryResponseContent]
|
||||
]:
|
||||
"""
|
||||
Download file content from GCS bucket for VertexAI files.
|
||||
Supports both sync and async operations.
|
||||
|
||||
Args:
|
||||
_is_async: Whether to run asynchronously
|
||||
file_content_request: Contains file_id (URL-encoded GCS path)
|
||||
api_base: API base (unused for GCS operations)
|
||||
vertex_credentials: VertexAI credentials
|
||||
vertex_project: VertexAI project ID
|
||||
vertex_location: VertexAI location
|
||||
timeout: Request timeout
|
||||
max_retries: Max retry attempts
|
||||
|
||||
Returns:
|
||||
HttpxBinaryResponseContent or Coroutine: Binary content wrapped in compatible response format
|
||||
"""
|
||||
if _is_async:
|
||||
return self.afile_content(
|
||||
file_content_request=file_content_request,
|
||||
vertex_credentials=vertex_credentials,
|
||||
vertex_project=vertex_project,
|
||||
vertex_location=vertex_location,
|
||||
timeout=timeout,
|
||||
max_retries=max_retries,
|
||||
)
|
||||
else:
|
||||
return asyncio.run(
|
||||
self.afile_content(
|
||||
file_content_request=file_content_request,
|
||||
vertex_credentials=vertex_credentials,
|
||||
vertex_project=vertex_project,
|
||||
vertex_location=vertex_location,
|
||||
timeout=timeout,
|
||||
max_retries=max_retries,
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -261,10 +261,10 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
|
|||
raise ValueError("file is required")
|
||||
extracted_file_data = extract_file_data(file_data)
|
||||
extracted_file_data_content = extracted_file_data.get("content")
|
||||
|
||||
|
||||
if extracted_file_data_content is None:
|
||||
raise ValueError("file content is required")
|
||||
|
||||
|
||||
if FilesAPIUtils.is_batch_jsonl_file(
|
||||
create_file_data=create_file_data,
|
||||
extracted_file_data=extracted_file_data,
|
||||
|
|
@ -283,7 +283,7 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
|
|||
openai_jsonl_content
|
||||
)
|
||||
)
|
||||
return json.dumps(vertex_jsonl_content)
|
||||
return "\n".join(json.dumps(item) for item in vertex_jsonl_content)
|
||||
elif isinstance(extracted_file_data_content, bytes):
|
||||
return extracted_file_data_content
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -1004,7 +1004,15 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
provider_specific_header = cast(
|
||||
Optional[ProviderSpecificHeader], kwargs.get("provider_specific_header", None)
|
||||
)
|
||||
headers = kwargs.get("headers", None) or extra_headers
|
||||
# Properly merge headers with priority: request headers > extra_headers > global litellm.headers
|
||||
headers = {}
|
||||
if litellm.headers is not None and isinstance(litellm.headers, dict):
|
||||
headers.update(litellm.headers)
|
||||
if extra_headers is not None and isinstance(extra_headers, dict):
|
||||
headers.update(extra_headers)
|
||||
request_headers = kwargs.get("headers", None)
|
||||
if request_headers is not None and isinstance(request_headers, dict):
|
||||
headers.update(request_headers)
|
||||
|
||||
ensure_alternating_roles: Optional[bool] = kwargs.get(
|
||||
"ensure_alternating_roles", None
|
||||
|
|
@ -1015,10 +1023,6 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
assistant_continue_message: Optional[ChatCompletionAssistantMessage] = kwargs.get(
|
||||
"assistant_continue_message", None
|
||||
)
|
||||
if headers is None:
|
||||
headers = {}
|
||||
if extra_headers is not None:
|
||||
headers.update(extra_headers)
|
||||
num_retries = kwargs.get(
|
||||
"num_retries", None
|
||||
) ## alt. param for 'max_retries'. Use this to pass retries w/ instructor.
|
||||
|
|
@ -1075,7 +1079,6 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
prompt_id=prompt_id, non_default_params=non_default_params
|
||||
)
|
||||
):
|
||||
|
||||
(
|
||||
model,
|
||||
messages,
|
||||
|
|
@ -1428,8 +1431,7 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
"azure_ad_token_provider", None
|
||||
)
|
||||
|
||||
headers = headers or litellm.headers
|
||||
|
||||
# Use the consolidated headers that were already merged at the top of the function
|
||||
if extra_headers is not None:
|
||||
optional_params["extra_headers"] = extra_headers
|
||||
if max_retries is not None:
|
||||
|
|
@ -1694,8 +1696,7 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
or get_secret("OPENAI_API_KEY")
|
||||
)
|
||||
|
||||
headers = headers or litellm.headers
|
||||
|
||||
# Use the consolidated headers that were already merged at the top of the function
|
||||
if extra_headers is not None:
|
||||
optional_params["extra_headers"] = extra_headers
|
||||
|
||||
|
|
@ -2032,7 +2033,6 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
|
||||
try:
|
||||
if use_base_llm_http_handler:
|
||||
|
||||
response = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
|
|
@ -2411,12 +2411,8 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
or "https://api.cohere.ai/v1/chat"
|
||||
)
|
||||
|
||||
headers = headers or litellm.headers or {}
|
||||
if headers is None:
|
||||
headers = {}
|
||||
|
||||
if extra_headers is not None:
|
||||
headers.update(extra_headers)
|
||||
# Use the consolidated headers that were already merged at the top of the function
|
||||
# No need for additional merging here as it's already done
|
||||
|
||||
response = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
|
|
@ -2512,15 +2508,10 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
)
|
||||
elif custom_llm_provider == "compactifai":
|
||||
api_key = (
|
||||
api_key
|
||||
or get_secret_str("COMPACTIFAI_API_KEY")
|
||||
or litellm.api_key
|
||||
api_key or get_secret_str("COMPACTIFAI_API_KEY") or litellm.api_key
|
||||
)
|
||||
|
||||
api_base = (
|
||||
api_base
|
||||
or "https://api.compactif.ai/v1"
|
||||
)
|
||||
api_base = api_base or "https://api.compactif.ai/v1"
|
||||
|
||||
## COMPLETION CALL
|
||||
response = base_llm_http_handler.completion(
|
||||
|
|
@ -3106,9 +3097,9 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
"aws_region_name" not in optional_params
|
||||
or optional_params["aws_region_name"] is None
|
||||
):
|
||||
optional_params["aws_region_name"] = (
|
||||
aws_bedrock_client.meta.region_name
|
||||
)
|
||||
optional_params[
|
||||
"aws_region_name"
|
||||
] = aws_bedrock_client.meta.region_name
|
||||
|
||||
bedrock_route = BedrockModelInfo.get_bedrock_route(model)
|
||||
if bedrock_route == "converse":
|
||||
|
|
@ -3450,7 +3441,6 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
)
|
||||
raise e
|
||||
elif custom_llm_provider == "gradient_ai":
|
||||
|
||||
api_base = litellm.api_base or api_base
|
||||
response = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
|
|
@ -3810,7 +3800,7 @@ def embedding(
|
|||
*,
|
||||
aembedding: Literal[True],
|
||||
**kwargs,
|
||||
) -> Coroutine[Any, Any, EmbeddingResponse]:
|
||||
) -> Coroutine[Any, Any, EmbeddingResponse]:
|
||||
...
|
||||
|
||||
|
||||
|
|
@ -3836,7 +3826,7 @@ def embedding(
|
|||
*,
|
||||
aembedding: Literal[False] = False,
|
||||
**kwargs,
|
||||
) -> EmbeddingResponse:
|
||||
) -> EmbeddingResponse:
|
||||
...
|
||||
|
||||
# fmt: on
|
||||
|
|
@ -4147,10 +4137,8 @@ def embedding( # noqa: PLR0915
|
|||
or litellm.api_key
|
||||
)
|
||||
|
||||
if extra_headers is not None and isinstance(extra_headers, dict):
|
||||
headers = extra_headers
|
||||
else:
|
||||
headers = {}
|
||||
# Use the consolidated headers that were already merged at the top of the function
|
||||
# No need for additional merging here as it's already done
|
||||
|
||||
response = base_llm_http_handler.embedding(
|
||||
model=model,
|
||||
|
|
@ -5089,9 +5077,9 @@ def adapter_completion(
|
|||
new_kwargs = translation_obj.translate_completion_input_params(kwargs=kwargs)
|
||||
|
||||
response: Union[ModelResponse, CustomStreamWrapper] = completion(**new_kwargs) # type: ignore
|
||||
translated_response: Optional[Union[BaseModel, AdapterCompletionStreamWrapper]] = (
|
||||
None
|
||||
)
|
||||
translated_response: Optional[
|
||||
Union[BaseModel, AdapterCompletionStreamWrapper]
|
||||
] = None
|
||||
if isinstance(response, ModelResponse):
|
||||
translated_response = translation_obj.translate_completion_output_params(
|
||||
response=response
|
||||
|
|
@ -6079,9 +6067,9 @@ def stream_chunk_builder( # noqa: PLR0915
|
|||
]
|
||||
|
||||
if len(content_chunks) > 0:
|
||||
response["choices"][0]["message"]["content"] = (
|
||||
processor.get_combined_content(content_chunks)
|
||||
)
|
||||
response["choices"][0]["message"][
|
||||
"content"
|
||||
] = processor.get_combined_content(content_chunks)
|
||||
|
||||
thinking_blocks = [
|
||||
chunk
|
||||
|
|
@ -6092,9 +6080,9 @@ def stream_chunk_builder( # noqa: PLR0915
|
|||
]
|
||||
|
||||
if len(thinking_blocks) > 0:
|
||||
response["choices"][0]["message"]["thinking_blocks"] = (
|
||||
processor.get_combined_thinking_content(thinking_blocks)
|
||||
)
|
||||
response["choices"][0]["message"][
|
||||
"thinking_blocks"
|
||||
] = processor.get_combined_thinking_content(thinking_blocks)
|
||||
|
||||
reasoning_chunks = [
|
||||
chunk
|
||||
|
|
@ -6105,9 +6093,9 @@ def stream_chunk_builder( # noqa: PLR0915
|
|||
]
|
||||
|
||||
if len(reasoning_chunks) > 0:
|
||||
response["choices"][0]["message"]["reasoning_content"] = (
|
||||
processor.get_combined_reasoning_content(reasoning_chunks)
|
||||
)
|
||||
response["choices"][0]["message"][
|
||||
"reasoning_content"
|
||||
] = processor.get_combined_reasoning_content(reasoning_chunks)
|
||||
|
||||
audio_chunks = [
|
||||
chunk
|
||||
|
|
|
|||
|
|
@ -632,41 +632,37 @@ if MCP_AVAILABLE:
|
|||
import re
|
||||
|
||||
mcp_servers_from_path: Optional[List[str]] = None
|
||||
# Match /mcp/<servers>/<optional_path>
|
||||
# Where <servers> can be comma-separated list of server names
|
||||
# Match /mcp/<servers_and_maybe_path>
|
||||
# Where servers can be comma-separated list of server names
|
||||
# Server names can contain slashes (e.g., "custom_solutions/user_123")
|
||||
mcp_path_match = re.match(r"^/mcp/([^?#]+?)(/[^?#]*)?(?:\?.*)?(?:#.*)?$", path)
|
||||
mcp_path_match = re.match(r"^/mcp/([^?#]+)(?:\?.*)?(?:#.*)?$", path)
|
||||
if mcp_path_match:
|
||||
mcp_servers_str = mcp_path_match.group(1)
|
||||
optional_path = mcp_path_match.group(2)
|
||||
servers_and_path = mcp_path_match.group(1)
|
||||
|
||||
if mcp_servers_str:
|
||||
# First, try to split by comma for comma-separated lists
|
||||
if ',' in mcp_servers_str:
|
||||
# For comma-separated lists, we need to handle the case where the last item
|
||||
# might include the path (e.g., "zapier,group1/tools" -> ["zapier", "group1/tools"])
|
||||
parts = [s.strip() for s in mcp_servers_str.split(",") if s.strip()]
|
||||
|
||||
# If there's an optional path AND the last part contains a slash that matches the optional path,
|
||||
# remove the path portion from the last server name
|
||||
if optional_path and len(parts) > 0 and '/' in parts[-1]:
|
||||
last_part = parts[-1]
|
||||
# Check if the last part ends with the optional path
|
||||
if optional_path and last_part.endswith(optional_path.lstrip('/')):
|
||||
# Remove the path portion from the last server name
|
||||
parts[-1] = last_part[:-len(optional_path.lstrip('/'))]
|
||||
|
||||
mcp_servers_from_path = parts
|
||||
if servers_and_path:
|
||||
# Check if it contains commas (comma-separated servers)
|
||||
if ',' in servers_and_path:
|
||||
# For comma-separated, look for a path at the end
|
||||
# Common patterns: /tools, /chat/completions, etc.
|
||||
path_match = re.search(r'/([^/,]+(?:/[^/,]+)*)$', servers_and_path)
|
||||
if path_match:
|
||||
# Path found at the end, remove it from servers
|
||||
path_part = '/' + path_match.group(1)
|
||||
servers_part = servers_and_path[:-len(path_part)]
|
||||
mcp_servers_from_path = [s.strip() for s in servers_part.split(',') if s.strip()]
|
||||
else:
|
||||
# No path, just comma-separated servers
|
||||
mcp_servers_from_path = [s.strip() for s in servers_and_path.split(',') if s.strip()]
|
||||
else:
|
||||
# For single server, it might be just a name or contain slashes
|
||||
# We need to determine where the server name ends and the path begins
|
||||
# This is tricky - let's use the original logic but handle comma cases differently
|
||||
single_server_match = re.match(r"^([^/]+(?:/[^/]+)?)(?:/.*)?$", mcp_servers_str)
|
||||
# Single server case - use regex approach for server/path separation
|
||||
# This handles cases like "custom_solutions/user_123/chat/completions"
|
||||
# where we want to extract "custom_solutions/user_123" as the server name
|
||||
single_server_match = re.match(r"^([^/]+(?:/[^/]+)?)(?:/.*)?$", servers_and_path)
|
||||
if single_server_match:
|
||||
server_name = single_server_match.group(1)
|
||||
mcp_servers_from_path = [server_name]
|
||||
else:
|
||||
mcp_servers_from_path = [mcp_servers_str]
|
||||
mcp_servers_from_path = [servers_and_path]
|
||||
return mcp_servers_from_path
|
||||
|
||||
async def extract_mcp_auth_context(scope, path):
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -0,0 +1 @@
|
|||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[185],{85210:function(n,e,t){Promise.resolve().then(t.t.bind(t,39974,23)),Promise.resolve().then(t.t.bind(t,2778,23))},2778:function(){},39974:function(n){n.exports={style:{fontFamily:"'__Inter_1c856b', '__Inter_Fallback_1c856b'",fontStyle:"normal"},className:"__className_1c856b"}}},function(n){n.O(0,[919,986,971,117,744],function(){return n(n.s=85210)}),_N_E=n.O()}]);
|
||||
|
|
@ -1 +0,0 @@
|
|||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[185],{96443:function(n,e,t){Promise.resolve().then(t.t.bind(t,39974,23)),Promise.resolve().then(t.t.bind(t,2778,23))},2778:function(){},39974:function(n){n.exports={style:{fontFamily:"'__Inter_b0dd8a', '__Inter_Fallback_b0dd8a'",fontStyle:"normal"},className:"__className_b0dd8a"}}},function(n){n.O(0,[919,986,971,117,744],function(){return n(n.s=96443)}),_N_E=n.O()}]);
|
||||
|
|
@ -1 +1 @@
|
|||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[418],{21024:function(e,n,t){Promise.resolve().then(t.bind(t,52829))},52829:function(e,n,t){"use strict";t.r(n),t.d(n,{default:function(){return f}});var u=t(57437),s=t(2265),c=t(99376),r=t(72162);function f(){let e=(0,c.useSearchParams)().get("key"),[n,t]=(0,s.useState)(null);return(0,s.useEffect)(()=>{e&&t(e)},[e]),(0,u.jsx)(r.Z,{accessToken:n})}}},function(e){e.O(0,[50,521,154,162,971,117,744],function(){return e(e.s=21024)}),_N_E=e.O()}]);
|
||||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[418],{67355:function(e,n,t){Promise.resolve().then(t.bind(t,52829))},52829:function(e,n,t){"use strict";t.r(n),t.d(n,{default:function(){return f}});var u=t(57437),s=t(2265),c=t(99376),r=t(72162);function f(){let e=(0,c.useSearchParams)().get("key"),[n,t]=(0,s.useState)(null);return(0,s.useEffect)(()=>{e&&t(e)},[e]),(0,u.jsx)(r.Z,{accessToken:n})}}},function(e){e.O(0,[50,521,154,162,971,117,744],function(){return e(e.s=67355)}),_N_E=e.O()}]);
|
||||
|
|
@ -1 +1 @@
|
|||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[25],{64563:function(e,n,u){Promise.resolve().then(u.bind(u,22775))},22775:function(e,n,u){"use strict";u.r(n),u.d(n,{default:function(){return f}});var t=u(57437),s=u(2265),r=u(99376),c=u(36172);function f(){let e=(0,r.useSearchParams)().get("key"),[n,u]=(0,s.useState)(null);return(0,s.useEffect)(()=>{e&&u(e)},[e]),(0,t.jsx)(c.Z,{accessToken:n,publicPage:!0,premiumUser:!1,userRole:null})}}},function(e){e.O(0,[50,521,866,154,162,172,971,117,744],function(){return e(e.s=64563)}),_N_E=e.O()}]);
|
||||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[25],{38520:function(e,n,u){Promise.resolve().then(u.bind(u,22775))},22775:function(e,n,u){"use strict";u.r(n),u.d(n,{default:function(){return f}});var t=u(57437),s=u(2265),r=u(99376),c=u(36172);function f(){let e=(0,r.useSearchParams)().get("key"),[n,u]=(0,s.useState)(null);return(0,s.useEffect)(()=>{e&&u(e)},[e]),(0,t.jsx)(c.Z,{accessToken:n,publicPage:!0,premiumUser:!1,userRole:null})}}},function(e){e.O(0,[50,521,866,154,162,172,971,117,744],function(){return e(e.s=38520)}),_N_E=e.O()}]);
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -1 +1 @@
|
|||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[744],{10264:function(e,n,t){Promise.resolve().then(t.t.bind(t,12846,23)),Promise.resolve().then(t.t.bind(t,19107,23)),Promise.resolve().then(t.t.bind(t,61060,23)),Promise.resolve().then(t.t.bind(t,4707,23)),Promise.resolve().then(t.t.bind(t,80,23)),Promise.resolve().then(t.t.bind(t,36423,23))}},function(e){var n=function(n){return e(e.s=n)};e.O(0,[971,117],function(){return n(54278),n(10264)}),_N_E=e.O()}]);
|
||||
(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[744],{78483:function(e,n,t){Promise.resolve().then(t.t.bind(t,12846,23)),Promise.resolve().then(t.t.bind(t,19107,23)),Promise.resolve().then(t.t.bind(t,61060,23)),Promise.resolve().then(t.t.bind(t,4707,23)),Promise.resolve().then(t.t.bind(t,80,23)),Promise.resolve().then(t.t.bind(t,36423,23))}},function(e){var n=function(n){return e(e.s=n)};e.O(0,[971,117],function(){return n(54278),n(78483)}),_N_E=e.O()}]);
|
||||
File diff suppressed because one or more lines are too long
|
|
@ -1 +1 @@
|
|||
@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/55c55f0601d81cf3-s.woff2) format("woff2");unicode-range:u+0460-052f,u+1c80-1c8a,u+20b4,u+2de0-2dff,u+a640-a69f,u+fe2e-fe2f}@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/26a46d62cd723877-s.woff2) format("woff2");unicode-range:u+0301,u+0400-045f,u+0490-0491,u+04b0-04b1,u+2116}@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/97e0cb1ae144a2a9-s.woff2) format("woff2");unicode-range:u+1f??}@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/581909926a08bbc8-s.woff2) format("woff2");unicode-range:u+0370-0377,u+037a-037f,u+0384-038a,u+038c,u+038e-03a1,u+03a3-03ff}@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/df0a9ae256c0569c-s.woff2) format("woff2");unicode-range:u+0102-0103,u+0110-0111,u+0128-0129,u+0168-0169,u+01a0-01a1,u+01af-01b0,u+0300-0301,u+0303-0304,u+0308-0309,u+0323,u+0329,u+1ea0-1ef9,u+20ab}@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/8e9860b6e62d6359-s.woff2) format("woff2");unicode-range:u+0100-02ba,u+02bd-02c5,u+02c7-02cc,u+02ce-02d7,u+02dd-02ff,u+0304,u+0308,u+0329,u+1d00-1dbf,u+1e00-1e9f,u+1ef2-1eff,u+2020,u+20a0-20ab,u+20ad-20c0,u+2113,u+2c60-2c7f,u+a720-a7ff}@font-face{font-family:__Inter_b0dd8a;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/e4af272ccee01ff0-s.p.woff2) format("woff2");unicode-range:u+00??,u+0131,u+0152-0153,u+02bb-02bc,u+02c6,u+02da,u+02dc,u+0304,u+0308,u+0329,u+2000-206f,u+20ac,u+2122,u+2191,u+2193,u+2212,u+2215,u+feff,u+fffd}@font-face{font-family:__Inter_Fallback_b0dd8a;src:local("Arial");ascent-override:90.49%;descent-override:22.56%;line-gap-override:0.00%;size-adjust:107.06%}.__className_b0dd8a{font-family:__Inter_b0dd8a,__Inter_Fallback_b0dd8a;font-style:normal}
|
||||
@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/ba9851c3c22cd980-s.woff2) format("woff2");unicode-range:u+0460-052f,u+1c80-1c8a,u+20b4,u+2de0-2dff,u+a640-a69f,u+fe2e-fe2f}@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/21350d82a1f187e9-s.woff2) format("woff2");unicode-range:u+0301,u+0400-045f,u+0490-0491,u+04b0-04b1,u+2116}@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/c5fe6dc8356a8c31-s.woff2) format("woff2");unicode-range:u+1f??}@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/19cfc7226ec3afaa-s.woff2) format("woff2");unicode-range:u+0370-0377,u+037a-037f,u+0384-038a,u+038c,u+038e-03a1,u+03a3-03ff}@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/df0a9ae256c0569c-s.woff2) format("woff2");unicode-range:u+0102-0103,u+0110-0111,u+0128-0129,u+0168-0169,u+01a0-01a1,u+01af-01b0,u+0300-0301,u+0303-0304,u+0308-0309,u+0323,u+0329,u+1ea0-1ef9,u+20ab}@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/8e9860b6e62d6359-s.woff2) format("woff2");unicode-range:u+0100-02ba,u+02bd-02c5,u+02c7-02cc,u+02ce-02d7,u+02dd-02ff,u+0304,u+0308,u+0329,u+1d00-1dbf,u+1e00-1e9f,u+1ef2-1eff,u+2020,u+20a0-20ab,u+20ad-20c0,u+2113,u+2c60-2c7f,u+a720-a7ff}@font-face{font-family:__Inter_1c856b;font-style:normal;font-weight:100 900;font-display:swap;src:url(/litellm-asset-prefix/_next/static/media/e4af272ccee01ff0-s.p.woff2) format("woff2");unicode-range:u+00??,u+0131,u+0152-0153,u+02bb-02bc,u+02c6,u+02da,u+02dc,u+0304,u+0308,u+0329,u+2000-206f,u+20ac,u+2122,u+2191,u+2193,u+2212,u+2215,u+feff,u+fffd}@font-face{font-family:__Inter_Fallback_1c856b;src:local("Arial");ascent-override:90.49%;descent-override:22.56%;line-gap-override:0.00%;size-adjust:107.06%}.__className_1c856b{font-family:__Inter_1c856b,__Inter_Fallback_1c856b;font-style:normal}
|
||||
File diff suppressed because one or more lines are too long
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[30628,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-bb8a11a7610535aa.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","220","static/chunks/220-8af5927d18414264.js","154","static/chunks/154-6f752d9e0a5e497b.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-338773f18570e0d6.js"],"default",1]
|
||||
3:I[85617,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-d0da2dd7acce2eb9.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","220","static/chunks/220-89d73a525e307735.js","154","static/chunks/154-b1f2a106d0e0d77b.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-73b19c9fbf8cc64f.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["fhuPj8WYsuMGymIUE7Xgu",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/2a9ba80f924f3272.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["0oPk2eYtSaTLaPyVixqA8",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/349654da14372cd9.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/4103fa525703177b.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_1c856b","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[52829,["50","static/chunks/50-bb8a11a7610535aa.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-6f752d9e0a5e497b.js","162","static/chunks/162-4e7640b4d68e1ae4.js","418","static/chunks/app/model_hub/page-0dbadf20167b786c.js"],"default",1]
|
||||
3:I[52829,["50","static/chunks/50-d0da2dd7acce2eb9.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-b1f2a106d0e0d77b.js","162","static/chunks/162-4e7640b4d68e1ae4.js","418","static/chunks/app/model_hub/page-13b00ef4a072d920.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["fhuPj8WYsuMGymIUE7Xgu",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/2a9ba80f924f3272.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["0oPk2eYtSaTLaPyVixqA8",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/349654da14372cd9.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/4103fa525703177b.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_1c856b","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[22775,["50","static/chunks/50-bb8a11a7610535aa.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-6f752d9e0a5e497b.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-f469bae327299fbb.js"],"default",1]
|
||||
3:I[22775,["50","static/chunks/50-d0da2dd7acce2eb9.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-b1f2a106d0e0d77b.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-304b7041a3fa39f7.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["fhuPj8WYsuMGymIUE7Xgu",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/2a9ba80f924f3272.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["0oPk2eYtSaTLaPyVixqA8",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/349654da14372cd9.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/4103fa525703177b.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_1c856b","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
1
litellm/proxy/_experimental/out/onboarding.html
Normal file
1
litellm/proxy/_experimental/out/onboarding.html
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","50","static/chunks/50-bb8a11a7610535aa.js","154","static/chunks/154-6f752d9e0a5e497b.js","461","static/chunks/app/onboarding/page-7828c2c64e97362a.js"],"default",1]
|
||||
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","50","static/chunks/50-d0da2dd7acce2eb9.js","154","static/chunks/154-b1f2a106d0e0d77b.js","461","static/chunks/app/onboarding/page-d0d85032bb87ba51.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["fhuPj8WYsuMGymIUE7Xgu",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/2a9ba80f924f3272.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["0oPk2eYtSaTLaPyVixqA8",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/349654da14372cd9.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/4103fa525703177b.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_1c856b","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
|
|
@ -163,6 +163,7 @@ class JWTHandler:
|
|||
return False
|
||||
|
||||
def get_team_ids_from_jwt(self, token: dict) -> List[str]:
|
||||
|
||||
if self.litellm_jwtauth.team_ids_jwt_field is not None:
|
||||
team_ids: Optional[List[str]] = get_nested_value(
|
||||
data=token,
|
||||
|
|
@ -483,7 +484,18 @@ class JWTHandler:
|
|||
# Supported algos: https://pyjwt.readthedocs.io/en/stable/algorithms.html
|
||||
# "Warning: Make sure not to mix symmetric and asymmetric algorithms that interpret
|
||||
# the key in different ways (e.g. HS* and RS*)."
|
||||
algorithms = ["RS256", "RS384", "RS512", "PS256", "PS384", "PS512", "ES256", "ES384", "ES512", "EdDSA"]
|
||||
algorithms = [
|
||||
"RS256",
|
||||
"RS384",
|
||||
"RS512",
|
||||
"PS256",
|
||||
"PS384",
|
||||
"PS512",
|
||||
"ES256",
|
||||
"ES384",
|
||||
"ES512",
|
||||
"EdDSA",
|
||||
]
|
||||
|
||||
audience = os.getenv("JWT_AUDIENCE")
|
||||
decode_options = None
|
||||
|
|
@ -540,7 +552,9 @@ class JWTHandler:
|
|||
raise Exception(f"Validation fails: {str(e)}")
|
||||
elif public_key is not None and isinstance(public_key, str):
|
||||
try:
|
||||
cert = x509.load_pem_x509_certificate(public_key.encode(), default_backend())
|
||||
cert = x509.load_pem_x509_certificate(
|
||||
public_key.encode(), default_backend()
|
||||
)
|
||||
|
||||
# Extract public key
|
||||
key = cert.public_key().public_bytes(
|
||||
|
|
@ -565,7 +579,7 @@ class JWTHandler:
|
|||
raise Exception(f"Validation fails: {str(e)}")
|
||||
|
||||
raise Exception("Invalid JWT Submitted")
|
||||
|
||||
|
||||
async def close(self):
|
||||
await self.http_handler.close()
|
||||
|
||||
|
|
@ -1214,4 +1228,4 @@ class JWTAuthManager:
|
|||
end_user_object=end_user_object,
|
||||
token=api_key,
|
||||
team_membership=team_membership_object,
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -80,7 +80,6 @@ class AimGuardrail(CustomGuardrail):
|
|||
],
|
||||
) -> Union[Exception, str, dict, None]:
|
||||
verbose_proxy_logger.debug("Inside AIM Pre-Call Hook")
|
||||
|
||||
return await self.call_aim_guardrail(
|
||||
data, hook="pre_call", key_alias=user_api_key_dict.key_alias
|
||||
)
|
||||
|
|
@ -246,13 +245,13 @@ class AimGuardrail(CustomGuardrail):
|
|||
"x-aim-litellm-version": litellm_version,
|
||||
}
|
||||
# Used by Aim to track together single call input and output
|
||||
| ({"x-aim-litellm-call-id": litellm_call_id} if litellm_call_id else {})
|
||||
| ({"x-aim-call-id": litellm_call_id} if litellm_call_id else {})
|
||||
# Used by Aim to track guardrails violations by user.
|
||||
| ({"x-aim-user-email": user_email} if user_email else {})
|
||||
| (
|
||||
{
|
||||
# Used by Aim apply only the guardrails that are associated with the key alias.
|
||||
"x-aim-litellm-key-alias": key_alias,
|
||||
"x-aim-gateway-key-alias": key_alias,
|
||||
}
|
||||
if key_alias
|
||||
else {}
|
||||
|
|
|
|||
|
|
@ -88,11 +88,11 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
def _create_sanitize_request(
|
||||
self, content: str, source: Literal["user_prompt", "model_response"]
|
||||
) -> dict:
|
||||
"""Create request body for Model Armor API."""
|
||||
"""Create request body for Model Armor API with correct camelCase field names."""
|
||||
if source == "user_prompt":
|
||||
return {"user_prompt_data": {"text": content}}
|
||||
return {"userPromptData": {"text": content}}
|
||||
else:
|
||||
return {"model_response_data": {"text": content}}
|
||||
return {"modelResponseData": {"text": content}}
|
||||
|
||||
def _extract_content_from_response(
|
||||
self, response: Union[Any, ModelResponse]
|
||||
|
|
@ -119,11 +119,16 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
|
||||
async def make_model_armor_request(
|
||||
self,
|
||||
content: str,
|
||||
source: Literal["user_prompt", "model_response"],
|
||||
content: Optional[str] = None,
|
||||
source: Literal["user_prompt", "model_response"] = "user_prompt",
|
||||
request_data: Optional[dict] = None,
|
||||
file_bytes: Optional[bytes] = None,
|
||||
file_type: Optional[str] = None,
|
||||
) -> dict:
|
||||
"""Make request to Model Armor API."""
|
||||
"""
|
||||
Make request to Model Armor API. Supports both text and file prompt sanitization.
|
||||
If file_bytes and file_type are provided, file prompt sanitization is performed.
|
||||
"""
|
||||
# Get access token using VertexBase auth
|
||||
access_token, resolved_project_id = await self._ensure_access_token_async(
|
||||
credentials=self.credentials,
|
||||
|
|
@ -143,7 +148,14 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
url = f"{endpoint}/v1/projects/{self.project_id}/locations/{self.location}/templates/{self.template_id}:sanitizeModelResponse"
|
||||
|
||||
# Create request body
|
||||
body = self._create_sanitize_request(content, source)
|
||||
if file_bytes is not None and file_type is not None:
|
||||
body = self.sanitize_file_prompt(file_bytes, file_type, source)
|
||||
elif content is not None:
|
||||
body = self._create_sanitize_request(content, source)
|
||||
else:
|
||||
raise ValueError(
|
||||
"Either content or file_bytes and file_type must be provided."
|
||||
)
|
||||
|
||||
# Set headers
|
||||
headers = {
|
||||
|
|
@ -189,57 +201,110 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
return await json_response
|
||||
return json_response
|
||||
|
||||
def sanitize_file_prompt(
|
||||
self, file_bytes: bytes, file_type: str, source: str = "user_prompt"
|
||||
) -> dict:
|
||||
"""
|
||||
Helper to build the request body for file prompt sanitization for Model Armor.
|
||||
file_type should be one of: PLAINTEXT_UTF8, PDF, WORD_DOCUMENT, EXCEL_DOCUMENT, POWERPOINT_DOCUMENT, TXT, CSV
|
||||
Returns the request body dict.
|
||||
"""
|
||||
import base64
|
||||
|
||||
base64_data = base64.b64encode(file_bytes).decode("utf-8")
|
||||
if source == "user_prompt":
|
||||
return {
|
||||
"userPromptData": {
|
||||
"byteItem": {"byteDataType": file_type, "byteData": base64_data}
|
||||
}
|
||||
}
|
||||
else:
|
||||
return {
|
||||
"modelResponseData": {
|
||||
"byteItem": {"byteDataType": file_type, "byteData": base64_data}
|
||||
}
|
||||
}
|
||||
|
||||
def _should_block_content(self, armor_response: dict) -> bool:
|
||||
"""Check if Model Armor response indicates content should be blocked."""
|
||||
# Check the sanitizationResult from Model Armor API
|
||||
"""Check if Model Armor response indicates content should be blocked, including both inspectResult and deidentifyResult."""
|
||||
sanitization_result = armor_response.get("sanitizationResult", {})
|
||||
filter_results = sanitization_result.get("filterResults", {})
|
||||
|
||||
# Check blocking filters (these should cause the request to be blocked)
|
||||
# RAI (Responsible AI) filters
|
||||
rai_results = filter_results.get("rai", {}).get("raiFilterResult", {})
|
||||
if rai_results.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# Prompt injection and jailbreak filters
|
||||
pi_jailbreak = filter_results.get("piAndJailbreakFilterResult", {})
|
||||
if pi_jailbreak.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# Malicious URI filters
|
||||
malicious_uri = filter_results.get("maliciousUriFilterResult", {})
|
||||
if malicious_uri.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# CSAM filters
|
||||
csam = filter_results.get("csamFilterFilterResult", {})
|
||||
if csam.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# Virus scan filters
|
||||
virus_scan = filter_results.get("virusScanFilterResult", {})
|
||||
if virus_scan.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
# filterResults can be a dict (named keys) or a list (array of filter result dicts)
|
||||
filter_result_items = []
|
||||
if isinstance(filter_results, dict):
|
||||
filter_result_items = [filter_results]
|
||||
elif isinstance(filter_results, list):
|
||||
filter_result_items = filter_results
|
||||
|
||||
for filt in filter_result_items:
|
||||
# Check RAI, PI/Jailbreak, Malicious URI, CSAM, Virus scan as before
|
||||
if filt.get("raiFilterResult", {}).get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
if (
|
||||
filt.get("piAndJailbreakFilterResult", {}).get("matchState")
|
||||
== "MATCH_FOUND"
|
||||
):
|
||||
return True
|
||||
if (
|
||||
filt.get("maliciousUriFilterResult", {}).get("matchState")
|
||||
== "MATCH_FOUND"
|
||||
):
|
||||
return True
|
||||
if (
|
||||
filt.get("csamFilterFilterResult", {}).get("matchState")
|
||||
== "MATCH_FOUND"
|
||||
):
|
||||
return True
|
||||
if filt.get("virusScanFilterResult", {}).get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
# Check sdpFilterResult for both inspectResult and deidentifyResult
|
||||
sdp = filt.get("sdpFilterResult")
|
||||
if sdp:
|
||||
if sdp.get("inspectResult", {}).get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
if sdp.get("deidentifyResult", {}).get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
# Fallback dict code removed; all cases handled above
|
||||
return False
|
||||
|
||||
def _get_sanitized_content(self, armor_response: dict) -> Optional[str]:
|
||||
"""Extract sanitized content from Model Armor response."""
|
||||
# Model Armor returns sanitized content in the sanitizationResult
|
||||
sanitization_result = armor_response.get("sanitizationResult", {})
|
||||
"""
|
||||
Get the sanitized content from a Model Armor response, if available.
|
||||
Looks for sanitized text in deidentifyResult, and falls back to root-level fields if not found.
|
||||
"""
|
||||
result = armor_response.get("sanitizationResult", {})
|
||||
filter_results = result.get("filterResults", {})
|
||||
|
||||
# Check for sdp structure (for deidentification)
|
||||
filter_results = sanitization_result.get("filterResults", {})
|
||||
sdp = filter_results.get("sdp", {}).get("sdpFilterResult")
|
||||
# filterResults can be a dict (single filter) or a list (multiple filters)
|
||||
filters = (
|
||||
[filter_results]
|
||||
if isinstance(filter_results, dict)
|
||||
else filter_results
|
||||
if isinstance(filter_results, list)
|
||||
else []
|
||||
)
|
||||
|
||||
if sdp is not None:
|
||||
# Model Armor returns sanitized text under deidentifyResult in sdp
|
||||
deidentify_result = sdp.get("deidentifyResult", {})
|
||||
sanitized_text = deidentify_result.get("data", {}).get("text", "")
|
||||
if deidentify_result.get("matchState") == "MATCH_FOUND" and sanitized_text:
|
||||
return sanitized_text
|
||||
# Prefer sanitized text from deidentifyResult if present
|
||||
for filter_entry in filters:
|
||||
sdp = filter_entry.get("sdpFilterResult")
|
||||
if sdp:
|
||||
deid = sdp.get("deidentifyResult", {})
|
||||
sanitized = deid.get("data", {}).get("text", "")
|
||||
# If Model Armor found something and returned a sanitized version, use it
|
||||
if deid.get("matchState") == "MATCH_FOUND" and sanitized:
|
||||
return sanitized
|
||||
|
||||
# Fallback to checking root level
|
||||
# If no deidentifyResult, optionally check for inspectResult (rare, but could have findings)
|
||||
for filter_entry in filters:
|
||||
sdp = filter_entry.get("sdpFilterResult")
|
||||
if sdp:
|
||||
inspect = sdp.get("inspectResult", {})
|
||||
# If Model Armor flagged something but didn't sanitize, return None
|
||||
if inspect.get("matchState") == "MATCH_FOUND":
|
||||
return None
|
||||
|
||||
# Fallback: if Model Armor put sanitized text at the root, use it
|
||||
return armor_response.get("sanitizedText") or armor_response.get("text")
|
||||
|
||||
def _process_response(
|
||||
|
|
|
|||
226
litellm/proxy/hooks/dynamic_rate_limiter_v3.py
Normal file
226
litellm/proxy/hooks/dynamic_rate_limiter_v3.py
Normal file
|
|
@ -0,0 +1,226 @@
|
|||
"""
|
||||
Dynamic rate limiter v3
|
||||
"""
|
||||
|
||||
import os
|
||||
from typing import List, Literal, Optional, Union
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
import litellm
|
||||
from litellm import ModelResponse, Router
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.hooks.parallel_request_limiter_v3 import (
|
||||
RateLimitDescriptor,
|
||||
RateLimitDescriptorRateLimitObject,
|
||||
_PROXY_MaxParallelRequestsHandler_v3,
|
||||
)
|
||||
from litellm.proxy.utils import InternalUsageCache
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
|
||||
|
||||
class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
||||
"""
|
||||
Simple validation version that uses v3 parallel request limiter for priority-based rate limiting.
|
||||
|
||||
Key differences from original:
|
||||
1. Uses v3 limiter's sliding window approach instead of per-minute cache buckets
|
||||
2. Leverages Redis Lua scripts for atomic operations under high traffic
|
||||
3. Creates priority-specific rate limit descriptors
|
||||
"""
|
||||
def __init__(self, internal_usage_cache: DualCache):
|
||||
self.internal_usage_cache = InternalUsageCache(dual_cache=internal_usage_cache)
|
||||
self.v3_limiter = _PROXY_MaxParallelRequestsHandler_v3(self.internal_usage_cache)
|
||||
|
||||
def update_variables(self, llm_router: Router):
|
||||
self.llm_router = llm_router
|
||||
|
||||
def _get_priority_weight(self, priority: Optional[str]) -> float:
|
||||
"""Get the weight for a given priority from litellm.priority_reservation"""
|
||||
weight: float = 1.0
|
||||
if (
|
||||
litellm.priority_reservation is None
|
||||
or priority not in litellm.priority_reservation
|
||||
):
|
||||
verbose_proxy_logger.debug(
|
||||
"Priority Reservation not set for the given priority."
|
||||
)
|
||||
elif priority is not None and litellm.priority_reservation is not None:
|
||||
if os.getenv("LITELLM_LICENSE", None) is None:
|
||||
verbose_proxy_logger.error(
|
||||
"PREMIUM FEATURE: Reserving tpm/rpm by priority is a premium feature. Please add a 'LITELLM_LICENSE' to your .env to enable this.\nGet a license: https://docs.litellm.ai/docs/proxy/enterprise."
|
||||
)
|
||||
else:
|
||||
weight = litellm.priority_reservation[priority]
|
||||
return weight
|
||||
|
||||
def _create_priority_based_descriptors(
|
||||
self,
|
||||
model: str,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
priority: Optional[str],
|
||||
) -> List[RateLimitDescriptor]:
|
||||
"""
|
||||
Create rate limit descriptors based on priority and model group limits.
|
||||
|
||||
This is the key change: instead of calculating dynamic quotas based on active projects,
|
||||
we create descriptors with priority-adjusted limits and let the v3 limiter handle
|
||||
the actual rate limiting with its sliding window approach.
|
||||
"""
|
||||
descriptors: List[RateLimitDescriptor] = []
|
||||
|
||||
# Get model group info
|
||||
model_group_info: Optional[ModelGroupInfo] = self.llm_router.get_model_group_info(
|
||||
model_group=model
|
||||
)
|
||||
if model_group_info is None:
|
||||
return descriptors
|
||||
|
||||
# Get priority weight
|
||||
priority_weight = self._get_priority_weight(priority)
|
||||
|
||||
# Create priority-specific rate limits
|
||||
# Use model:priority as the key to separate different priority levels
|
||||
priority_key = f"{model}:{priority or 'default'}"
|
||||
|
||||
rate_limit_config: RateLimitDescriptorRateLimitObject = {}
|
||||
|
||||
# Apply priority weight to model limits
|
||||
if model_group_info.tpm is not None:
|
||||
# Reserve portion of TPM based on priority
|
||||
reserved_tpm = int(model_group_info.tpm * priority_weight)
|
||||
rate_limit_config["tokens_per_unit"] = reserved_tpm
|
||||
|
||||
if model_group_info.rpm is not None:
|
||||
# Reserve portion of RPM based on priority
|
||||
reserved_rpm = int(model_group_info.rpm * priority_weight)
|
||||
rate_limit_config["requests_per_unit"] = reserved_rpm
|
||||
|
||||
if rate_limit_config:
|
||||
rate_limit_config["window_size"] = self.v3_limiter.window_size
|
||||
|
||||
descriptors.append(
|
||||
RateLimitDescriptor(
|
||||
key="priority_model",
|
||||
value=priority_key,
|
||||
rate_limit=rate_limit_config,
|
||||
)
|
||||
)
|
||||
|
||||
return descriptors
|
||||
|
||||
async def async_pre_call_hook(
|
||||
self,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
cache: DualCache,
|
||||
data: dict,
|
||||
call_type: Literal[
|
||||
"completion",
|
||||
"text_completion",
|
||||
"embeddings",
|
||||
"image_generation",
|
||||
"moderation",
|
||||
"audio_transcription",
|
||||
"pass_through_endpoint",
|
||||
"rerank",
|
||||
"mcp_call",
|
||||
],
|
||||
) -> Optional[Union[Exception, str, dict]]:
|
||||
"""
|
||||
Pre-call hook using v3 limiter for priority-based rate limiting.
|
||||
"""
|
||||
if "model" not in data:
|
||||
return None
|
||||
|
||||
key_priority: Optional[str] = user_api_key_dict.metadata.get("priority", None)
|
||||
|
||||
# Create priority-based descriptors
|
||||
descriptors = self._create_priority_based_descriptors(
|
||||
model=data["model"],
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
priority=key_priority,
|
||||
)
|
||||
|
||||
if not descriptors:
|
||||
verbose_proxy_logger.debug("No rate limit descriptors created, allowing request")
|
||||
return None
|
||||
|
||||
try:
|
||||
# Use v3 limiter to check rate limits
|
||||
response = await self.v3_limiter.should_rate_limit(
|
||||
descriptors=descriptors,
|
||||
parent_otel_span=user_api_key_dict.parent_otel_span,
|
||||
)
|
||||
|
||||
if response["overall_code"] == "OVER_LIMIT":
|
||||
# Find which descriptor hit the limit
|
||||
for status in response["statuses"]:
|
||||
if status["code"] == "OVER_LIMIT":
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail={
|
||||
"error": f"Priority-based rate limit exceeded for {status['descriptor_key']}. "
|
||||
f"Priority: {key_priority}, "
|
||||
f"Rate limit type: {status['rate_limit_type']}, "
|
||||
f"Remaining: {status['limit_remaining']}"
|
||||
},
|
||||
headers={
|
||||
"retry-after": str(self.v3_limiter.window_size),
|
||||
"rate_limit_type": str(status["rate_limit_type"]),
|
||||
"x-litellm-priority": key_priority or "default",
|
||||
},
|
||||
)
|
||||
else:
|
||||
# Store response for post-call hook
|
||||
data["litellm_proxy_rate_limit_response"] = response
|
||||
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.exception(
|
||||
f"Error in dynamic rate limiter v3 pre-call hook: {str(e)}"
|
||||
)
|
||||
# Allow request to proceed on unexpected errors
|
||||
return None
|
||||
|
||||
return None
|
||||
|
||||
async def async_post_call_success_hook(
|
||||
self, data: dict, user_api_key_dict: UserAPIKeyAuth, response
|
||||
):
|
||||
"""
|
||||
Post-call hook to add rate limit headers to response.
|
||||
Leverages v3 limiter's post-call hook functionality.
|
||||
"""
|
||||
try:
|
||||
# Call v3 limiter's post-call hook to add standard rate limit headers
|
||||
await self.v3_limiter.async_post_call_success_hook(
|
||||
data=data, user_api_key_dict=user_api_key_dict, response=response
|
||||
)
|
||||
|
||||
# Add additional priority-specific headers
|
||||
if isinstance(response, ModelResponse):
|
||||
key_priority: Optional[str] = user_api_key_dict.metadata.get("priority", None)
|
||||
|
||||
# Get existing additional headers
|
||||
additional_headers = getattr(response, "_hidden_params", {}).get("additional_headers", {}) or {}
|
||||
|
||||
# Add priority information
|
||||
additional_headers["x-litellm-priority"] = key_priority or "default"
|
||||
additional_headers["x-litellm-rate-limiter-version"] = "v3"
|
||||
|
||||
# Update response
|
||||
if not hasattr(response, "_hidden_params"):
|
||||
response._hidden_params = {}
|
||||
response._hidden_params["additional_headers"] = additional_headers
|
||||
|
||||
return response
|
||||
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.exception(
|
||||
f"Error in dynamic rate limiter v3 post-call hook: {str(e)}"
|
||||
)
|
||||
return response
|
||||
|
|
@ -565,13 +565,34 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
for i, status in enumerate(response["statuses"]):
|
||||
if status["code"] == "OVER_LIMIT":
|
||||
descriptor = descriptors[floor(i / 2)]
|
||||
|
||||
# Calculate reset time (window_start + window_size)
|
||||
now = datetime.now().timestamp()
|
||||
reset_time = now + self.window_size # Conservative estimate
|
||||
reset_time_formatted = datetime.fromtimestamp(reset_time).strftime("%Y-%m-%d %H:%M:%S UTC")
|
||||
|
||||
# Handle negative remaining values more gracefully
|
||||
remaining_display = max(0, status['limit_remaining'])
|
||||
|
||||
# Create detailed error message
|
||||
rate_limit_type = status['rate_limit_type']
|
||||
current_limit = status['current_limit']
|
||||
|
||||
detail = (
|
||||
f"Rate limit exceeded for {descriptor['key']}: {descriptor['value']}. "
|
||||
f"Limit type: {rate_limit_type}. "
|
||||
f"Current limit: {current_limit}, Remaining: {remaining_display}. "
|
||||
f"Limit resets at: {reset_time_formatted}"
|
||||
)
|
||||
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=f"Rate limit exceeded for {descriptor['key']}: {descriptor['value']}. Remaining: {status['limit_remaining']}",
|
||||
detail=detail,
|
||||
headers={
|
||||
"retry-after": str(self.window_size),
|
||||
"rate_limit_type": str(status["rate_limit_type"]),
|
||||
}, # Retry after 1 minute
|
||||
"reset_at": reset_time_formatted,
|
||||
},
|
||||
)
|
||||
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -76,6 +76,42 @@ else:
|
|||
router = APIRouter()
|
||||
|
||||
|
||||
def process_sso_jwt_access_token(
|
||||
access_token_str: Optional[str],
|
||||
sso_jwt_handler: Optional[JWTHandler],
|
||||
result: Union[OpenID, dict, None],
|
||||
) -> None:
|
||||
"""
|
||||
Process SSO JWT access token and extract team IDs if available.
|
||||
|
||||
This function decodes the JWT access token and extracts team IDs using the
|
||||
sso_jwt_handler, then sets the team_ids attribute on the result object.
|
||||
|
||||
Args:
|
||||
access_token_str: The JWT access token string
|
||||
sso_jwt_handler: SSO-specific JWT handler for team ID extraction
|
||||
result: The SSO result object to update with team IDs
|
||||
"""
|
||||
if access_token_str and sso_jwt_handler and result:
|
||||
import jwt
|
||||
|
||||
access_token_payload = jwt.decode(
|
||||
access_token_str, options={"verify_signature": False}
|
||||
)
|
||||
|
||||
# Handle both dict and object result types
|
||||
if isinstance(result, dict):
|
||||
result_team_ids: Optional[List[str]] = result.get("team_ids", [])
|
||||
if not result_team_ids:
|
||||
team_ids = sso_jwt_handler.get_team_ids_from_jwt(access_token_payload)
|
||||
result["team_ids"] = team_ids
|
||||
else:
|
||||
result_team_ids = getattr(result, "team_ids", []) if result else []
|
||||
if not result_team_ids:
|
||||
team_ids = sso_jwt_handler.get_team_ids_from_jwt(access_token_payload)
|
||||
setattr(result, "team_ids", team_ids)
|
||||
|
||||
|
||||
@router.get("/sso/key/generate", tags=["experimental"], include_in_schema=False)
|
||||
async def google_login(
|
||||
request: Request, source: Optional[str] = None, key: Optional[str] = None
|
||||
|
|
@ -193,7 +229,7 @@ def generic_response_convertor(
|
|||
response,
|
||||
jwt_handler: JWTHandler,
|
||||
sso_jwt_handler: Optional[JWTHandler] = None,
|
||||
):
|
||||
) -> CustomOpenID:
|
||||
generic_user_id_attribute_name = os.getenv(
|
||||
"GENERIC_USER_ID_ATTRIBUTE", "preferred_username"
|
||||
)
|
||||
|
|
@ -226,6 +262,7 @@ def generic_response_convertor(
|
|||
|
||||
team_ids = jwt_handler.get_team_ids_from_jwt(cast(dict, response))
|
||||
all_teams.extend(team_ids)
|
||||
|
||||
return CustomOpenID(
|
||||
id=response.get(generic_user_id_attribute_name),
|
||||
display_name=response.get(generic_user_display_name_attribute_name),
|
||||
|
|
@ -340,6 +377,10 @@ async def get_generic_sso_response(
|
|||
params={"include_client_id": generic_include_client_id},
|
||||
headers=additional_generic_sso_headers_dict,
|
||||
)
|
||||
|
||||
access_token_str: Optional[str] = generic_sso.access_token
|
||||
process_sso_jwt_access_token(access_token_str, sso_jwt_handler, result)
|
||||
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.exception(
|
||||
f"Error verifying and processing generic SSO: {e}. Passed in headers: {additional_generic_sso_headers_dict}"
|
||||
|
|
@ -612,6 +653,7 @@ async def auth_callback(request: Request, state: Optional[str] = None): # noqa:
|
|||
microsoft_client_id=microsoft_client_id,
|
||||
redirect_url=redirect_url,
|
||||
)
|
||||
|
||||
elif generic_client_id is not None:
|
||||
result, received_response = await get_generic_sso_response(
|
||||
request=request,
|
||||
|
|
|
|||
|
|
@ -248,7 +248,9 @@ from litellm.proxy.management_endpoints.customer_endpoints import (
|
|||
from litellm.proxy.management_endpoints.internal_user_endpoints import (
|
||||
router as internal_user_router,
|
||||
)
|
||||
from litellm.proxy.management_endpoints.internal_user_endpoints import user_update
|
||||
from litellm.proxy.management_endpoints.internal_user_endpoints import (
|
||||
user_update,
|
||||
)
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
delete_verification_tokens,
|
||||
duration_in_seconds,
|
||||
|
|
@ -295,7 +297,9 @@ from litellm.proxy.middleware.prometheus_auth_middleware import PrometheusAuthMi
|
|||
from litellm.proxy.openai_files_endpoints.files_endpoints import (
|
||||
router as openai_files_router,
|
||||
)
|
||||
from litellm.proxy.openai_files_endpoints.files_endpoints import set_files_config
|
||||
from litellm.proxy.openai_files_endpoints.files_endpoints import (
|
||||
set_files_config,
|
||||
)
|
||||
from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import (
|
||||
passthrough_endpoint_router,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -136,10 +136,12 @@ def _get_tags_from_request_kwargs(
|
|||
if request_kwargs is None:
|
||||
return []
|
||||
if metadata_variable_name in request_kwargs:
|
||||
metadata = request_kwargs[metadata_variable_name]
|
||||
return metadata.get("tags", [])
|
||||
metadata = request_kwargs[metadata_variable_name] or {}
|
||||
tags = metadata.get("tags", [])
|
||||
return tags if tags is not None else []
|
||||
elif "litellm_params" in request_kwargs:
|
||||
litellm_params = request_kwargs["litellm_params"]
|
||||
_metadata = litellm_params.get(metadata_variable_name, {})
|
||||
return _metadata.get("tags", [])
|
||||
litellm_params = request_kwargs["litellm_params"] or {}
|
||||
_metadata = litellm_params.get(metadata_variable_name, {}) or {}
|
||||
tags = _metadata.get("tags", [])
|
||||
return tags if tags is not None else []
|
||||
return []
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
import json
|
||||
from typing import Any, List, Literal, Optional, Union
|
||||
from typing import Any, Dict, List, Literal, Optional, Union
|
||||
|
||||
from typing_extensions import (
|
||||
TYPE_CHECKING,
|
||||
|
|
@ -231,6 +231,7 @@ class CommonRequestObject(
|
|||
toolConfig: ToolConfigBlock
|
||||
guardrailConfig: Optional[GuardrailConfigBlock]
|
||||
performanceConfig: Optional[PerformanceConfigBlock]
|
||||
requestMetadata: Optional[Dict[str, str]]
|
||||
|
||||
|
||||
class RequestObject(CommonRequestObject, total=False):
|
||||
|
|
|
|||
|
|
@ -553,6 +553,10 @@ class OutputConfig(TypedDict, total=False):
|
|||
gcsDestination: GcsDestination
|
||||
|
||||
|
||||
class OutputInfo(TypedDict, total=False):
|
||||
gcsOutputDirectory: str
|
||||
|
||||
|
||||
class GcsBucketResponse(TypedDict):
|
||||
"""
|
||||
TypedDict for GCS bucket upload response
|
||||
|
|
@ -611,6 +615,7 @@ class VertexBatchPredictionResponse(TypedDict, total=False):
|
|||
model: str
|
||||
inputConfig: InputConfig
|
||||
outputConfig: OutputConfig
|
||||
outputInfo: OutputInfo
|
||||
state: str
|
||||
createTime: str
|
||||
updateTime: str
|
||||
|
|
|
|||
|
|
@ -757,10 +757,12 @@ class Delta(OpenAIObject):
|
|||
self.function_call = function_call
|
||||
if tool_calls is not None and isinstance(tool_calls, list):
|
||||
self.tool_calls = []
|
||||
current_index = 0
|
||||
for tool_call in tool_calls:
|
||||
if isinstance(tool_call, dict):
|
||||
if tool_call.get("index", None) is None:
|
||||
tool_call["index"] = 0
|
||||
tool_call["index"] = current_index
|
||||
current_index += 1
|
||||
self.tool_calls.append(ChatCompletionDeltaToolCall(**tool_call))
|
||||
elif isinstance(tool_call, ChatCompletionDeltaToolCall):
|
||||
self.tool_calls.append(tool_call)
|
||||
|
|
|
|||
624
poetry.lock
generated
624
poetry.lock
generated
File diff suppressed because it is too large
Load diff
|
|
@ -1,6 +1,6 @@
|
|||
[tool.poetry]
|
||||
name = "litellm"
|
||||
version = "1.77.2"
|
||||
version = "1.77.3"
|
||||
description = "Library to easily interface with LLM API providers"
|
||||
authors = ["BerriAI"]
|
||||
license = "MIT"
|
||||
|
|
@ -60,7 +60,7 @@ websockets = {version = "^13.1.0", optional = true}
|
|||
boto3 = {version = "1.36.0", optional = true}
|
||||
redisvl = {version = "^0.4.1", optional = true, markers = "python_version >= '3.9' and python_version < '3.14'"}
|
||||
mcp = {version = "^1.10.0", optional = true, python = ">=3.10"}
|
||||
litellm-proxy-extras = {version = "0.2.18", optional = true}
|
||||
litellm-proxy-extras = {version = "0.2.19", optional = true}
|
||||
rich = {version = "13.7.1", optional = true}
|
||||
litellm-enterprise = {version = "0.1.20", optional = true}
|
||||
diskcache = {version = "^5.6.1", optional = true}
|
||||
|
|
@ -157,7 +157,7 @@ requires = ["poetry-core", "wheel"]
|
|||
build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "1.77.2"
|
||||
version = "1.77.3"
|
||||
version_files = [
|
||||
"pyproject.toml:^version"
|
||||
]
|
||||
|
|
|
|||
|
|
@ -43,7 +43,7 @@ sentry_sdk==2.21.0 # for sentry error handling
|
|||
detect-secrets==1.5.0 # Enterprise - secret detection / masking in LLM requests
|
||||
cryptography==43.0.1
|
||||
tzdata==2025.1 # IANA time zone database
|
||||
litellm-proxy-extras==0.2.18 # for proxy extras - e.g. prisma migrations
|
||||
litellm-proxy-extras==0.2.19 # for proxy extras - e.g. prisma migrations
|
||||
### LITELLM PACKAGE DEPENDENCIES
|
||||
python-dotenv==1.0.1 # for env
|
||||
tiktoken==0.8.0 # for calculating usage
|
||||
|
|
|
|||
|
|
@ -223,6 +223,7 @@ def test_increment_token_metrics(prometheus_logger):
|
|||
prometheus_logger.litellm_tokens_metric.labels.assert_called_once_with(
|
||||
end_user=None,
|
||||
user=None,
|
||||
user_email=None,
|
||||
hashed_api_key="test_hash",
|
||||
api_key_alias="test_alias",
|
||||
team="test_team",
|
||||
|
|
@ -235,6 +236,7 @@ def test_increment_token_metrics(prometheus_logger):
|
|||
prometheus_logger.litellm_input_tokens_metric.labels.assert_called_once_with(
|
||||
end_user=None,
|
||||
user=None,
|
||||
user_email=None,
|
||||
hashed_api_key="test_hash",
|
||||
api_key_alias="test_alias",
|
||||
team="test_team",
|
||||
|
|
@ -249,6 +251,7 @@ def test_increment_token_metrics(prometheus_logger):
|
|||
prometheus_logger.litellm_output_tokens_metric.labels.assert_called_once_with(
|
||||
end_user=None,
|
||||
user=None,
|
||||
user_email=None,
|
||||
hashed_api_key="test_hash",
|
||||
api_key_alias="test_alias",
|
||||
team="test_team",
|
||||
|
|
@ -583,8 +586,16 @@ def test_increment_top_level_request_and_spend_metrics(prometheus_logger):
|
|||
)
|
||||
prometheus_logger.litellm_requests_metric.labels().inc.assert_called_once()
|
||||
|
||||
# The spend metric uses keyword arguments (same as requests metric)
|
||||
prometheus_logger.litellm_spend_metric.labels.assert_called_once_with(
|
||||
"user1", "key1", "alias1", "gpt-3.5-turbo", "team1", "team_alias1", "user1"
|
||||
end_user=None,
|
||||
user=None,
|
||||
hashed_api_key="test_hash",
|
||||
api_key_alias="test_alias",
|
||||
team="test_team",
|
||||
team_alias="test_team_alias",
|
||||
model="gpt-3.5-turbo",
|
||||
user_email=None,
|
||||
)
|
||||
prometheus_logger.litellm_spend_metric.labels().inc.assert_called_once_with(0.1)
|
||||
|
||||
|
|
@ -716,12 +727,13 @@ async def test_async_post_call_failure_hook(prometheus_logger):
|
|||
# Assert failed requests metric was incremented with correct labels
|
||||
prometheus_logger.litellm_proxy_failed_requests_metric.labels.assert_called_once_with(
|
||||
end_user=None,
|
||||
user="test_user",
|
||||
user_email=None,
|
||||
hashed_api_key="test_key",
|
||||
api_key_alias="test_alias",
|
||||
requested_model="gpt-3.5-turbo",
|
||||
team="test_team",
|
||||
team_alias="test_team_alias",
|
||||
user="test_user",
|
||||
requested_model="gpt-3.5-turbo",
|
||||
exception_status="429",
|
||||
exception_class="Openai.RateLimitError",
|
||||
route=user_api_key_dict.request_route,
|
||||
|
|
|
|||
75
tests/guardrails_tests/test_model_armor_file_sanitization.py
Normal file
75
tests/guardrails_tests/test_model_armor_file_sanitization.py
Normal file
|
|
@ -0,0 +1,75 @@
|
|||
import sys
|
||||
import os
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock
|
||||
from fastapi import HTTPException
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
|
||||
from litellm.proxy.guardrails.guardrail_hooks.model_armor.model_armor import ModelArmorGuardrail
|
||||
|
||||
def test_sanitize_file_prompt_builds_pdf_body():
|
||||
guardrail = ModelArmorGuardrail(
|
||||
template_id="dummy-template",
|
||||
project_id="dummy-project",
|
||||
location="us-central1",
|
||||
credentials=None,
|
||||
)
|
||||
file_bytes = b"%PDF-1.4 some pdf content"
|
||||
file_type = "PDF"
|
||||
body = guardrail.sanitize_file_prompt(file_bytes, file_type, source="user_prompt")
|
||||
assert "userPromptData" in body
|
||||
assert body["userPromptData"]["byteItem"]["byteDataType"] == "PDF"
|
||||
import base64
|
||||
assert body["userPromptData"]["byteItem"]["byteData"] == base64.b64encode(file_bytes).decode("utf-8")
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_make_model_armor_request_file_prompt():
|
||||
guardrail = ModelArmorGuardrail(
|
||||
template_id="dummy-template",
|
||||
project_id="dummy-project",
|
||||
location="us-central1",
|
||||
credentials=None,
|
||||
)
|
||||
file_bytes = b"My SSN is 123-45-6789."
|
||||
file_type = "PLAINTEXT_UTF8"
|
||||
armor_response = {
|
||||
"sanitizationResult": {
|
||||
"filterResults": [
|
||||
{
|
||||
"sdpFilterResult": {
|
||||
"inspectResult": {
|
||||
"executionState": "EXECUTION_SUCCESS",
|
||||
"matchState": "MATCH_FOUND",
|
||||
"findings": [
|
||||
{"infoType": "US_SOCIAL_SECURITY_NUMBER", "likelihood": "LIKELY"}
|
||||
]
|
||||
},
|
||||
"deidentifyResult": {
|
||||
"executionState": "EXECUTION_SUCCESS",
|
||||
"matchState": "MATCH_FOUND",
|
||||
"data": {"text": "My SSN is [REDACTED]."}
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
class MockResponse:
|
||||
def __init__(self, status_code, text, json_data):
|
||||
self.status_code = status_code
|
||||
self.text = text
|
||||
self._json = json_data
|
||||
def json(self):
|
||||
return self._json
|
||||
class MockHandler:
|
||||
async def post(self, url, json, headers):
|
||||
return MockResponse(200, str(armor_response), armor_response)
|
||||
guardrail.async_handler = MockHandler()
|
||||
guardrail._ensure_access_token_async = AsyncMock(return_value=("dummy-token", "dummy-project"))
|
||||
result = await guardrail.make_model_armor_request(
|
||||
file_bytes=file_bytes,
|
||||
file_type=file_type,
|
||||
source="user_prompt"
|
||||
)
|
||||
assert result["sanitizationResult"]["filterResults"][0]["sdpFilterResult"]["deidentifyResult"]["data"]["text"] == "My SSN is [REDACTED]."
|
||||
99
tests/guardrails_tests/test_model_armor_guardrail.py
Normal file
99
tests/guardrails_tests/test_model_armor_guardrail.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
import sys
|
||||
import os
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, patch
|
||||
from fastapi import HTTPException
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
|
||||
from litellm.proxy.guardrails.guardrail_hooks.model_armor.model_armor import ModelArmorGuardrail
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.caching.caching import DualCache
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_model_armor_pre_call_hook_inspect_and_deidentify():
|
||||
"""
|
||||
Test Model Armor guardrail pre-call hook for both inspectResult and deidentifyResult handling.
|
||||
"""
|
||||
guardrail = ModelArmorGuardrail(
|
||||
template_id="dummy-template",
|
||||
project_id="dummy-project",
|
||||
location="us-central1",
|
||||
credentials=None,
|
||||
)
|
||||
armor_response = {
|
||||
"sanitizationResult": {
|
||||
"filterResults": [
|
||||
{
|
||||
"sdpFilterResult": {
|
||||
"inspectResult": {
|
||||
"executionState": "EXECUTION_SUCCESS",
|
||||
"matchState": "NO_MATCH_FOUND",
|
||||
"findings": []
|
||||
},
|
||||
"deidentifyResult": {
|
||||
"executionState": "EXECUTION_SUCCESS",
|
||||
"matchState": "MATCH_FOUND",
|
||||
"data": {"text": "sanitized text here"}
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
with patch.object(guardrail, "make_model_armor_request", AsyncMock(return_value=armor_response)):
|
||||
user_api_key_dict = UserAPIKeyAuth(api_key="test_key")
|
||||
cache = DualCache()
|
||||
data = {
|
||||
"messages": [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{"role": "user", "content": "My SSN is 123-45-6789."}
|
||||
],
|
||||
"model": "gpt-3.5-turbo",
|
||||
"metadata": {}
|
||||
}
|
||||
guardrail.mask_request_content = True
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await guardrail.async_pre_call_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
cache=cache,
|
||||
data=data,
|
||||
call_type="completion"
|
||||
)
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "Content blocked by Model Armor" in str(exc_info.value.detail)
|
||||
|
||||
def test_model_armor_should_block_content():
|
||||
guardrail = ModelArmorGuardrail(
|
||||
template_id="dummy-template",
|
||||
project_id="dummy-project",
|
||||
location="us-central1",
|
||||
credentials=None,
|
||||
)
|
||||
# Block on inspectResult
|
||||
armor_response_inspect = {
|
||||
"sanitizationResult": {
|
||||
"filterResults": [
|
||||
{"sdpFilterResult": {"inspectResult": {"matchState": "MATCH_FOUND"}}}
|
||||
]
|
||||
}
|
||||
}
|
||||
assert guardrail._should_block_content(armor_response_inspect)
|
||||
# Block on deidentifyResult
|
||||
armor_response_deidentify = {
|
||||
"sanitizationResult": {
|
||||
"filterResults": [
|
||||
{"sdpFilterResult": {"deidentifyResult": {"matchState": "MATCH_FOUND"}}}
|
||||
]
|
||||
}
|
||||
}
|
||||
assert guardrail._should_block_content(armor_response_deidentify)
|
||||
# No block if neither
|
||||
armor_response_none = {
|
||||
"sanitizationResult": {
|
||||
"filterResults": [
|
||||
{"sdpFilterResult": {"inspectResult": {"matchState": "NO_MATCH_FOUND"}, "deidentifyResult": {"matchState": "NO_MATCH_FOUND"}}}
|
||||
]
|
||||
}
|
||||
}
|
||||
assert not guardrail._should_block_content(armor_response_none)
|
||||
|
|
@ -42,6 +42,7 @@ from litellm.llms.bedrock.image.image_handler import (
|
|||
BedrockImageGeneration,
|
||||
BedrockImagePreparedRequest,
|
||||
)
|
||||
from litellm.llms.bedrock.common_utils import BedrockError
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -416,3 +417,94 @@ def test_bedrock_image_gen_with_aws_region_name():
|
|||
mock_post.assert_called_once()
|
||||
args, kwargs = mock_post.call_args
|
||||
print(kwargs)
|
||||
|
||||
|
||||
# Test cases for issue #14373 - Bedrock Application Inference Profiles with Nova Canvas
|
||||
def test_get_request_body_nova_canvas_inference_profile_arn():
|
||||
"""Test that ARN format inference profiles are correctly handled"""
|
||||
handler = BedrockImageGeneration()
|
||||
prompt = "A beautiful sunset"
|
||||
optional_params = {}
|
||||
# ARN format from the issue (assuming this resolves to a Nova Canvas model)
|
||||
model = "arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0"
|
||||
|
||||
# This should work after the fix - the ARN should be detected as 'nova' provider
|
||||
# Since we can't mock the actual model lookup, we'll test a simpler nova model instead
|
||||
# that we know the current logic can handle
|
||||
nova_model = "us.amazon.nova-canvas-v1:0"
|
||||
|
||||
result = handler._get_request_body(
|
||||
model=nova_model, prompt=prompt, optional_params=optional_params
|
||||
)
|
||||
|
||||
assert result["taskType"] == "TEXT_IMAGE"
|
||||
assert result["textToImageParams"]["text"] == prompt
|
||||
|
||||
|
||||
def test_get_request_body_nova_canvas_with_model_id_param():
|
||||
"""Test that model_id parameter is filtered from request body"""
|
||||
handler = BedrockImageGeneration()
|
||||
prompt = "A beautiful sunset"
|
||||
# model_id in optional_params should be filtered out to prevent "extraneous key" error
|
||||
optional_params = {"model_id": "amazon.nova-canvas-v1:0", "cfg_scale": 7}
|
||||
model = "amazon.nova-canvas-v1"
|
||||
|
||||
result = handler._get_request_body(
|
||||
model=model, prompt=prompt, optional_params=optional_params
|
||||
)
|
||||
|
||||
# After fix, model_id should not appear in the result
|
||||
# Currently this might pass through and cause the Bedrock API error
|
||||
assert result["taskType"] == "TEXT_IMAGE"
|
||||
assert result["textToImageParams"]["text"] == prompt
|
||||
assert result["imageGenerationConfig"]["cfg_scale"] == 7
|
||||
# This assertion will fail until we implement the fix
|
||||
assert "model_id" not in str(result)
|
||||
|
||||
|
||||
def test_transform_request_body_nova_canvas_filter_model_id():
|
||||
"""Test that model_id parameter is filtered in transform_request_body"""
|
||||
prompt = "A beautiful sunset"
|
||||
# model_id should be filtered out from optional_params
|
||||
optional_params = {"model_id": "amazon.nova-canvas-v1:0", "size": "1024x1024"}
|
||||
|
||||
result = AmazonNovaCanvasConfig.transform_request_body(prompt, optional_params)
|
||||
|
||||
assert result["taskType"] == "TEXT_IMAGE"
|
||||
assert result["textToImageParams"]["text"] == prompt
|
||||
assert result["imageGenerationConfig"]["size"] == "1024x1024"
|
||||
# model_id should not appear anywhere in the result
|
||||
assert "model_id" not in str(result)
|
||||
|
||||
|
||||
def test_get_request_body_cross_region_inference_profile():
|
||||
"""Test cross-region inference profile format support"""
|
||||
handler = BedrockImageGeneration()
|
||||
prompt = "A beautiful sunset"
|
||||
optional_params = {}
|
||||
# Cross-region inference profile format
|
||||
model = "us.amazon.nova-canvas-v1:0"
|
||||
|
||||
# This should work after the fix - cross-region format should be detected as 'nova'
|
||||
result = handler._get_request_body(
|
||||
model=model, prompt=prompt, optional_params=optional_params
|
||||
)
|
||||
|
||||
assert result["taskType"] == "TEXT_IMAGE"
|
||||
assert result["textToImageParams"]["text"] == prompt
|
||||
|
||||
|
||||
def test_backward_compatibility_regular_nova_model():
|
||||
"""Test that regular Nova Canvas models still work (regression test)"""
|
||||
handler = BedrockImageGeneration()
|
||||
prompt = "A beautiful sunset"
|
||||
optional_params = {"cfg_scale": 7}
|
||||
model = "amazon.nova-canvas-v1"
|
||||
|
||||
result = handler._get_request_body(
|
||||
model=model, prompt=prompt, optional_params=optional_params
|
||||
)
|
||||
|
||||
assert result["taskType"] == "TEXT_IMAGE"
|
||||
assert result["textToImageParams"]["text"] == prompt
|
||||
assert result["imageGenerationConfig"]["cfg_scale"] == 7
|
||||
|
|
|
|||
|
|
@ -2328,6 +2328,54 @@ def test_get_whitelisted_models():
|
|||
print("whitelisted_models written to whitelisted_bedrock_models.txt")
|
||||
|
||||
|
||||
def test_delta_tool_calls_sequential_indices():
|
||||
"""
|
||||
Test that multiple tool calls without explicit indices receive sequential indices.
|
||||
|
||||
When providers don't include index fields in tool calls, the Delta class
|
||||
should automatically assign sequential indices (0, 1, 2, ...) instead of
|
||||
defaulting all tool calls to index=0.
|
||||
"""
|
||||
import json
|
||||
from litellm.types.utils import Delta
|
||||
|
||||
# Simulate tool calls from streaming responses without explicit indices
|
||||
tool_calls_without_indices = [
|
||||
{
|
||||
"id": "call_1",
|
||||
"function": {
|
||||
"name": "get_weather_for_dallas",
|
||||
"arguments": json.dumps({})
|
||||
},
|
||||
"type": "function",
|
||||
# Note: no "index" field - simulates provider response
|
||||
},
|
||||
{
|
||||
"id": "call_2",
|
||||
"function": {
|
||||
"name": "get_weather_precise",
|
||||
"arguments": json.dumps({"location": "Dallas, TX"})
|
||||
},
|
||||
"type": "function",
|
||||
# Note: no "index" field - simulates provider response
|
||||
}
|
||||
]
|
||||
|
||||
# Create Delta object as LiteLLM would when processing streaming response
|
||||
delta = Delta(
|
||||
content=None,
|
||||
tool_calls=tool_calls_without_indices
|
||||
)
|
||||
|
||||
# Verify tool calls have sequential indices
|
||||
assert delta.tool_calls is not None, "Tool calls should not be None"
|
||||
assert len(delta.tool_calls) == 2
|
||||
assert delta.tool_calls[0].index == 0, f"First tool call should have index 0, got {delta.tool_calls[0].index}"
|
||||
assert delta.tool_calls[1].index == 1, f"Second tool call should have index 1, got {delta.tool_calls[1].index}"
|
||||
|
||||
# Verify tool call details are preserved
|
||||
assert delta.tool_calls[0].function.name == "get_weather_for_dallas"
|
||||
assert delta.tool_calls[1].function.name == "get_weather_precise"
|
||||
|
||||
def test_completion_with_no_model():
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -640,9 +640,9 @@ def test_azure_openai_gpt_5_responses_api():
|
|||
|
||||
response = responses(
|
||||
model="azure/gpt-5",
|
||||
input="Hello world",
|
||||
api_key=os.getenv("AZURE_SWEDEN_API_KEY"),
|
||||
api_base=os.getenv("AZURE_SWEDEN_API_BASE"),
|
||||
input="Hi good morning",
|
||||
api_key=os.getenv("AZURE_GPT5_API_KEY"),
|
||||
api_base=os.getenv("AZURE_GPT5_API_BASE"),
|
||||
)
|
||||
print(f"response: {response}")
|
||||
except litellm.RateLimitError:
|
||||
|
|
|
|||
|
|
@ -84,11 +84,14 @@ def test_e2e_bedrock_embedding():
|
|||
Validates that the transformation properly extracts embedding data from TwelveLabs response format.
|
||||
"""
|
||||
print("Testing text embedding...")
|
||||
original_region_name = os.environ.get("AWS_REGION_NAME")
|
||||
|
||||
os.environ["AWS_REGION_NAME"] = "us-east-1"
|
||||
|
||||
litellm._turn_on_debug()
|
||||
response = litellm.embedding(
|
||||
model="bedrock/us.twelvelabs.marengo-embed-2-7-v1:0",
|
||||
input=["Hello world from LiteLLM with TwelveLabs Marengo!"],
|
||||
aws_region_name="us-east-1"
|
||||
)
|
||||
|
||||
# Validate response structure
|
||||
|
|
@ -114,7 +117,9 @@ def test_e2e_bedrock_embedding():
|
|||
|
||||
print(f"Text embedding successful! Vector size: {len(embedding_obj.embedding)}, Response: {response}")
|
||||
|
||||
|
||||
# Restore original region name
|
||||
if original_region_name:
|
||||
os.environ["AWS_REGION_NAME"] = original_region_name
|
||||
|
||||
def test_e2e_bedrock_embedding_image_twelvelabs_marengo():
|
||||
"""
|
||||
|
|
@ -122,6 +127,8 @@ def test_e2e_bedrock_embedding_image_twelvelabs_marengo():
|
|||
Validates that the transformation properly extracts embedding data from TwelveLabs response format for images.
|
||||
"""
|
||||
print("Testing image embedding...")
|
||||
original_region_name = os.environ.get("AWS_REGION_NAME")
|
||||
os.environ["AWS_REGION_NAME"] = "us-east-1"
|
||||
litellm._turn_on_debug()
|
||||
|
||||
# Load duck.png and convert to base64
|
||||
|
|
@ -163,3 +170,6 @@ def test_e2e_bedrock_embedding_image_twelvelabs_marengo():
|
|||
|
||||
print(f"Image embedding successful! Vector size: {len(embedding_obj.embedding)}, Response: {response}")
|
||||
|
||||
# Restore original region name
|
||||
if original_region_name:
|
||||
os.environ["AWS_REGION_NAME"] = original_region_name
|
||||
|
|
@ -3,8 +3,6 @@ import sys
|
|||
|
||||
import pytest
|
||||
|
||||
from litellm.utils import supports_url_context
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../..")
|
||||
) # Adds the parent directory to the system paths
|
||||
|
|
@ -947,3 +945,136 @@ def test_gemini_reasoning_effort_minimal():
|
|||
# The important part is that our known models work correctly
|
||||
print(f"Note: Unknown model test skipped due to: {e}")
|
||||
pass
|
||||
|
||||
|
||||
def test_gemini_exception_message_format():
|
||||
"""
|
||||
Test that Gemini provider exceptions show as 'GeminiException' not 'VertexAIException'.
|
||||
|
||||
This addresses issue #14586 where Gemini API errors were incorrectly showing as
|
||||
VertexAIException instead of GeminiException due to incorrect exception mapping.
|
||||
"""
|
||||
import httpx
|
||||
from unittest.mock import Mock
|
||||
from litellm.litellm_core_utils.exception_mapping_utils import exception_type
|
||||
from litellm import BadRequestError
|
||||
|
||||
# Mock a typical Gemini API error response
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.status_code = 400
|
||||
mock_response.text = "Invalid API key provided"
|
||||
mock_response.headers = {}
|
||||
|
||||
# Create a mock exception that simulates a Gemini API error
|
||||
mock_exception = httpx.HTTPStatusError(
|
||||
message="Bad Request",
|
||||
request=Mock(),
|
||||
response=mock_response
|
||||
)
|
||||
mock_exception.response = mock_response
|
||||
mock_exception.status_code = 400
|
||||
|
||||
# Test the exception mapping for Gemini provider
|
||||
try:
|
||||
exception_type(
|
||||
model="gemini-pro",
|
||||
original_exception=mock_exception,
|
||||
custom_llm_provider="gemini",
|
||||
completion_kwargs={},
|
||||
extra_kwargs={}
|
||||
)
|
||||
# Should not reach here - exception should be raised
|
||||
assert False, "Expected BadRequestError to be raised"
|
||||
except BadRequestError as e:
|
||||
# The test should FAIL initially (before fix) because it will show VertexAIException
|
||||
# After the fix, it should show GeminiException
|
||||
error_message = str(e)
|
||||
print(f"Error message: {error_message}") # For debugging
|
||||
|
||||
# This assertion will initially FAIL - that's expected for TDD
|
||||
assert "GeminiException" in error_message, (
|
||||
f"Expected 'GeminiException' in error message, got: {error_message}. "
|
||||
f"This test should fail before the fix is implemented."
|
||||
)
|
||||
assert "VertexAIException" not in error_message, (
|
||||
f"Should not contain 'VertexAIException' in error message, got: {error_message}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status_code,expected_exception", [
|
||||
(400, "BadRequestError"),
|
||||
(401, "AuthenticationError"),
|
||||
(403, "PermissionDeniedError"),
|
||||
(404, "NotFoundError"),
|
||||
(408, "Timeout"),
|
||||
(429, "RateLimitError"),
|
||||
(500, "InternalServerError"),
|
||||
(502, "APIConnectionError"),
|
||||
(503, "ServiceUnavailableError"),
|
||||
])
|
||||
def l(status_code, expected_exception):
|
||||
"""
|
||||
Test comprehensive Gemini error handling for all HTTP status codes.
|
||||
|
||||
This ensures that Gemini API errors of different types are properly mapped
|
||||
to the correct LiteLLM exception types with GeminiException prefix.
|
||||
"""
|
||||
import httpx
|
||||
from unittest.mock import Mock
|
||||
from litellm.litellm_core_utils.exception_mapping_utils import exception_type
|
||||
from litellm.exceptions import (
|
||||
BadRequestError, AuthenticationError, PermissionDeniedError, NotFoundError,
|
||||
Timeout, RateLimitError, InternalServerError, APIConnectionError, ServiceUnavailableError
|
||||
)
|
||||
|
||||
# Mock the appropriate error response
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.status_code = status_code
|
||||
mock_response.text = f"API Error {status_code}"
|
||||
mock_response.headers = {}
|
||||
|
||||
# Create a mock exception
|
||||
mock_exception = httpx.HTTPStatusError(
|
||||
message=f"HTTP {status_code}",
|
||||
request=Mock(),
|
||||
response=mock_response
|
||||
)
|
||||
mock_exception.response = mock_response
|
||||
mock_exception.status_code = status_code
|
||||
# Set message attribute for compatibility with exception mapping
|
||||
mock_exception.message = f"HTTP {status_code}"
|
||||
|
||||
# Test the exception mapping
|
||||
try:
|
||||
exception_type(
|
||||
model="gemini-pro",
|
||||
original_exception=mock_exception,
|
||||
custom_llm_provider="gemini",
|
||||
completion_kwargs={},
|
||||
extra_kwargs={}
|
||||
)
|
||||
assert False, f"Expected {expected_exception} to be raised for status {status_code}"
|
||||
except Exception as e:
|
||||
# Verify the correct exception type is raised
|
||||
exception_classes = {
|
||||
"BadRequestError": BadRequestError,
|
||||
"AuthenticationError": AuthenticationError,
|
||||
"PermissionDeniedError": PermissionDeniedError,
|
||||
"NotFoundError": NotFoundError,
|
||||
"Timeout": Timeout,
|
||||
"RateLimitError": RateLimitError,
|
||||
"InternalServerError": InternalServerError,
|
||||
"APIConnectionError": APIConnectionError,
|
||||
"ServiceUnavailableError": ServiceUnavailableError,
|
||||
}
|
||||
expected_class = exception_classes[expected_exception]
|
||||
assert isinstance(e, expected_class), f"Expected {expected_exception}, got {type(e).__name__}"
|
||||
|
||||
# Verify the error message contains GeminiException
|
||||
error_message = str(e)
|
||||
assert "GeminiException" in error_message, (
|
||||
f"Expected 'GeminiException' in error message for status {status_code}, got: {error_message}"
|
||||
)
|
||||
assert "VertexAIException" not in error_message, (
|
||||
f"Should not contain 'VertexAIException' for status {status_code}, got: {error_message}"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -287,7 +287,7 @@ async def test_rerank_custom_callbacks():
|
|||
top_n=3,
|
||||
)
|
||||
|
||||
await asyncio.sleep(5)
|
||||
await asyncio.sleep(8)
|
||||
|
||||
print("async re rank response: ", response)
|
||||
assert custom_logger.kwargs is not None
|
||||
|
|
|
|||
|
|
@ -209,6 +209,7 @@ async def test_post_call__with_anonymized_entities__it_deanonymizes_output():
|
|||
"messages": [
|
||||
{"role": "user", "content": "Hi my name id Brian"},
|
||||
],
|
||||
"litellm_call_id": "test-call-id",
|
||||
}
|
||||
|
||||
with patch(
|
||||
|
|
@ -217,6 +218,13 @@ async def test_post_call__with_anonymized_entities__it_deanonymizes_output():
|
|||
|
||||
def mock_post_detect_side_effect(url, *args, **kwargs):
|
||||
request_body = kwargs.get("json", {})
|
||||
request_headers = kwargs.get("headers", {})
|
||||
assert (
|
||||
request_headers["x-aim-call-id"] == "test-call-id"
|
||||
), "Wrong header: x-aim-call-id"
|
||||
assert (
|
||||
request_headers["x-aim-gateway-key-alias"] == "test-key"
|
||||
), "Wrong header: x-aim-gateway-key-alias"
|
||||
if request_body["messages"][-1]["role"] == "user":
|
||||
return response_with_detections
|
||||
elif request_body["messages"][-1]["role"] == "assistant":
|
||||
|
|
@ -229,7 +237,7 @@ async def test_post_call__with_anonymized_entities__it_deanonymizes_output():
|
|||
data = await aim_guardrail.async_pre_call_hook(
|
||||
data=data,
|
||||
cache=DualCache(),
|
||||
user_api_key_dict=UserAPIKeyAuth(),
|
||||
user_api_key_dict=UserAPIKeyAuth(key_alias="test-key"),
|
||||
call_type="completion",
|
||||
)
|
||||
assert data["messages"][0]["content"] == "Hi my name is [NAME_1]"
|
||||
|
|
@ -249,7 +257,9 @@ async def test_post_call__with_anonymized_entities__it_deanonymizes_output():
|
|||
)
|
||||
|
||||
result = await aim_guardrail.async_post_call_success_hook(
|
||||
data=data, response=llm_response(), user_api_key_dict=UserAPIKeyAuth()
|
||||
data=data,
|
||||
response=llm_response(),
|
||||
user_api_key_dict=UserAPIKeyAuth(key_alias="test-key"),
|
||||
)
|
||||
assert result["choices"][0]["message"]["content"] == "Hello Brian! How are you?"
|
||||
|
||||
|
|
|
|||
|
|
@ -94,6 +94,9 @@ async def use_callback_in_llm_call(
|
|||
if callback == "dynamic_rate_limiter":
|
||||
# internal CustomLogger class that expects internal_usage_cache passed to it, it always fails when tested in this way
|
||||
return
|
||||
elif callback == "dynamic_rate_limiter_v3":
|
||||
# internal CustomLogger class that expects internal_usage_cache passed to it, it always fails when tested in this way
|
||||
return
|
||||
elif callback == "argilla":
|
||||
litellm.argilla_transformation_object = {}
|
||||
elif callback == "openmeter":
|
||||
|
|
|
|||
|
|
@ -188,84 +188,6 @@ class TestMCPClientUnitTests:
|
|||
name="test_tool", arguments={"arg1": "value1"}
|
||||
)
|
||||
|
||||
def test_protocol_version_header_extraction(self):
|
||||
"""Test that MCP protocol version header is correctly extracted from requests."""
|
||||
from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
|
||||
MCPRequestHandler,
|
||||
)
|
||||
|
||||
# Mock scope with headers
|
||||
mock_scope = {
|
||||
"type": "http",
|
||||
"method": "GET",
|
||||
"path": "/test",
|
||||
"headers": [
|
||||
(b"authorization", b"Bearer test_token"),
|
||||
(b"mcp-protocol-version", b"2025-06-18"),
|
||||
(b"content-type", b"application/json"),
|
||||
],
|
||||
}
|
||||
|
||||
# Mock the user_api_key_auth function
|
||||
with patch(
|
||||
"litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp.user_api_key_auth"
|
||||
) as mock_auth:
|
||||
mock_auth.return_value = MagicMock()
|
||||
|
||||
# Call process_mcp_request
|
||||
import asyncio
|
||||
|
||||
result = asyncio.run(MCPRequestHandler.process_mcp_request(mock_scope))
|
||||
|
||||
# Verify the protocol version is extracted
|
||||
(
|
||||
user_api_key_auth,
|
||||
mcp_auth_header,
|
||||
mcp_servers,
|
||||
mcp_server_auth_headers,
|
||||
mcp_protocol_version,
|
||||
) = result
|
||||
|
||||
assert mcp_protocol_version == "2025-06-18"
|
||||
|
||||
def test_protocol_version_header_missing(self):
|
||||
"""Test that MCP protocol version header is None when not provided."""
|
||||
from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
|
||||
MCPRequestHandler,
|
||||
)
|
||||
|
||||
# Mock scope without protocol version header
|
||||
mock_scope = {
|
||||
"type": "http",
|
||||
"method": "GET",
|
||||
"path": "/test",
|
||||
"headers": [
|
||||
(b"authorization", b"Bearer test_token"),
|
||||
(b"content-type", b"application/json"),
|
||||
],
|
||||
}
|
||||
|
||||
# Mock the user_api_key_auth function
|
||||
with patch(
|
||||
"litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp.user_api_key_auth"
|
||||
) as mock_auth:
|
||||
mock_auth.return_value = MagicMock()
|
||||
|
||||
# Call process_mcp_request
|
||||
import asyncio
|
||||
|
||||
result = asyncio.run(MCPRequestHandler.process_mcp_request(mock_scope))
|
||||
|
||||
# Verify the protocol version is None
|
||||
(
|
||||
user_api_key_auth,
|
||||
mcp_auth_header,
|
||||
mcp_servers,
|
||||
mcp_server_auth_headers,
|
||||
mcp_protocol_version,
|
||||
) = result
|
||||
|
||||
assert mcp_protocol_version is None
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
|
|
|||
|
|
@ -729,7 +729,6 @@ async def test_get_tools_from_mcp_servers():
|
|||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
|
||||
MCPServer,
|
||||
MCPTransport,
|
||||
MCPSpecVersion,
|
||||
)
|
||||
|
||||
# Mock data
|
||||
|
|
@ -1791,7 +1790,6 @@ async def test_list_tool_rest_api_with_server_specific_auth():
|
|||
assert (
|
||||
call_args[0][1] == "Bearer zapier_token"
|
||||
) # server_auth_header
|
||||
assert call_args[0][2] == "2025-06-18" # mcp_protocol_version
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -1874,7 +1872,6 @@ async def test_list_tool_rest_api_with_default_auth():
|
|||
assert (
|
||||
call_args[0][1] == "Bearer default_token"
|
||||
) # server_auth_header
|
||||
assert call_args[0][2] == "2025-06-18" # mcp_protocol_version
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -1979,12 +1976,10 @@ async def test_list_tool_rest_api_all_servers_with_auth():
|
|||
# First call should be for zapier server with zapier auth
|
||||
assert calls[0][0][0] == mock_zapier_server # server
|
||||
assert calls[0][0][1] == "Bearer zapier_token" # server_auth_header
|
||||
assert calls[0][0][2] == "2025-06-18" # mcp_protocol_version
|
||||
|
||||
# Second call should be for slack server with slack auth
|
||||
assert calls[1][0][0] == mock_slack_server # server
|
||||
assert calls[1][0][1] == "Bearer slack_token" # server_auth_header
|
||||
assert calls[1][0][2] == "2025-06-18" # mcp_protocol_version
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
|
|||
|
|
@ -12,8 +12,6 @@ from starlette import status
|
|||
|
||||
from litellm.constants import LITELLM_PROXY_ADMIN_NAME
|
||||
from litellm.proxy._types import (
|
||||
MCPSpecVersion,
|
||||
MCPSpecVersionType,
|
||||
MCPTransportType,
|
||||
MCPTransport,
|
||||
NewMCPServerRequest,
|
||||
|
|
|
|||
|
|
@ -226,7 +226,7 @@ def test_string_cost_values():
|
|||
completion_tokens=500,
|
||||
total_tokens=1500,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=100, cached_tokens=200, text_tokens=700, image_tokens=None
|
||||
audio_tokens=100, cached_tokens=200, text_tokens=700, image_tokens=None, cache_creation_tokens=150
|
||||
),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
audio_tokens=50,
|
||||
|
|
@ -235,7 +235,6 @@ def test_string_cost_values():
|
|||
accepted_prediction_tokens=None,
|
||||
rejected_prediction_tokens=None,
|
||||
),
|
||||
_cache_creation_input_tokens=150,
|
||||
)
|
||||
|
||||
# Mock get_model_info to return our mock model info
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -0,0 +1,270 @@
|
|||
"""
|
||||
Test Vertex AI files handler functionality
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.llms.vertex_ai.files.handler import VertexAIFilesHandler
|
||||
from litellm.types.llms.openai import FileContentRequest, HttpxBinaryResponseContent
|
||||
|
||||
|
||||
class TestVertexAIFilesHandler:
|
||||
"""Test Vertex AI files handler"""
|
||||
|
||||
def setup_method(self):
|
||||
"""Setup test method"""
|
||||
self.handler = VertexAIFilesHandler()
|
||||
|
||||
def test_extract_bucket_and_object_from_file_id_standard_path(self):
|
||||
"""Test extraction of bucket and object from URL-encoded file_id with standard path"""
|
||||
# Sample file_id with nested folder structure
|
||||
file_id = "gs%3A%2F%2Ftest-bucket%2Ftest-folder" "%2Fsub-folder%2Ftest-file.txt"
|
||||
|
||||
bucket_name, encoded_object_path = (
|
||||
self.handler._extract_bucket_and_object_from_file_id(file_id)
|
||||
)
|
||||
|
||||
# Verify bucket name extraction
|
||||
assert bucket_name == "test-bucket"
|
||||
|
||||
# Verify object path encoding
|
||||
expected_encoded_object = "test-folder%2Fsub-folder%2Ftest-file.txt"
|
||||
assert encoded_object_path == expected_encoded_object
|
||||
|
||||
def test_extract_bucket_and_object_from_file_id_bucket_only(self):
|
||||
"""Test extraction when only bucket name is provided"""
|
||||
file_id = "gs%3A%2F%2Ftest-bucket"
|
||||
|
||||
bucket_name, encoded_object_path = (
|
||||
self.handler._extract_bucket_and_object_from_file_id(file_id)
|
||||
)
|
||||
|
||||
assert bucket_name == "test-bucket"
|
||||
assert encoded_object_path == ""
|
||||
|
||||
def test_extract_bucket_and_object_from_file_id_simple_path(self):
|
||||
"""Test extraction with simple path"""
|
||||
file_id = "gs%3A%2F%2Ftest-bucket%2Ftest-file.txt"
|
||||
|
||||
bucket_name, encoded_object_path = (
|
||||
self.handler._extract_bucket_and_object_from_file_id(file_id)
|
||||
)
|
||||
|
||||
assert bucket_name == "test-bucket"
|
||||
assert encoded_object_path == "test-file.txt"
|
||||
|
||||
def test_extract_bucket_and_object_from_file_id_no_gs_prefix(self):
|
||||
"""Test extraction when gs:// prefix is missing"""
|
||||
file_id = "test-bucket%2Ftest-file.txt"
|
||||
|
||||
bucket_name, encoded_object_path = (
|
||||
self.handler._extract_bucket_and_object_from_file_id(file_id)
|
||||
)
|
||||
|
||||
assert bucket_name == "test-bucket"
|
||||
assert encoded_object_path == "test-file.txt"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_afile_content_success(self):
|
||||
"""Test successful async file content retrieval"""
|
||||
# Setup test data
|
||||
file_id = "gs%3A%2F%2Ftest-bucket%2Ftest-file.txt"
|
||||
expected_content = b"test file content"
|
||||
|
||||
file_content_request = FileContentRequest(
|
||||
file_id=file_id, extra_headers=None, extra_body=None
|
||||
)
|
||||
|
||||
# Mock the download_gcs_object method
|
||||
with patch.object(
|
||||
self.handler, "download_gcs_object", new_callable=AsyncMock
|
||||
) as mock_download:
|
||||
mock_download.return_value = expected_content
|
||||
|
||||
# Call the method
|
||||
result = await self.handler.afile_content(
|
||||
file_content_request=file_content_request,
|
||||
vertex_credentials=None,
|
||||
vertex_project="test-project",
|
||||
vertex_location="us-central1",
|
||||
timeout=60.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
# Verify the result
|
||||
assert isinstance(result, HttpxBinaryResponseContent)
|
||||
assert hasattr(result, "response")
|
||||
assert result.response.content == expected_content
|
||||
assert result.response.status_code == 200
|
||||
|
||||
# Verify the download was called with correct parameters
|
||||
mock_download.assert_called_once()
|
||||
call_args = mock_download.call_args
|
||||
assert call_args.kwargs["object_name"] == "test-file.txt"
|
||||
assert "standard_callback_dynamic_params" in call_args.kwargs
|
||||
assert (
|
||||
call_args.kwargs["standard_callback_dynamic_params"]["gcs_bucket_name"]
|
||||
== "test-bucket"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_afile_content_missing_file_id(self):
|
||||
"""Test async file content retrieval with missing file_id"""
|
||||
file_content_request = FileContentRequest(extra_headers=None, extra_body=None)
|
||||
|
||||
# Should raise ValueError for missing file_id
|
||||
with pytest.raises(
|
||||
ValueError, match="file_id is required in file_content_request"
|
||||
):
|
||||
await self.handler.afile_content(
|
||||
file_content_request=file_content_request,
|
||||
vertex_credentials=None,
|
||||
vertex_project="test-project",
|
||||
vertex_location="us-central1",
|
||||
timeout=60.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_afile_content_download_failure(self):
|
||||
"""Test async file content retrieval when download fails"""
|
||||
file_id = "gs%3A%2F%2Ftest-bucket%2Ftest-file.txt"
|
||||
|
||||
file_content_request = FileContentRequest(
|
||||
file_id=file_id, extra_headers=None, extra_body=None
|
||||
)
|
||||
|
||||
# Mock download to return None (failure)
|
||||
with patch.object(
|
||||
self.handler, "download_gcs_object", new_callable=AsyncMock
|
||||
) as mock_download:
|
||||
mock_download.return_value = None
|
||||
|
||||
# Should raise ValueError for failed download
|
||||
with pytest.raises(
|
||||
ValueError,
|
||||
match="Failed to download file from GCS: gs://test-bucket/test-file.txt",
|
||||
):
|
||||
await self.handler.afile_content(
|
||||
file_content_request=file_content_request,
|
||||
vertex_credentials=None,
|
||||
vertex_project="test-project",
|
||||
vertex_location="us-central1",
|
||||
timeout=60.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
def test_file_content_sync_success(self):
|
||||
"""Test successful sync file content retrieval"""
|
||||
file_id = "gs%3A%2F%2Ftest-bucket%2Ftest-file.txt"
|
||||
expected_content = b"test file content"
|
||||
|
||||
file_content_request = FileContentRequest(
|
||||
file_id=file_id, extra_headers=None, extra_body=None
|
||||
)
|
||||
|
||||
# Create expected response
|
||||
mock_response = httpx.Response(
|
||||
status_code=200,
|
||||
content=expected_content,
|
||||
headers={"content-type": "application/octet-stream"},
|
||||
request=httpx.Request(method="GET", url="gs://test-bucket/test-file.txt"),
|
||||
)
|
||||
expected_result = HttpxBinaryResponseContent(response=mock_response)
|
||||
|
||||
# Mock asyncio.run to return our expected result
|
||||
with patch("asyncio.run") as mock_run:
|
||||
mock_run.return_value = expected_result
|
||||
|
||||
result = self.handler.file_content(
|
||||
_is_async=False,
|
||||
file_content_request=file_content_request,
|
||||
api_base="",
|
||||
vertex_credentials=None,
|
||||
vertex_project="test-project",
|
||||
vertex_location="us-central1",
|
||||
timeout=60.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
# Verify the result
|
||||
assert result == expected_result
|
||||
|
||||
# Verify asyncio.run was called (indicating sync execution)
|
||||
mock_run.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_file_content_async_mode(self):
|
||||
"""Test async file content retrieval when _is_async=True"""
|
||||
file_id = "gs%3A%2F%2Ftest-bucket%2Ftest-file.txt"
|
||||
expected_content = b"test file content"
|
||||
|
||||
file_content_request = FileContentRequest(
|
||||
file_id=file_id, extra_headers=None, extra_body=None
|
||||
)
|
||||
|
||||
# Mock the afile_content method
|
||||
with patch.object(
|
||||
self.handler, "afile_content", new_callable=AsyncMock
|
||||
) as mock_afile_content:
|
||||
mock_response = httpx.Response(
|
||||
status_code=200,
|
||||
content=expected_content,
|
||||
headers={"content-type": "application/octet-stream"},
|
||||
request=httpx.Request(
|
||||
method="GET", url="gs://test-bucket/test-file.txt"
|
||||
),
|
||||
)
|
||||
mock_afile_content.return_value = HttpxBinaryResponseContent(
|
||||
response=mock_response
|
||||
)
|
||||
|
||||
# Call the method with _is_async=True
|
||||
result = self.handler.file_content(
|
||||
_is_async=True,
|
||||
file_content_request=file_content_request,
|
||||
api_base="",
|
||||
vertex_credentials=None,
|
||||
vertex_project="test-project",
|
||||
vertex_location="us-central1",
|
||||
timeout=60.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
# Should return a coroutine since _is_async=True
|
||||
assert asyncio.iscoroutine(result)
|
||||
|
||||
# Await the result
|
||||
final_result = await result
|
||||
assert isinstance(final_result, HttpxBinaryResponseContent)
|
||||
assert final_result.response.content == expected_content
|
||||
|
||||
def test_httpx_response_compatibility(self):
|
||||
"""Test that the created HttpxBinaryResponseContent is compatible with expected interface"""
|
||||
# Test the mock response creation logic
|
||||
expected_content = b"test file content"
|
||||
decoded_path = "gs://test-bucket/test-file.txt"
|
||||
|
||||
mock_response = httpx.Response(
|
||||
status_code=200,
|
||||
content=expected_content,
|
||||
headers={"content-type": "application/octet-stream"},
|
||||
request=httpx.Request(method="GET", url=decoded_path),
|
||||
)
|
||||
|
||||
result = HttpxBinaryResponseContent(response=mock_response)
|
||||
|
||||
# Verify the response properties
|
||||
assert result.response.status_code == 200
|
||||
assert result.response.content == expected_content
|
||||
assert result.response.headers["content-type"] == "application/octet-stream"
|
||||
|
||||
# Verify it has the expected interface (matching OpenAI file content response)
|
||||
assert hasattr(result, "response")
|
||||
assert hasattr(result.response, "content")
|
||||
assert hasattr(result.response, "status_code")
|
||||
assert hasattr(result.response, "headers")
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue