Merge branch 'main' into litellm_dev_09_11_2025_p1

This commit is contained in:
Krish Dholakia 2025-09-12 19:59:28 -07:00 • committed by GitHub
commit a11f50d8ba
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
104 changed files with 3960 additions and 453 deletions

View file

@ -671,6 +671,7 @@ jobs:
pip install mypy
pip install "google-generativeai==0.3.2"
pip install "google-cloud-aiplatform==1.43.0"
pip install "google-genai==1.22.0"
pip install pyarrow
pip install "boto3==1.36.0"
pip install "aioboto3==13.4.0"

View file

@ -0,0 +1,25 @@
from openai import OpenAI
client = OpenAI(
base_url="http://0.0.0.0:4000",
api_key="sk-1234",
)
BEDROCK_BATCH_MODEL = "bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0"
# Upload file
batch_input_file = client.files.create(
file=open("./bedrock_batch_completions.jsonl", "rb"),
purpose="batch",
extra_body={"target_model_names": BEDROCK_BATCH_MODEL}
)
print(batch_input_file)
# Create batch
batch = client.batches.create(
input_file_id=batch_input_file.id,
endpoint="/v1/chat/completions",
completion_window="24h",
metadata={"description": "Test batch job"},
)
print(batch)

View file

@ -0,0 +1,128 @@
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}

View file

@ -7,7 +7,7 @@ Covers Batches, Files
| Feature | Supported | Notes |
|-------|-------|-------|
| Supported Providers | OpenAI, Azure, Vertex | - |
| Supported Providers | OpenAI, Azure, Vertex, Bedrock | - |
| ✨ Cost Tracking | ✅ | LiteLLM Enterprise only |
| Logging | ✅ | Works across all logging integrations |
@ -178,6 +178,7 @@ print("list_batches_response=", list_batches_response)
### [Azure OpenAI](./providers/azure#azure-batches-api)
### [OpenAI](#quick-start)
### [Vertex AI](./providers/vertex#batch-apis)
### [Bedrock](./providers/bedrock_batches)
## How Cost Tracking for Batches API Works

View file

@ -0,0 +1,180 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Bedrock Batches
Use Amazon Bedrock Batch Inference API through LiteLLM.
| Property | Details |
|----------|---------|
| Description | Amazon Bedrock Batch Inference allows you to run inference on large datasets asynchronously |
| Provider Doc | [AWS Bedrock Batch Inference ↗](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html) |
## Overview
Use this to:
- Run batch inference on large datasets with Bedrock models
- Control batch model access by key/user/team (same as chat completion models)
- Manage S3 storage for batch input/output files
## (Proxy Admin) Usage
Here's how to give developers access to your Bedrock Batch models.
### 1. Setup config.yaml
- Specify `mode: batch` for each model: Allows developers to know this is a batch model
- Configure S3 bucket and AWS credentials for batch operations
```yaml showLineNumbers title="litellm_config.yaml"
model_list:
- model_name: "bedrock-batch-claude"
litellm_params:
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
#########################################################
########## batch specific params ########################
s3_bucket_name: litellm-proxy
s3_region_name: us-west-2
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
model_info:
mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model
```
**Required Parameters:**
| Parameter | Description |
|-----------|-------------|
| `s3_bucket_name` | S3 bucket for batch input/output files |
| `s3_region_name` | AWS region for S3 bucket |
| `s3_access_key_id` | AWS access key for S3 bucket |
| `s3_secret_access_key` | AWS secret key for S3 bucket |
| `aws_batch_role_arn` | IAM role ARN for Bedrock batch operations. Bedrock Batch APIs require an IAM role ARN to be set. |
| `mode: batch` | Indicates to LiteLLM this is a batch model |
### 2. Create Virtual Key
```bash showLineNumbers title="create_virtual_key.sh"
curl -L -X POST 'https://{PROXY_BASE_URL}/key/generate' \
-H 'Authorization: Bearer ${PROXY_API_KEY}' \
-H 'Content-Type: application/json' \
-d '{"models": ["bedrock-batch-claude"]}'
```
You can now use the virtual key to access the batch models (See Developer flow).
## (Developer) Usage
Here's how to create a LiteLLM managed file and execute Bedrock Batch CRUD operations with the file.
### 1. Create request.jsonl
- Check models available via `/model_group/info`
- See all models with `mode: batch`
- Set `model` in .jsonl to the model from `/model_group/info`
```json showLineNumbers title="bedrock_batch_completions.jsonl"
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock-batch-claude", "messages": [{"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello world!"}], "max_tokens": 1000}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock-batch-claude", "messages": [{"role": "system", "content": "You are an unhelpful assistant."}, {"role": "user", "content": "Hello world!"}], "max_tokens": 1000}}
```
Expectation:
- LiteLLM translates this to the bedrock deployment specific value (e.g. `bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0`)
### 2. Upload File
Specify `target_model_names: "<model-name>"` to enable LiteLLM managed files and request validation.
model-name should be the same as the model-name in the request.jsonl
<Tabs>
<TabItem value="python" label="Python">
```python showLineNumbers title="bedrock_batch.py"
from openai import OpenAI
client = OpenAI(
base_url="http://0.0.0.0:4000",
api_key="sk-1234",
)
# Upload file
batch_input_file = client.files.create(
file=open("./bedrock_batch_completions.jsonl", "rb"), # {"model": "bedrock-batch-claude"} <-> {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0"}
purpose="batch",
extra_body={"target_model_names": "bedrock-batch-claude"}
)
print(batch_input_file)
```
</TabItem>
<TabItem value="curl" label="Curl">
```bash showLineNumbers title="Upload File"
curl http://localhost:4000/v1/files \
-H "Authorization: Bearer sk-1234" \
-F purpose="batch" \
-F file="@bedrock_batch_completions.jsonl" \
-F extra_body='{"target_model_names": "bedrock-batch-claude"}'
```
</TabItem>
</Tabs>
**Where is the file written?**:
The file is written to S3 bucket specified in your config and prepared for Bedrock batch inference.
### 3. Create the batch
<Tabs>
<TabItem value="python" label="Python">
```python showLineNumbers title="bedrock_batch.py"
...
# Create batch
batch = client.batches.create(
input_file_id=batch_input_file.id,
endpoint="/v1/chat/completions",
completion_window="24h",
metadata={"description": "Test batch job"},
)
print(batch)
```
</TabItem>
<TabItem value="curl" label="Curl">
```bash showLineNumbers title="Create Batch Request"
curl http://localhost:4000/v1/batches \
-H "Authorization: Bearer sk-1234" \
-H "Content-Type: application/json" \
-d '{
"input_file_id": "file-abc123",
"endpoint": "/v1/chat/completions",
"completion_window": "24h",
"metadata": {"description": "Test batch job"}
}'
```
</TabItem>
</Tabs>
## FAQ
### Where are my files written?
When a `target_model_names` is specified, the file is written to the S3 bucket configured in your Bedrock batch model configuration.
### What models are supported?
LiteLLM only supports Bedrock Anthropic Models for Batch API. If you want other bedrock models file an issue [here](https://github.com/BerriAI/litellm/issues/new/choose).
## Further Reading
- [AWS Bedrock Batch Inference Documentation](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html)
- [LiteLLM Managed Batches](../proxy/managed_batches)
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)

View file

@ -1,4 +1,4 @@
# Dashscope
# Dashscope (Qwen API)
https://dashscope.console.aliyun.com/
**We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests**

View file

@ -11,13 +11,13 @@ The proxy also supports json logs. [See here](#json-logs)
**via cli**
```bash
```bash showLineNumbers
$ litellm --debug
```
**via env**
```python
```python showLineNumbers
os.environ["LITELLM_LOG"] = "INFO"
```
@ -25,25 +25,25 @@ os.environ["LITELLM_LOG"] = "INFO"
**via cli**
```bash
```bash showLineNumbers
$ litellm --detailed_debug
```
**via env**
```python
```python showLineNumbers
os.environ["LITELLM_LOG"] = "DEBUG"
```
### Debug Logs
Run the proxy with `--detailed_debug` to view detailed debug logs
```shell
```shell showLineNumbers
litellm --config /path/to/config.yaml --detailed_debug
```
When making requests you should see the POST request sent by LiteLLM to the LLM on the Terminal output
```shell
```shell showLineNumbers
POST Request Sent from LiteLLM:
curl -X POST \
https://api.openai.com/v1/chat/completions \
@ -51,25 +51,63 @@ https://api.openai.com/v1/chat/completions \
-d '{"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}'
```
## Debug single request
Pass in `litellm_request_debug=True` in the request body
```bash showLineNumbers
curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-d '{
"model":"fake-openai-endpoint",
"messages": [{"role": "user","content": "How many r in the word strawberry?"}],
"litellm_request_debug": true
}'
```
This will emit the raw request sent by LiteLLM to the API Provider and raw response received from the API Provider for **just** this request in the logs.
```bash showLineNumbers
INFO: Uvicorn running on http://0.0.0.0:4000 (Press CTRL+C to quit)
20:14:06 - LiteLLM:WARNING: litellm_logging.py:938 -
POST Request Sent from LiteLLM:
curl -X POST \
https://exampleopenaiendpoint-production.up.railway.app/chat/completions \
-H 'Authorization: Be****ey' -H 'Content-Type: application/json' \
-d '{'model': 'fake', 'messages': [{'role': 'user', 'content': 'How many r in the word strawberry?'}], 'stream': False}'
20:14:06 - LiteLLM:WARNING: litellm_logging.py:1015 - RAW RESPONSE:
{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-3.5-turbo-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}}
INFO: 127.0.0.1:56155 - "POST /chat/completions HTTP/1.1" 200 OK
```
## JSON LOGS
Set `JSON_LOGS="True"` in your env:
```bash
```bash showLineNumbers
export JSON_LOGS="True"
```
**OR**
Set `json_logs: true` in your yaml:
```yaml
```yaml showLineNumbers
litellm_settings:
json_logs: true
```
Start proxy
```bash
```bash showLineNumbers
$ litellm
```
@ -80,7 +118,7 @@ The proxy will now all logs in json format.
Turn off fastapi's default 'INFO' logs
1. Turn on 'json logs'
```yaml
```yaml showLineNumbers
litellm_settings:
json_logs: true
```
@ -89,20 +127,20 @@ litellm_settings:
Only get logs if an error occurs.
```bash
```bash showLineNumbers
LITELLM_LOG="ERROR"
```
3. Start proxy
```bash
```bash showLineNumbers
$ litellm
```
Expected Output:
```bash
```bash showLineNumbers
# no info statements
```
@ -119,14 +157,14 @@ This can be caused due to all your models hitting rate limit errors, causing the
How to control this?
- Adjust the cooldown time
```yaml
```yaml showLineNumbers
router_settings:
cooldown_time: 0 # 👈 KEY CHANGE
```
- Disable Cooldowns [NOT RECOMMENDED]
```yaml
```yaml showLineNumbers
router_settings:
disable_cooldowns: True
```

View file

@ -410,6 +410,7 @@ const sidebars = {
items: [
"providers/bedrock",
"providers/bedrock_agents",
"providers/bedrock_batches",
"providers/bedrock_vector_store",
]
},

View file

@ -19,6 +19,7 @@ from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast
import httpx
import litellm
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.azure.batches.handler import AzureBatchesAPI
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
@ -38,6 +39,7 @@ from litellm.utils import (
ProviderConfigManager,
client,
get_litellm_params,
get_llm_provider,
supports_httpx_timeout,
)
@ -49,6 +51,45 @@ base_llm_http_handler = BaseLLMHTTPHandler()
#################################################
def _resolve_timeout(
optional_params: GenericLiteLLMParams,
kwargs: Dict[str, Any],
custom_llm_provider: str,
default_timeout: float = 600.0,
) -> float:
"""
Resolve timeout value from various sources and handle httpx.Timeout objects.
Args:
optional_params: GenericLiteLLMParams object containing timeout
kwargs: Additional kwargs that may contain request_timeout
custom_llm_provider: Provider name for httpx timeout support check
default_timeout: Default timeout value to use
Returns:
Resolved timeout as float
"""
timeout = optional_params.timeout or kwargs.get("request_timeout", default_timeout) or default_timeout
# Handle httpx.Timeout objects
if isinstance(timeout, httpx.Timeout):
if supports_httpx_timeout(custom_llm_provider) is False:
# Extract read timeout for providers that don't support httpx.Timeout
read_timeout = timeout.read or default_timeout
return float(read_timeout)
else:
# For providers that support httpx.Timeout, we still need to return a float
# This case might need to be handled differently based on the actual use case
return float(timeout.read or default_timeout)
# Handle None case
if timeout is None:
return float(default_timeout)
# Handle numeric values (int, float, string representations)
return float(timeout)
@client
async def acreate_batch(
completion_window: Literal["24h"],
@ -118,13 +159,23 @@ def create_batch(
litellm_call_id = kwargs.get("litellm_call_id", None)
proxy_server_request = kwargs.get("proxy_server_request", None)
model_info = kwargs.get("model_info", None)
model: Optional[str] = kwargs.get("model", None)
try:
if model is not None:
model, _, _, _ = get_llm_provider(
model=model,
custom_llm_provider=None,
)
except Exception as e:
verbose_logger.exception(f"litellm.batches.main.py::create_batch() - Error inferring custom_llm_provider - {str(e)}")
_is_async = kwargs.pop("acreate_batch", False) is True
litellm_params = dict(GenericLiteLLMParams(**kwargs))
litellm_logging_obj: LiteLLMLoggingObj = cast(LiteLLMLoggingObj, kwargs.get("litellm_logging_obj", None))
### TIMEOUT LOGIC ###
timeout = optional_params.timeout or kwargs.get("request_timeout", 600) or 600
timeout = _resolve_timeout(optional_params, kwargs, custom_llm_provider)
litellm_logging_obj.update_environment_variables(
model=None,
model=model,
user=None,
optional_params=optional_params.model_dump(),
litellm_params={
@ -138,18 +189,6 @@ def create_batch(
},
custom_llm_provider=custom_llm_provider,
)
if (
timeout is not None
and isinstance(timeout, httpx.Timeout)
and supports_httpx_timeout(custom_llm_provider) is False
):
read_timeout = timeout.read or 600
timeout = read_timeout # default 10 min timeout
elif timeout is not None and not isinstance(timeout, httpx.Timeout):
timeout = float(timeout) # type: ignore
elif timeout is None:
timeout = 600.0
_create_batch_request = CreateBatchRequest(
@ -160,10 +199,13 @@ def create_batch(
extra_headers=extra_headers,
extra_body=extra_body,
)
provider_config = ProviderConfigManager.get_provider_batches_config(
model="",
provider=LlmProviders(custom_llm_provider),
)
if model is not None:
provider_config = ProviderConfigManager.get_provider_batches_config(
model=model,
provider=LlmProviders(custom_llm_provider),
)
else:
provider_config = None
if provider_config is not None:
response = base_llm_http_handler.create_batch(
provider_config=provider_config,
@ -179,6 +221,7 @@ def create_batch(
and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
else None,
timeout=timeout,
model=model,
)
return response
api_base: Optional[str] = None

View file

@ -15,7 +15,7 @@ DEFAULT_SQS_FLUSH_INTERVAL_SECONDS = int(
os.getenv("DEFAULT_SQS_FLUSH_INTERVAL_SECONDS", 10)
)
DEFAULT_NUM_WORKERS_LITELLM_PROXY = int(
os.getenv("DEFAULT_NUM_WORKERS_LITELLM_PROXY", os.cpu_count() or 4)
os.getenv("DEFAULT_NUM_WORKERS_LITELLM_PROXY", 1)
)
DEFAULT_SQS_BATCH_SIZE = int(os.getenv("DEFAULT_SQS_BATCH_SIZE", 512))
SQS_SEND_MESSAGE_ACTION = "SendMessage"
@ -60,7 +60,9 @@ DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO = int(
os.getenv("DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO", 128)
)
DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE = int(
os.getenv("DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE", 512)
os.getenv(
"DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE", 512
)
)
# Generic fallback for unknown models
@ -949,7 +951,9 @@ LITELLM_CLI_SESSION_TOKEN_PREFIX = "litellm-session-token"
DB_SPEND_UPDATE_JOB_NAME = "db_spend_update_job"
PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME = "prometheus_emit_budget_metrics"
CLOUDZERO_EXPORT_USAGE_DATA_JOB_NAME = "cloudzero_export_usage_data"
CLOUDZERO_MAX_FETCHED_DATA_RECORDS = int(os.getenv("CLOUDZERO_MAX_FETCHED_DATA_RECORDS", 50000))
CLOUDZERO_MAX_FETCHED_DATA_RECORDS = int(
os.getenv("CLOUDZERO_MAX_FETCHED_DATA_RECORDS", 50000)
)
SPEND_LOG_CLEANUP_JOB_NAME = "spend_log_cleanup"
SPEND_LOG_RUN_LOOPS = int(os.getenv("SPEND_LOG_RUN_LOOPS", 500))
SPEND_LOG_CLEANUP_BATCH_SIZE = int(os.getenv("SPEND_LOG_CLEANUP_BATCH_SIZE", 1000))

View file

@ -344,6 +344,11 @@ def cost_per_token( # noqa: PLR0915
return perplexity_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "xai":
return xai_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "dashscope":
from litellm.llms.dashscope.cost_calculator import (
cost_per_token as dashscope_cost_per_token,
)
return dashscope_cost_per_token(model=model, usage=usage_block)
else:
model_info = _cached_get_model_info_helper(
model=model, custom_llm_provider=custom_llm_provider

27
litellm/files/utils.py Normal file
View file

@ -0,0 +1,27 @@
from typing import Optional
from litellm.types.llms.openai import CreateFileRequest
from litellm.types.utils import ExtractedFileData
class FilesAPIUtils:
"""
Utils for files API interface on litellm
"""
@staticmethod
def is_batch_jsonl_file(create_file_data: CreateFileRequest, extracted_file_data: ExtractedFileData) -> bool:
"""
Check if the file is a batch jsonl file
"""
return (
create_file_data.get("purpose") == "batch"
and FilesAPIUtils.valid_content_type(extracted_file_data.get("content_type"))
and extracted_file_data.get("content") is not None
)
@staticmethod
def valid_content_type(content_type: Optional[str]) -> bool:
"""
Check if the content type is valid
"""
return content_type in set(["application/jsonl", "application/octet-stream"])

View file

@ -224,6 +224,9 @@ async def agenerate_content(
loop = asyncio.get_event_loop()
kwargs["agenerate_content"] = True
# Handle generationConfig parameter from kwargs for backward compatibility
if "generationConfig" in kwargs and config is None:
config = kwargs.pop("generationConfig")
# get custom llm provider so we can use this for mapping exceptions
if custom_llm_provider is None:
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
@ -288,6 +291,9 @@ def generate_content(
try:
_is_async = kwargs.pop("agenerate_content", False) is True
# Handle generationConfig parameter from kwargs for backward compatibility
if "generationConfig" in kwargs and config is None:
config = kwargs.pop("generationConfig")
# Check for mock response first
litellm_params = GenericLiteLLMParams(**kwargs)
if litellm_params.mock_response and isinstance(
@ -374,6 +380,9 @@ async def agenerate_content_stream(
try:
kwargs["agenerate_content_stream"] = True
# Handle generationConfig parameter from kwargs for backward compatibility
if "generationConfig" in kwargs and config is None:
config = kwargs.pop("generationConfig")
# get custom llm provider so we can use this for mapping exceptions
if custom_llm_provider is None:
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
@ -461,6 +470,9 @@ def generate_content_stream(
# Remove any async-related flags since this is the sync function
_is_async = kwargs.pop("agenerate_content_stream", False)
# Handle generationConfig parameter from kwargs for backward compatibility
if "generationConfig" in kwargs and config is None:
config = kwargs.pop("generationConfig")
# Setup the call
setup_result = GenerateContentHelper.setup_generate_content_call(
model=model,

View file

@ -62,6 +62,7 @@ def get_litellm_params(
use_litellm_proxy: Optional[bool] = None,
api_version: Optional[str] = None,
max_retries: Optional[int] = None,
litellm_request_debug: Optional[bool] = None,
**kwargs,
) -> dict:
litellm_params = {
@ -118,5 +119,6 @@ def get_litellm_params(
"vertex_credentials": kwargs.get("vertex_credentials"),
"vertex_project": kwargs.get("vertex_project"),
"use_litellm_proxy": use_litellm_proxy,
"litellm_request_debug": litellm_request_debug,
}
return litellm_params

View file

@ -245,6 +245,7 @@ class Logging(LiteLLMLoggingBaseClass):
global supabaseClient, promptLayerLogger, weightsBiasesLogger, logfireLogger, capture_exception, add_breadcrumb, lunaryLogger, logfireLogger, prometheusLogger, slack_app
custom_pricing: bool = False
stream_options = None
litellm_request_debug: bool = False
def __init__(
self,
@ -470,6 +471,7 @@ class Logging(LiteLLMLoggingBaseClass):
**self.litellm_params,
**scrub_sensitive_keys_in_metadata(litellm_params),
}
self.litellm_request_debug = litellm_params.get("litellm_request_debug", False)
self.logger_fn = litellm_params.get("logger_fn", None)
verbose_logger.debug(f"self.optional_params: {self.optional_params}")
@ -907,13 +909,19 @@ class Logging(LiteLLMLoggingBaseClass):
Prints the RAW curl command sent from LiteLLM
"""
if _is_debugging_on():
if _is_debugging_on() or self.litellm_request_debug:
if json_logs:
masked_headers = self._get_masked_headers(headers)
verbose_logger.debug(
"POST Request Sent from LiteLLM",
extra={"api_base": {api_base}, **masked_headers},
)
if self.litellm_request_debug:
verbose_logger.warning( # .warning ensures this shows up in all environments
"POST Request Sent from LiteLLM",
extra={"api_base": {api_base}, **masked_headers},
)
else:
verbose_logger.debug(
"POST Request Sent from LiteLLM",
extra={"api_base": {api_base}, **masked_headers},
)
else:
headers = additional_args.get("headers", {})
if headers is None:
@ -926,7 +934,12 @@ class Logging(LiteLLMLoggingBaseClass):
additional_args=additional_args,
data=data,
)
verbose_logger.debug(f"\033[92m{curl_command}\033[0m\n")
if self.litellm_request_debug:
verbose_logger.warning(
f"\033[92m{curl_command}\033[0m\n"
) # .warning ensures this shows up in all environments
else:
verbose_logger.debug(f"\033[92m{curl_command}\033[0m\n")
def _get_request_body(self, data: dict) -> str:
return str(data)
@ -983,8 +996,14 @@ class Logging(LiteLLMLoggingBaseClass):
self.model_call_details["additional_args"] = additional_args
self.model_call_details["log_event_type"] = "post_api_call"
if self.litellm_request_debug:
attr = "warning"
else:
attr = "debug"
if json_logs:
verbose_logger.debug(
callattr = getattr(verbose_logger, attr)
callattr(
"RAW RESPONSE:\n{}\n\n".format(
self.model_call_details.get(
"original_response", self.model_call_details
@ -992,7 +1011,8 @@ class Logging(LiteLLMLoggingBaseClass):
),
)
else:
print_verbose(
callattr = getattr(verbose_logger, attr)
callattr(
"RAW RESPONSE:\n{}\n\n".format(
self.model_call_details.get(
"original_response", self.model_call_details
@ -1714,12 +1734,16 @@ class Logging(LiteLLMLoggingBaseClass):
response_obj=result,
start_time=start_time,
end_time=end_time,
litellm_call_id=current_call_id
if (
current_call_id := litellm_params.get("litellm_call_id")
)
is not None
else str(uuid.uuid4()),
litellm_call_id=(
current_call_id
if (
current_call_id := litellm_params.get(
"litellm_call_id"
)
)
is not None
else str(uuid.uuid4())
),
print_verbose=print_verbose,
)
if callback == "wandb" and weightsBiasesLogger is not None:
@ -3367,6 +3391,7 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
return galileo_logger # type: ignore
elif logging_integration == "cloudzero":
from litellm.integrations.cloudzero.cloudzero import CloudZeroLogger
for callback in _in_memory_loggers:
if isinstance(callback, CloudZeroLogger):
return callback # type: ignore
@ -3594,6 +3619,7 @@ def get_custom_logger_compatible_class( # noqa: PLR0915
return callback
elif logging_integration == "cloudzero":
from litellm.integrations.cloudzero.cloudzero import CloudZeroLogger
for callback in _in_memory_loggers:
if isinstance(callback, CloudZeroLogger):
return callback
@ -4504,7 +4530,7 @@ def get_standard_logging_object_payload(
def emit_standard_logging_payload(payload: StandardLoggingPayload):
if os.getenv("LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD"):
print(json.dumps(payload, indent=4)) # noqa
print(json.dumps(payload, indent=4)) # noqa
def get_standard_logging_metadata(

View file

@ -1,10 +1,21 @@
import asyncio
import contextlib
from typing import Coroutine, Optional
import contextvars
from typing import Coroutine, Optional, TypedDict
from litellm._logging import verbose_logger
class LoggingTask(TypedDict):
"""
A logging task with its associated context to ensure logging is executed in
the original task's context.
"""
coroutine: Coroutine
context: contextvars.Context
class LoggingWorker:
"""
A simple, async logging worker that processes log coroutines in the background.
@ -13,77 +24,84 @@ class LoggingWorker:
This leads to a +200 RPS performance improvement when using LiteLLM Python SDK or Proxy Server.
- Use this to queue coroutine tasks that are not critical to the main flow of the application. e.g Success/Error callbacks, logging, etc.
"""
LOGGING_WORKER_MAX_QUEUE_SIZE = 50_000
LOGGING_WORKER_MAX_TIME_PER_COROUTINE = 20.0
MAX_ITERATIONS_TO_CLEAR_QUEUE = 200
MAX_TIME_TO_CLEAR_QUEUE = 5.0
def __init__(
self,
timeout: float = LOGGING_WORKER_MAX_TIME_PER_COROUTINE,
self,
timeout: float = LOGGING_WORKER_MAX_TIME_PER_COROUTINE,
max_queue_size: int = LOGGING_WORKER_MAX_QUEUE_SIZE,
):
self.timeout = timeout
self.max_queue_size = max_queue_size
self._queue: Optional[asyncio.Queue] = None
self._queue: Optional[asyncio.Queue[LoggingTask]] = None
self._worker_task: Optional[asyncio.Task] = None
def _ensure_queue(self) -> None:
"""Initialize the queue if it doesn't exist."""
if self._queue is None:
self._queue = asyncio.Queue(maxsize=self.max_queue_size)
def start(self) -> None:
"""Start the logging worker. Idempotent - safe to call multiple times."""
self._ensure_queue()
if self._worker_task is None or self._worker_task.done():
self._worker_task = asyncio.create_task(self._worker_loop())
async def _worker_loop(self) -> None:
"""Main worker loop that processes log coroutines sequentially."""
try:
if self._queue is None:
return
while True:
# Process one coroutine at a time to keep event loop load predictable
coroutine = await self._queue.get()
task = await self._queue.get()
try:
await asyncio.wait_for(coroutine, timeout=self.timeout)
# Run the coroutine in its original context
await asyncio.wait_for(
task["context"].run(asyncio.create_task, task["coroutine"]),
timeout=self.timeout,
)
except Exception as e:
verbose_logger.exception(f"LoggingWorker error: {e}")
pass
finally:
self._queue.task_done()
except asyncio.CancelledError:
verbose_logger.debug("LoggingWorker cancelled during shutdown")
# Attempt to clear remaining items to prevent "never awaited" warnings
await self.clear_queue()
def enqueue(self, coroutine: Coroutine) -> None:
"""
Add a coroutine to the logging queue.
Add a coroutine to the logging queue.
Hot path: never blocks, drops logs if queue is full.
"""
if self._queue is None:
return
try:
self._queue.put_nowait(coroutine)
# Capture the current context when enqueueing
task = LoggingTask(coroutine=coroutine, context=contextvars.copy_context())
self._queue.put_nowait(task)
except asyncio.QueueFull as e:
verbose_logger.exception(f"LoggingWorker queue is full: {e}")
# Drop logs on overload to protect request throughput
pass
def ensure_initialized_and_enqueue(self, async_coroutine: Coroutine):
"""
Ensure the logging worker is initialized and enqueue the coroutine.
"""
self.start()
self.enqueue(async_coroutine)
async def stop(self) -> None:
"""Stop the logging worker and clean up resources."""
if self._worker_task:
@ -91,34 +109,42 @@ class LoggingWorker:
with contextlib.suppress(Exception):
await self._worker_task
self._worker_task = None
async def flush(self) -> None:
"""Flush the logging queue."""
if self._queue is None:
return
while not self._queue.empty():
await self._queue.join()
async def clear_queue(self):
"""
Clear the queue with a maximum time limit.
"""
if self._queue is None:
return
start_time = asyncio.get_event_loop().time()
for _ in range(self.MAX_ITERATIONS_TO_CLEAR_QUEUE):
# Check if we've exceeded the maximum time
if asyncio.get_event_loop().time() - start_time >= self.MAX_TIME_TO_CLEAR_QUEUE:
verbose_logger.warning(f"clear_queue exceeded max_time of {self.MAX_TIME_TO_CLEAR_QUEUE}s, stopping early")
if (
asyncio.get_event_loop().time() - start_time
>= self.MAX_TIME_TO_CLEAR_QUEUE
):
verbose_logger.warning(
f"clear_queue exceeded max_time of {self.MAX_TIME_TO_CLEAR_QUEUE}s, stopping early"
)
break
try:
coroutine = self._queue.get_nowait()
task = self._queue.get_nowait()
# Await the coroutine to properly execute and avoid "never awaited" warnings
try:
await asyncio.wait_for(coroutine, timeout=self.timeout)
await asyncio.wait_for(
task["context"].run(asyncio.create_task, task["coroutine"]),
timeout=self.timeout,
)
except Exception:
# Suppress errors during cleanup
pass
@ -129,4 +155,3 @@ class LoggingWorker:
# Global instance for backward compatibility
GLOBAL_LOGGING_WORKER = LoggingWorker()

View file

@ -28,10 +28,6 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
TextBlock,
)
def __init__(self, completion_stream: Any, model: str):
super().__init__(completion_stream)
self.model = model
sent_first_chunk: bool = False
sent_content_block_start: bool = False
sent_content_block_finish: bool = False
@ -39,6 +35,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
sent_last_message: bool = False
holding_chunk: Optional[Any] = None
holding_stop_reason_chunk: Optional[Any] = None
queued_usage_chunk: bool = False
current_content_block_index: int = 0
current_content_block_start: ContentBlockContentBlockDict = TextBlock(
type="text",
@ -47,6 +44,10 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
pending_new_content_block: bool = False
chunk_queue: deque = deque() # Queue for buffering multiple chunks
def __init__(self, completion_stream: Any, model: str):
super().__init__(completion_stream)
self.model = model
def __next__(self):
from .transformation import LiteLLMAnthropicMessagesAdapter
@ -217,77 +218,83 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
# Queue the merged chunk and reset
self.chunk_queue.append(merged_chunk)
self.queued_usage_chunk = True
self.holding_stop_reason_chunk = None
return self.chunk_queue.popleft()
# Check if this processed chunk has a stop_reason - hold it for next chunk
if should_start_new_block and not self.sent_content_block_finish:
# Queue the sequence: content_block_stop -> content_block_start -> current_chunk
if not self.queued_usage_chunk:
if should_start_new_block and not self.sent_content_block_finish:
# Queue the sequence: content_block_stop -> content_block_start -> current_chunk
# 1. Stop current content block
self.chunk_queue.append(
{
"type": "content_block_stop",
"index": max(self.current_content_block_index - 1, 0),
}
)
# 1. Stop current content block
self.chunk_queue.append(
{
"type": "content_block_stop",
"index": max(self.current_content_block_index - 1, 0),
}
)
# 2. Start new content block
self.chunk_queue.append(
{
"type": "content_block_start",
"index": self.current_content_block_index,
"content_block": self.current_content_block_start,
}
)
# 2. Start new content block
self.chunk_queue.append(
{
"type": "content_block_start",
"index": self.current_content_block_index,
"content_block": self.current_content_block_start,
}
)
# 3. Queue the current chunk (don't lose it!)
self.chunk_queue.append(processed_chunk)
# Reset state for new block
self.sent_content_block_finish = False
# Return the first queued item
return self.chunk_queue.popleft()
if (
processed_chunk["type"] == "message_delta"
and self.sent_content_block_finish is False
):
# Queue both the content_block_stop and the holding chunk
self.chunk_queue.append(
{
"type": "content_block_stop",
"index": self.current_content_block_index,
}
)
self.sent_content_block_finish = True
if processed_chunk.get("delta", {}).get("stop_reason") is not None:
self.holding_stop_reason_chunk = processed_chunk
else:
# 3. Queue the current chunk (don't lose it!)
self.chunk_queue.append(processed_chunk)
return self.chunk_queue.popleft()
elif self.holding_chunk is not None:
# Queue both chunks
self.chunk_queue.append(self.holding_chunk)
self.chunk_queue.append(processed_chunk)
self.holding_chunk = None
return self.chunk_queue.popleft()
else:
# Queue the current chunk
self.chunk_queue.append(processed_chunk)
return self.chunk_queue.popleft()
# Reset state for new block
self.sent_content_block_finish = False
# Return the first queued item
return self.chunk_queue.popleft()
if (
processed_chunk["type"] == "message_delta"
and self.sent_content_block_finish is False
):
# Queue both the content_block_stop and the holding chunk
self.chunk_queue.append(
{
"type": "content_block_stop",
"index": self.current_content_block_index,
}
)
self.sent_content_block_finish = True
if (
processed_chunk.get("delta", {}).get("stop_reason")
is not None
):
self.holding_stop_reason_chunk = processed_chunk
else:
self.chunk_queue.append(processed_chunk)
return self.chunk_queue.popleft()
elif self.holding_chunk is not None:
# Queue both chunks
self.chunk_queue.append(self.holding_chunk)
self.chunk_queue.append(processed_chunk)
self.holding_chunk = None
return self.chunk_queue.popleft()
else:
# Queue the current chunk
self.chunk_queue.append(processed_chunk)
return self.chunk_queue.popleft()
# Handle any remaining held chunks after stream ends
if self.holding_stop_reason_chunk is not None:
self.chunk_queue.append(self.holding_stop_reason_chunk)
self.holding_stop_reason_chunk = None
if not self.queued_usage_chunk:
if self.holding_stop_reason_chunk is not None:
self.chunk_queue.append(self.holding_stop_reason_chunk)
self.holding_stop_reason_chunk = None
if self.holding_chunk is not None:
self.chunk_queue.append(self.holding_chunk)
self.holding_chunk = None
if self.holding_chunk is not None:
self.chunk_queue.append(self.holding_chunk)
self.holding_chunk = None
if not self.sent_last_message:
self.sent_last_message = True

View file

@ -124,15 +124,13 @@ class BedrockBatchesConfig(BaseAWSLLM, BaseBatchesConfig):
"AWS IAM role ARN is required for Bedrock batch jobs. "
"Set 'aws_batch_role_arn' in litellm_params or AWS_BATCH_ROLE_ARN env var"
)
# Get the actual Bedrock model ID using common utility
bedrock_model_id = self.common_utils.extract_model_from_s3_file_path(input_file_id, optional_params)
if not bedrock_model_id:
raise ValueError("Could not determine Bedrock model ID. Ensure the model is specified in the input file or passed as a parameter.")
if not model:
raise ValueError("Could not determine Bedrock model ID. Please pass `model` in your request body.")
# Generate job name with the correct model ID using common utility
job_name = self.common_utils.generate_unique_job_name(bedrock_model_id, prefix="litellm")
job_name = self.common_utils.generate_unique_job_name(model, prefix="litellm")
output_key = f"litellm-batch-outputs/{job_name}/"
# Build input data config
@ -151,7 +149,7 @@ class BedrockBatchesConfig(BaseAWSLLM, BaseBatchesConfig):
# Create Bedrock batch request with proper typing
bedrock_request: BedrockCreateBatchRequest = {
"modelId": bedrock_model_id,
"modelId": model,
"jobName": job_name,
"inputDataConfig": input_data_config,
"outputDataConfig": output_data_config,

View file

@ -6,6 +6,8 @@ from typing import Any, Dict, List, Optional, Tuple, Union
from httpx import Headers, Response
from litellm._logging import verbose_logger
from litellm.files.utils import FilesAPIUtils
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.files.transformation import (
@ -21,6 +23,7 @@ from litellm.types.llms.openai import (
PathLike,
)
from litellm.types.utils import ExtractedFileData, LlmProviders
from litellm.utils import get_llm_provider
from ..base_aws_llm import BaseAWSLLM
from ..common_utils import BedrockError
@ -111,6 +114,10 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
# Remove bedrock/ prefix if present
if _model.startswith("bedrock/"):
_model = _model[8:]
# Replace colons with hyphens for Bedrock S3 URI compliance
_model = _model.replace(":", "-")
object_name = f"litellm-bedrock-files-{_model}-{uuid.uuid4()}.jsonl"
return object_name
@ -191,24 +198,6 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
) -> dict:
return optional_params
def _get_bedrock_provider_from_model(self, model: str) -> Optional[str]:
"""
Extract provider from Bedrock model name
"""
if model.startswith("anthropic."):
return "anthropic"
elif model.startswith("cohere."):
return "cohere"
elif model.startswith("meta.") or model.startswith("llama"):
return "meta"
elif model.startswith("mistral."):
return "mistral"
elif model.startswith("ai21."):
return "ai21"
elif model.startswith("amazon."):
return "amazon"
else:
return None
def _map_openai_to_bedrock_params(
self,
@ -218,11 +207,12 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
"""
Transform OpenAI request body to Bedrock-compatible modelInput parameters using existing transformation logic
"""
from litellm.types.utils import LlmProviders
_model = openai_request_body.get("model", "")
messages = openai_request_body.get("messages", [])
# Use existing Anthropic transformation logic for Anthropic models
if provider == "anthropic":
if provider == LlmProviders.ANTHROPIC:
from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import (
AmazonAnthropicClaudeConfig,
)
@ -231,16 +221,22 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
# Extract optional params (everything except model and messages)
optional_params = {k: v for k, v in openai_request_body.items() if k not in ["model", "messages"]}
mapped_params = anthropic_config.map_openai_params(
non_default_params={},
optional_params=optional_params,
model=_model,
drop_params=False
)
# Transform using existing Anthropic logic
bedrock_params = anthropic_config.transform_request(
model=_model,
messages=messages,
optional_params=optional_params,
optional_params=mapped_params,
litellm_params={},
headers={}
)
return bedrock_params
else:
# For other providers, use basic mapping
@ -278,9 +274,17 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
# Extract the request body from OpenAI format
openai_body = _openai_jsonl_content.get("body", {})
model = openai_body.get("model", "")
try:
model, _, _, _ = get_llm_provider(
model=model,
custom_llm_provider=None,
)
except Exception as e:
verbose_logger.exception(f"litellm.llms.bedrock.files.transformation.py::_transform_openai_jsonl_content_to_bedrock_jsonl_content() - Error inferring custom_llm_provider - {str(e)}")
# Determine provider from model name
provider = self._get_bedrock_provider_from_model(model)
provider = self.get_bedrock_invoke_provider(model)
# Transform to Bedrock modelInput format
model_input = self._map_openai_to_bedrock_params(
@ -315,11 +319,13 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
extracted_file_data = extract_file_data(file_data)
extracted_file_data_content = extracted_file_data.get("content")
if extracted_file_data_content is None:
raise ValueError("file content is required")
# Get and transform the file content
if (
create_file_data.get("purpose") == "batch"
and extracted_file_data.get("content_type") == "application/jsonl"
and extracted_file_data_content is not None
if FilesAPIUtils.is_batch_jsonl_file(
create_file_data=create_file_data,
extracted_file_data=extracted_file_data,
):
## Transform JSONL content to Bedrock format
original_file_content = self._get_content_from_openai_file(
@ -357,6 +363,8 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
api_base=api_base,
optional_params=optional_params,
)
litellm_params["upload_url"] = api_base
# Return a dict that tells the HTTP handler exactly what to do
return {
@ -440,6 +448,56 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
return dict(aws_request.headers), signed_body
def _convert_https_url_to_s3_uri(self, https_url: str) -> tuple[str, str]:
"""
Convert HTTPS S3 URL to s3:// URI format.
Args:
https_url: HTTPS S3 URL (e.g., "https://s3.us-west-2.amazonaws.com/bucket/key")
Returns:
Tuple of (s3_uri, filename)
Example:
Input: "https://s3.us-west-2.amazonaws.com/litellm-proxy/file.jsonl"
Output: ("s3://litellm-proxy/file.jsonl", "file.jsonl")
"""
import re
# Match HTTPS S3 URL patterns
# Pattern 1: https://s3.region.amazonaws.com/bucket/key
# Pattern 2: https://bucket.s3.region.amazonaws.com/key
pattern1 = r"https://s3\.([^.]+)\.amazonaws\.com/([^/]+)/(.+)"
pattern2 = r"https://([^.]+)\.s3\.([^.]+)\.amazonaws\.com/(.+)"
match1 = re.match(pattern1, https_url)
match2 = re.match(pattern2, https_url)
if match1:
# Pattern: https://s3.region.amazonaws.com/bucket/key
region, bucket, key = match1.groups()
s3_uri = f"s3://{bucket}/{key}"
elif match2:
# Pattern: https://bucket.s3.region.amazonaws.com/key
bucket, region, key = match2.groups()
s3_uri = f"s3://{bucket}/{key}"
else:
# Fallback: try to extract bucket and key from URL path
from urllib.parse import urlparse
parsed = urlparse(https_url)
path_parts = parsed.path.lstrip('/').split('/', 1)
if len(path_parts) >= 2:
bucket, key = path_parts[0], path_parts[1]
s3_uri = f"s3://{bucket}/{key}"
else:
raise ValueError(f"Unable to parse S3 URL: {https_url}")
# Extract filename from key
filename = key.split("/")[-1] if "/" in key else key
return s3_uri, filename
def transform_create_file_response(
self,
model: Optional[str],
@ -452,21 +510,18 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
"""
# For S3 uploads, we typically get an ETag and other metadata
response_headers = raw_response.headers
# Extract S3 object information from the response
# S3 PUT object returns ETag and other metadata in headers
content_length = response_headers.get("Content-Length", "0")
# Extract bucket and key from the request URL or litellm_params
bucket_name = litellm_params.get("s3_bucket_name") or os.getenv("AWS_S3_BUCKET_NAME")
# Generate file ID in S3 format
object_key = getattr(logging_obj, 'object_key', None) or f"file-{int(time.time())}"
file_id = f"s3://{bucket_name}/{object_key}"
# Extract filename from object key
filename = object_key.split("/")[-1] if "/" in object_key else object_key
# Use the actual upload URL that was used for the S3 upload
upload_url = litellm_params.get("upload_url")
file_id: str = ""
filename: str = ""
if upload_url:
# Convert HTTPS S3 URL to s3:// URI format
file_id, filename = self._convert_https_url_to_s3_uri(upload_url)
return OpenAIFileObject(
purpose="batch", # Default purpose for Bedrock files
id=file_id,

View file

@ -2201,7 +2201,6 @@ class BaseLLMHTTPHandler:
litellm_params=litellm_params,
optional_params={},
)
if _is_async:
return self.async_create_file(
transformed_request=transformed_request,
@ -2218,6 +2217,7 @@ class BaseLLMHTTPHandler:
sync_httpx_client = _get_httpx_client()
else:
sync_httpx_client = client
if isinstance(transformed_request, dict) and "method" in transformed_request:
# Handle pre-signed requests (e.g., from Bedrock S3 uploads)
@ -2283,11 +2283,15 @@ class BaseLLMHTTPHandler:
provider_config=provider_config,
)
# Store the upload URL in litellm_params for the transformation method
litellm_params_with_url = dict(litellm_params)
litellm_params_with_url["upload_url"] = api_base
return provider_config.transform_create_file_response(
model=None,
raw_response=upload_response,
logging_obj=logging_obj,
litellm_params=litellm_params,
litellm_params=litellm_params_with_url,
)
async def async_create_file(
@ -2408,15 +2412,19 @@ class BaseLLMHTTPHandler:
_is_async: bool = False,
client: Optional[Union["HTTPHandler", "AsyncHTTPHandler"]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
model: Optional[str] = None,
) -> Union["LiteLLMBatch", Coroutine[Any, Any, "LiteLLMBatch"]]:
"""
Creates a batch using provider-specific batch creation process
"""
# get config from model, custom llm provider
if model is None:
raise ValueError("model is required for create_batch")
headers = provider_config.validate_environment(
api_key=api_key,
headers=headers,
model="",
model=model,
messages=[],
optional_params={},
litellm_params=litellm_params,
@ -2425,7 +2433,7 @@ class BaseLLMHTTPHandler:
api_base = provider_config.get_complete_batch_url(
api_base=api_base,
api_key=api_key,
model="",
model=model,
optional_params={},
litellm_params=litellm_params,
data=create_batch_data,
@ -2435,7 +2443,7 @@ class BaseLLMHTTPHandler:
# Get the transformed request data
transformed_request = provider_config.transform_create_batch_request(
model="",
model=model,
create_batch_data=create_batch_data,
litellm_params=litellm_params,
optional_params={},
@ -2495,7 +2503,7 @@ class BaseLLMHTTPHandler:
litellm_params_with_request = {**litellm_params, "original_batch_request": create_batch_data}
return provider_config.transform_create_batch_response(
model=None,
model=model,
raw_response=batch_response,
logging_obj=logging_obj,
litellm_params=litellm_params_with_request,
@ -2512,6 +2520,7 @@ class BaseLLMHTTPHandler:
client: Optional[Union["HTTPHandler", "AsyncHTTPHandler"]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
create_batch_data: Optional["CreateBatchRequest"] = None,
model: Optional[str] = None,
):
"""
Async version of create_batch
@ -2572,7 +2581,7 @@ class BaseLLMHTTPHandler:
litellm_params_with_request = {**litellm_params, "original_batch_request": create_batch_data or {}}
return provider_config.transform_create_batch_response(
model=None,
model=model,
raw_response=batch_response,
logging_obj=logging_obj,
litellm_params=litellm_params_with_request,

View file

@ -1,21 +1,155 @@
"""
Cost calculator for DeepSeek Chat models.
Cost calculator for Dashscope Chat models.
Handles prompt caching scenario.
Handles tiered pricing and prompt caching scenarios.
"""
from typing import Tuple
from dataclasses import dataclass
from typing import List, Optional, Tuple
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import Usage
from litellm.types.utils import ModelInfo, Usage
from litellm.utils import get_model_info
@dataclass
class TokenBreakdown:
"""Token breakdown for cost calculation."""
text_tokens: int
cached_tokens: int
completion_tokens: int
reasoning_tokens: int
def _extract_token_breakdown(usage: Usage) -> TokenBreakdown:
"""Extract token counts from usage, handling cached and reasoning tokens."""
cached_tokens = 0
if usage.prompt_tokens_details and hasattr(usage.prompt_tokens_details, "cached_tokens"):
cached_tokens = usage.prompt_tokens_details.cached_tokens or 0
text_tokens = usage.prompt_tokens - cached_tokens
reasoning_tokens = 0
if (hasattr(usage, "completion_tokens_details") and
usage.completion_tokens_details and
hasattr(usage.completion_tokens_details, "reasoning_tokens")):
reasoning_tokens = usage.completion_tokens_details.reasoning_tokens or 0
completion_tokens = (usage.completion_tokens or 0) - reasoning_tokens
return TokenBreakdown(text_tokens, cached_tokens, completion_tokens, reasoning_tokens)
def _calculate_tiered_cost(
tokens: int,
tiered_pricing: List[dict],
cost_key: str,
fallback_cost_key: Optional[str] = None
) -> float:
"""Calculate cost using tiered pricing structure.
Finds the appropriate tier based on token count and applies that tier's rate to all tokens.
"""
if not tiered_pricing or tokens <= 0:
return 0.0
# Find the appropriate tier for the token count
for tier in tiered_pricing:
tier_range = tier.get("range", [])
if len(tier_range) != 2:
continue
range_start, range_end = tier_range
# Check if tokens fall within this tier's range
if range_start <= tokens <= range_end:
cost_per_token = tier.get(cost_key) or tier.get(fallback_cost_key, 0)
return tokens * cost_per_token
# If no tier matches, use the last tier (highest tier)
if tiered_pricing:
last_tier = tiered_pricing[-1]
cost_per_token = last_tier.get(cost_key) or last_tier.get(fallback_cost_key, 0)
return tokens * cost_per_token
return 0.0
def _calculate_flat_cost(tokens: int, cost_per_token: float) -> float:
"""Calculate cost using flat pricing."""
return tokens * cost_per_token
def _calculate_prompt_cost(breakdown: TokenBreakdown, model_info: ModelInfo, tiered_pricing: Optional[List[dict]]) -> float:
"""Calculate total prompt cost including cached tokens."""
if tiered_pricing:
text_cost = _calculate_tiered_cost(
tokens=breakdown.text_tokens,
tiered_pricing=tiered_pricing,
cost_key="input_cost_per_token"
)
cache_cost = _calculate_tiered_cost(
tokens=breakdown.cached_tokens,
tiered_pricing=tiered_pricing,
cost_key="cache_read_input_token_cost"
)
return text_cost + cache_cost
input_cost = model_info.get("input_cost_per_token", 0.0)
cache_cost = model_info.get("cache_read_input_token_cost", input_cost) or input_cost
return (_calculate_flat_cost(tokens=breakdown.text_tokens, cost_per_token=input_cost) +
_calculate_flat_cost(tokens=breakdown.cached_tokens, cost_per_token=cache_cost))
def _calculate_completion_cost(breakdown: TokenBreakdown, model_info: ModelInfo, tiered_pricing: Optional[List[dict]]) -> float:
"""Calculate total completion cost including reasoning tokens."""
if tiered_pricing:
completion_cost = _calculate_tiered_cost(
tokens=breakdown.completion_tokens,
tiered_pricing=tiered_pricing,
cost_key="output_cost_per_token"
)
reasoning_cost = _calculate_tiered_cost(
tokens=breakdown.reasoning_tokens,
tiered_pricing=tiered_pricing,
cost_key="output_cost_per_reasoning_token",
fallback_cost_key="output_cost_per_token"
)
return completion_cost + reasoning_cost
output_cost = model_info.get("output_cost_per_token", 0.0)
reasoning_cost = model_info.get("output_cost_per_reasoning_token", output_cost) or output_cost
return (_calculate_flat_cost(tokens=breakdown.completion_tokens, cost_per_token=output_cost) +
_calculate_flat_cost(tokens=breakdown.reasoning_tokens, cost_per_token=reasoning_cost))
def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]:
"""
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
Follows the same logic as Anthropic's cost per token calculation.
Calculate cost per token for Dashscope models.
Supports both tiered and flat pricing with cached and reasoning tokens.
Args:
model: Model name without provider prefix
usage: LiteLLM Usage block
Returns:
Tuple[float, float] - (prompt_cost_in_usd, completion_cost_in_usd)
"""
return generic_cost_per_token(
model=model, usage=usage, custom_llm_provider="deepseek"
model_info = get_model_info(model=model, custom_llm_provider="dashscope")
breakdown = _extract_token_breakdown(usage)
tiered_pricing = model_info.get("tiered_pricing") if isinstance(model_info.get("tiered_pricing"), list) else None
prompt_cost = _calculate_prompt_cost(
breakdown=breakdown,
model_info=model_info,
tiered_pricing=tiered_pricing
)
completion_cost = _calculate_completion_cost(
breakdown=breakdown,
model_info=model_info,
tiered_pricing=tiered_pricing
)
return prompt_cost, completion_cost

View file

@ -169,18 +169,20 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
if tool is None:
return None
kwags: dict = {
# Build DatabricksFunction explicitly to avoid parameter conflicts
function_params: DatabricksFunction = {
"name": tool["name"],
"parameters": cast(dict, tool.get("input_schema") or {})
}
# Only add description if it exists
description = tool.get("description")
if description is not None:
kwags["description"] = cast(Union[dict, str], description)
function_params["description"] = cast(Union[dict, str], description)
return DatabricksTool(
type="function",
function=DatabricksFunction(name=tool["name"], **kwags),
function=function_params,
)
def _map_openai_to_dbrx_tool(self, model: str, tools: List) -> List[DatabricksTool]:

View file

@ -11,7 +11,39 @@ if TYPE_CHECKING:
else:
GenerateContentContentListUnionDict = Any
class GoogleAIStudioTokenCounter:
def _clean_contents_for_gemini_api(self, contents: Any) -> Any:
"""
Clean up contents to remove unsupported fields for the Gemini API.
The Google Gemini API doesn't recognize the 'id' field in function responses,
so we need to remove it to prevent 400 Bad Request errors.
Args:
contents: The contents to clean up
Returns:
Cleaned contents with unsupported fields removed
"""
import copy
from google.genai.types import FunctionResponse
cleaned_contents = copy.deepcopy(contents)
for content in cleaned_contents:
parts = content["parts"]
for part in parts:
if "functionResponse" in part:
function_response_data = part["functionResponse"]
function_response_part = FunctionResponse(**function_response_data)
function_response_part.id = None
part["functionResponse"] = function_response_part.model_dump(
exclude_none=True
)
return cleaned_contents
def _construct_url(self, model: str, api_base: Optional[str] = None) -> str:
"""
@ -20,7 +52,6 @@ class GoogleAIStudioTokenCounter:
base_url = api_base or "https://generativelanguage.googleapis.com"
return f"{base_url}/v1beta/models/{model}:countTokens"
async def validate_environment(
self,
api_base: Optional[str] = None,
@ -33,7 +64,8 @@ class GoogleAIStudioTokenCounter:
Returns a Tuple of headers and url for the Google Gen AI Studio countTokens endpoint.
"""
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
headers = GoogleGenAIConfig().validate_environment(
headers = GoogleGenAIConfig().validate_environment(
api_key=api_key,
headers=headers,
model=model,
@ -54,7 +86,7 @@ class GoogleAIStudioTokenCounter:
) -> Dict[str, Any]:
"""
Count tokens using Google Gen AI Studio countTokens endpoint.
Args:
contents: The content to count tokens for (Google Gen AI format)
Example: [{"parts": [{"text": "Hello world"}]}]
@ -63,7 +95,7 @@ class GoogleAIStudioTokenCounter:
api_base: Optional API base URL (defaults to Google Gen AI Studio)
timeout: Optional timeout for the request
**kwargs: Additional parameters
Returns:
Dict containing token count information from Google Gen AI Studio API.
Example response:
@ -77,14 +109,13 @@ class GoogleAIStudioTokenCounter:
}
]
}
Raises:
ValueError: If API key is missing
litellm.APIError: If the API call fails
litellm.APIConnectionError: If the connection fails
Exception: For any other unexpected errors
"""
# Set up API base URL
# Prepare headers
headers, url = await self.validate_environment(
@ -94,46 +125,40 @@ class GoogleAIStudioTokenCounter:
model=model,
litellm_params=kwargs,
)
# Prepare request body
request_body = {
"contents": contents
}
# Prepare request body - clean up contents to remove unsupported fields
cleaned_contents = self._clean_contents_for_gemini_api(contents)
request_body = {"contents": cleaned_contents}
async_httpx_client = get_async_httpx_client(
llm_provider=LlmProviders.GEMINI,
)
try:
response = await async_httpx_client.post(
url=url,
headers=headers,
json=request_body
url=url, headers=headers, json=request_body
)
# Check for HTTP errors
response.raise_for_status()
# Parse response
result = response.json()
return result
except httpx.HTTPStatusError as e:
error_msg = f"Google Gen AI Studio API error: {e.response.status_code} - {e.response.text}"
raise litellm.APIError(
message=error_msg,
llm_provider="gemini",
model=model,
status_code=e.response.status_code
status_code=e.response.status_code,
) from e
except httpx.RequestError as e:
error_msg = f"Request to Google Gen AI Studio failed: {str(e)}"
raise litellm.APIConnectionError(
message=error_msg,
llm_provider="gemini",
model=model
message=error_msg, llm_provider="gemini", model=model
) from e
except Exception as e:
error_msg = f"Unexpected error during token counting: {str(e)}"
raise Exception(error_msg) from e

View file

@ -6,6 +6,7 @@ from typing import Any, Dict, List, Optional, Tuple, Union
from httpx import Headers, Response
from litellm.files.utils import FilesAPIUtils
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.files.transformation import (
@ -260,10 +261,13 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
raise ValueError("file is required")
extracted_file_data = extract_file_data(file_data)
extracted_file_data_content = extracted_file_data.get("content")
if (
create_file_data.get("purpose") == "batch"
and extracted_file_data.get("content_type") == "application/jsonl"
and extracted_file_data_content is not None
if extracted_file_data_content is None:
raise ValueError("file content is required")
if FilesAPIUtils.is_batch_jsonl_file(
create_file_data=create_file_data,
extracted_file_data=extracted_file_data,
):
## 1. If jsonl, check if there's a model name
file_content = self._get_content_from_openai_file(

View file

@ -1,7 +1,7 @@
"""
Transformation for Calling Google models in their native format.
"""
from typing import Literal, Optional, Union
from typing import Dict, Literal, Optional, Union
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
from litellm.types.router import GenericLiteLLMParams
@ -11,20 +11,20 @@ class VertexAIGoogleGenAIConfig(GoogleGenAIConfig):
"""
Configuration for calling Google models in their native format.
"""
HEADER_NAME = "Authorization"
BEARER_PREFIX = "Bearer"
@property
def custom_llm_provider(self) -> Literal["gemini", "vertex_ai"]:
return "vertex_ai"
def validate_environment(
self,
self,
api_key: Optional[str],
headers: Optional[dict],
model: str,
litellm_params: Optional[Union[GenericLiteLLMParams, dict]]
litellm_params: Optional[Union[GenericLiteLLMParams, dict]],
) -> dict:
default_headers = {
"Content-Type": "application/json",
@ -36,4 +36,65 @@ class VertexAIGoogleGenAIConfig(GoogleGenAIConfig):
default_headers.update(headers)
return default_headers
def _camel_to_snake(self, camel_str: str) -> str:
"""Convert camelCase to snake_case"""
import re
return re.sub(r"(?<!^)(?=[A-Z])", "_", camel_str).lower()
def map_generate_content_optional_params(
self,
generate_content_config_dict,
model: str,
):
"""
Map Google GenAI parameters to provider-specific format.
Args:
generate_content_optional_params: Optional parameters for generate content
model: The model name
Returns:
Mapped parameters for the provider
"""
from litellm.types.google_genai.main import GenerateContentConfigDict
_generate_content_config_dict = GenerateContentConfigDict()
for param, value in generate_content_config_dict.items():
camel_case_key = self._camel_to_snake(param)
_generate_content_config_dict[camel_case_key] = value
return dict(_generate_content_config_dict)
def transform_generate_content_request(
self,
model: str,
contents: any,
tools: Optional[any],
generate_content_config_dict: Dict,
system_instruction: Optional[any] = None,
) -> dict:
"""
Transform the generate content request for Vertex AI.
Since Vertex AI natively supports Google GenAI format, we can pass most fields directly.
"""
# Build the request in Google GenAI format that Vertex AI expects
result = {
"model": model,
"contents": contents,
}
# Add tools if provided
if tools:
result["tools"] = tools
# Add systemInstruction if provided
if system_instruction:
result["systemInstruction"] = system_instruction
# Handle generationConfig - Vertex AI expects it in the same format
if generate_content_config_dict:
result["generationConfig"] = generate_content_config_dict
return result

View file

@ -150,9 +150,9 @@ from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from .llms.custom_llm import CustomLLM, custom_chat_llm_router
from .llms.databricks.embed.handler import DatabricksEmbeddingHandler
from .llms.deprecated_providers import aleph_alpha, palm
from .llms.gemini.common_utils import get_api_key_from_env
from .llms.groq.chat.handler import GroqChatCompletion
from .llms.heroku.chat.transformation import HerokuChatConfig
from .llms.gemini.common_utils import get_api_key_from_env
from .llms.huggingface.embedding.handler import HuggingFaceEmbedding
from .llms.nlp_cloud.chat.handler import completion as nlp_cloud_chat_completion
from .llms.oci.chat.transformation import OCIChatConfig
@ -358,7 +358,9 @@ async def acompletion(
logprobs: Optional[bool] = None,
top_logprobs: Optional[int] = None,
deployment_id=None,
reasoning_effort: Optional[Literal["none", "minimal", "low", "medium", "high", "default"]] = None,
reasoning_effort: Optional[
Literal["none", "minimal", "low", "medium", "high", "default"]
] = None,
safety_identifier: Optional[str] = None,
# set api_base, api_version, api_key
base_url: Optional[str] = None,
@ -504,7 +506,9 @@ async def acompletion(
}
if custom_llm_provider is None:
_, custom_llm_provider, _, _ = get_llm_provider(
model=model, custom_llm_provider=custom_llm_provider, api_base=completion_kwargs.get("base_url", None)
model=model,
custom_llm_provider=custom_llm_provider,
api_base=completion_kwargs.get("base_url", None),
)
fallbacks = fallbacks or litellm.model_fallbacks
@ -899,7 +903,9 @@ def completion( # type: ignore # noqa: PLR0915
logit_bias: Optional[dict] = None,
user: Optional[str] = None,
# openai v1.0+ new params
reasoning_effort: Optional[Literal["none", "minimal", "low", "medium", "high", "default"]] = None,
reasoning_effort: Optional[
Literal["none", "minimal", "low", "medium", "high", "default"]
] = None,
response_format: Optional[Union[dict, Type[BaseModel]]] = None,
seed: Optional[int] = None,
tools: Optional[List] = None,
@ -1116,10 +1122,12 @@ def completion( # type: ignore # noqa: PLR0915
)
if provider_specific_header is not None:
headers.update(ProviderSpecificHeaderUtils.get_provider_specific_headers(
provider_specific_header=provider_specific_header,
custom_llm_provider=custom_llm_provider,
))
headers.update(
ProviderSpecificHeaderUtils.get_provider_specific_headers(
provider_specific_header=provider_specific_header,
custom_llm_provider=custom_llm_provider,
)
)
if model_response is not None and hasattr(model_response, "_hidden_params"):
model_response._hidden_params["custom_llm_provider"] = custom_llm_provider
@ -1325,6 +1333,7 @@ def completion( # type: ignore # noqa: PLR0915
azure_scope=kwargs.get("azure_scope"),
max_retries=max_retries,
timeout=timeout,
litellm_request_debug=kwargs.get("litellm_request_debug", False),
)
cast(LiteLLMLoggingObj, logging).update_environment_variables(
model=model,
@ -2712,9 +2721,7 @@ def completion( # type: ignore # noqa: PLR0915
)
api_key = (
api_key
or litellm.api_key
or get_secret("VERCEL_AI_GATEWAY_API_KEY")
api_key or litellm.api_key or get_secret("VERCEL_AI_GATEWAY_API_KEY")
)
vercel_site_url = get_secret("VERCEL_SITE_URL") or "https://litellm.ai"
@ -2730,7 +2737,7 @@ def completion( # type: ignore # noqa: PLR0915
vercel_headers.update(_headers)
headers = vercel_headers
## Load Config
config = litellm.VercelAIGatewayConfig.get_config()
for k, v in config.items():
@ -3712,7 +3719,9 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse:
func_with_context = partial(ctx.run, func)
_, custom_llm_provider, _, _ = get_llm_provider(
model=model, custom_llm_provider=custom_llm_provider, api_base=kwargs.get("api_base", None)
model=model,
custom_llm_provider=custom_llm_provider,
api_base=kwargs.get("api_base", None),
)
# Await normally
@ -5780,7 +5789,14 @@ async def ahealth_check(
input=input or ["test"],
),
"audio_speech": lambda: litellm.aspeech(
**{**_filter_model_params(model_params), **({"voice": "alloy"} if "voice" not in _filter_model_params(model_params) else {})},
**{
**_filter_model_params(model_params),
**(
{"voice": "alloy"}
if "voice" not in _filter_model_params(model_params)
else {}
),
},
input=prompt or "test",
),
"audio_transcription": lambda: litellm.atranscription(

View file

@ -6,6 +6,7 @@
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
"output_cost_per_reasoning_token": 0.0,
"input_cost_per_audio_token": 0.0,
"litellm_provider": "one of https://docs.litellm.ai/docs/providers",
"mode": "one of: chat, embedding, completion, image_generation, audio_transcription, audio_speech, image_generation, moderation, rerank",
"supports_function_calling": true,
@ -19045,34 +19046,43 @@
"max_tokens": 32768,
"max_input_tokens": 30720,
"max_output_tokens": 8192,
"input_cost_per_token": 1.6e-06,
"output_cost_per_token": 6.4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-latest": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo-latest": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 16384,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"output_cost_per_reasoning_token": 5e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-30b-a3b": {
"max_tokens": 131072,
@ -19083,7 +19093,272 @@
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-max-preview": {
"max_tokens": 262144,
"max_input_tokens": 258048,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 6e-06},
{"range": [32e3, 128e3], "input_cost_per_token": 2.4e-06, "output_cost_per_token": 1.2e-05},
{"range": [128e3, 252e3], "input_cost_per_token": 3.0e-06, "output_cost_per_token": 1.5e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-flash": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-coder": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 16384,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.5e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-plus": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06, "cache_read_input_token_cost": 1e-07},
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06, "cache_read_input_token_cost": 1.8e-07},
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "cache_read_input_token_cost": 3e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05, "cache_read_input_token_cost": 6e-07}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-plus-2025-07-22": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06},
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06},
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05},
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-flash": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06, "cache_read_input_token_cost": 8e-08},
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06, "cache_read_input_token_cost": 1.2e-07},
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06, "cache_read_input_token_cost": 2e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06, "cache_read_input_token_cost": 4e-07}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-flash-2025-07-28": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06},
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06},
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-09-11": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-07-28": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-07-14": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"output_cost_per_reasoning_token": 4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-04-28": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"output_cost_per_reasoning_token": 4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-01-25": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 8192,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-flash-2025-07-28": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"output_cost_per_reasoning_token": 5e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo-2025-04-28": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 16384,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"output_cost_per_reasoning_token": 5e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo-2024-11-01": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 8192,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwq-plus": {
"max_tokens": 131072,
"max_input_tokens": 98304,
"max_output_tokens": 8192,
"input_cost_per_token": 8e-07,
"output_cost_per_token": 2.4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"moonshot/moonshot-v1-8k": {
"max_tokens": 8192,

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

Binary file not shown.

After

Width:  |  Height:  |  Size: 48 KiB

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[75832,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","220","static/chunks/220-1c8d82f7ce7658c4.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-8dc8d9524a1f3965.js"],"default",1]
3:I[30628,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","220","static/chunks/220-5061c4cea850d728.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-127adcf8da2b5294.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -2,6 +2,6 @@
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","50","static/chunks/50-fe160ecfa8bc4059.js","154","static/chunks/154-fff436ed72b19a24.js","461","static/chunks/app/onboarding/page-3c5840c907b0a5c8.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -1617,6 +1617,20 @@ class ConfigList(LiteLLMPydanticObjectBase):
)
class UserHeaderMapping(LiteLLMPydanticObjectBase):
"""
Map an incoming HTTP header to a LiteLLM user role.
"""
header_name: str
litellm_user_role: Literal[
LitellmUserRoles.INTERNAL_USER,
LitellmUserRoles.CUSTOMER,
]
model_config = {
"extra": "forbid",
}
class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
"""
Documents all the fields supported by `general_settings` in config.yaml
@ -1721,6 +1735,11 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
default=None,
description="Set-up pass-through endpoints for provider-specific endpoints. Docs - https://docs.litellm.ai/docs/proxy/pass_through",
)
user_header_name: Optional[str] = Field(
None,
description="[DEPRECATED] Use 'user_header_mappings' instead. When set, the header value is treated as the end user id unless overridden by user_header_mappings.",
)
user_header_mappings: Optional[List[UserHeaderMapping]] = None
class ConfigYAML(LiteLLMPydanticObjectBase):

View file

@ -473,6 +473,22 @@ def _has_user_setup_sso():
return sso_setup
def get_customer_user_header_from_mapping(user_id_mapping) -> Optional[str]:
"""Return the header_name mapped to CUSTOMER role, if any (dict-based)."""
if not user_id_mapping:
return None
items = user_id_mapping if isinstance(user_id_mapping, list) else [user_id_mapping]
for item in items:
if not isinstance(item, dict):
continue
role = item.get("litellm_user_role")
header_name = item.get("header_name")
if role is None or not header_name:
continue
if str(role).lower() == str(LitellmUserRoles.CUSTOMER).lower():
return header_name
return None
def get_end_user_id_from_request_body(
request_body: dict, request_headers: Optional[dict] = None
@ -481,20 +497,34 @@ def get_end_user_id_from_request_body(
# and to ensure it's fetched at runtime.
from litellm.proxy.proxy_server import general_settings
# Check 1: Custom Header from general_settings.user_header_name (only if request_headers is provided)
# Check 1 : Follow the user header mappings feature, if not found, then check for deprecated user_header_name (only if request_headers is provided)
# User query: "system not respecting user_header_name property"
# This implies the key in general_settings is 'user_header_name'.
if request_headers is not None:
user_id_header_config_key = "user_header_name"
custom_header_name_to_check: Optional[str] = None
custom_header_name_to_check = general_settings.get(user_id_header_config_key)
# Prefer user mappings (new behavior)
user_id_mapping = general_settings.get("user_header_mappings", None)
if user_id_mapping:
custom_header_name_to_check = get_customer_user_header_from_mapping(
user_id_mapping
)
if custom_header_name_to_check and isinstance(custom_header_name_to_check, str):
# Fallback to deprecated user_header_name if mapping did not specify
if not custom_header_name_to_check:
user_id_header_config_key = "user_header_name"
value = general_settings.get(user_id_header_config_key)
if isinstance(value, str) and value.strip() != "":
custom_header_name_to_check = value
# If we have a header name to check, try to read it from request headers
if isinstance(custom_header_name_to_check, str):
for header_name, header_value in request_headers.items():
if header_name.lower() == custom_header_name_to_check.lower():
user_id_from_header = header_value
if user_id_from_header.strip():
return str(user_id_from_header)
user_id_str = str(user_id_from_header) if user_id_from_header is not None else ""
if user_id_str.strip():
return user_id_str
# Check 2: 'user' field in request_body (commonly OpenAI)
if "user" in request_body and request_body["user"] is not None:

View file

@ -18,6 +18,19 @@ model_list:
litellm_params:
model: "groq/*"
api_key: os.environ/GROQ_API_KEY
- model_name: bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0
litellm_params:
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
#########################################################
########## batch specific params ########################
s3_bucket_name: litellm-proxy
s3_region_name: us-west-2
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
model_info:
mode: batch
litellm_settings:
# set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production
drop_params: True

View file

@ -17,13 +17,13 @@ except ImportError:
# List of all available hooks that can be enabled
PROXY_HOOKS = {
"max_budget_limiter": _PROXY_MaxBudgetLimiter,
"parallel_request_limiter": _PROXY_MaxParallelRequestsHandler,
"parallel_request_limiter": _PROXY_MaxParallelRequestsHandler_v3,
"cache_control_check": _PROXY_CacheControlCheck,
}
## FEATURE FLAG HOOKS ##
if os.getenv("EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING", "false").lower() == "true":
PROXY_HOOKS["parallel_request_limiter"] = _PROXY_MaxParallelRequestsHandler_v3
if os.getenv("LEGACY_MULTI_INSTANCE_RATE_LIMITING", "false").lower() == "true":
PROXY_HOOKS["parallel_request_limiter"] = _PROXY_MaxParallelRequestsHandler
### update PROXY_HOOKS with ENTERPRISE_PROXY_HOOKS ###

View file

@ -6,6 +6,7 @@ This is currently in development and not yet ready for production.
import os
from datetime import datetime
from math import floor
from typing import (
TYPE_CHECKING,
Any,
@ -17,7 +18,7 @@ from typing import (
Union,
cast,
)
from math import floor
from fastapi import HTTPException
from litellm import DualCache
@ -95,6 +96,7 @@ end
return results
"""
class RateLimitDescriptorRateLimitObject(TypedDict, total=False):
requests_per_unit: Optional[int]
tokens_per_unit: Optional[int]
@ -266,7 +268,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
if current_limit is None or rate_limit_type is None:
continue
if counter_value is not None and int(counter_value) + 1 > current_limit:
if counter_value is not None and int(counter_value) > current_limit:
overall_code = "OVER_LIMIT"
item_code = "OVER_LIMIT"
@ -480,10 +482,15 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
},
)
)
# Team Member rate limits
if user_api_key_dict.user_id and (user_api_key_dict.team_member_rpm_limit is not None or user_api_key_dict.team_member_tpm_limit is not None):
team_member_value = f"{user_api_key_dict.team_id}:{user_api_key_dict.user_id}"
if user_api_key_dict.user_id and (
user_api_key_dict.team_member_rpm_limit is not None
or user_api_key_dict.team_member_tpm_limit is not None
):
team_member_value = (
f"{user_api_key_dict.team_id}:{user_api_key_dict.user_id}"
)
descriptors.append(
RateLimitDescriptor(
key="team_member",
@ -557,13 +564,13 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
# Find which descriptor hit the limit
for i, status in enumerate(response["statuses"]):
if status["code"] == "OVER_LIMIT":
descriptor = descriptors[floor(i/2)]
descriptor = descriptors[floor(i / 2)]
raise HTTPException(
status_code=429,
detail=f"Rate limit exceeded for {descriptor['key']}: {descriptor['value']}. Remaining: {status['limit_remaining']}",
headers={
"retry-after": str(self.window_size),
"rate_limit_type": str(status["rate_limit_type"])
"rate_limit_type": str(status["rate_limit_type"]),
}, # Retry after 1 minute
)
@ -613,7 +620,9 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
# Check if script is available
if self.token_increment_script is None:
verbose_proxy_logger.debug("TTL preservation script not available, using regular pipeline")
verbose_proxy_logger.debug(
"TTL preservation script not available, using regular pipeline"
)
await self.internal_usage_cache.dual_cache.async_increment_cache_pipeline(
increment_list=pipeline_operations,
litellm_parent_otel_span=parent_otel_span,
@ -628,7 +637,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
for op in pipeline_operations:
# Convert None TTL to 0 for Lua script
ttl_value = op["ttl"] if op["ttl"] is not None else 0
verbose_proxy_logger.debug(
f"Executing TTL-preserving increment for key={op['key']}, "
f"increment={op['increment_value']}, ttl={ttl_value}"
@ -693,16 +702,15 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
)
# Get metadata from kwargs
user_api_key = kwargs["litellm_params"]["metadata"].get("user_api_key")
user_api_key_user_id = kwargs["litellm_params"]["metadata"].get(
"user_api_key_user_id"
litellm_metadata = kwargs["litellm_params"]["metadata"]
if litellm_metadata is None:
return
user_api_key = litellm_metadata.get("user_api_key")
user_api_key_user_id = litellm_metadata.get("user_api_key_user_id")
user_api_key_team_id = litellm_metadata.get("user_api_key_team_id")
user_api_key_end_user_id = kwargs.get("user") or litellm_metadata.get(
"user_api_key_end_user_id"
)
user_api_key_team_id = kwargs["litellm_params"]["metadata"].get(
"user_api_key_team_id"
)
user_api_key_end_user_id = kwargs.get("user") or kwargs["litellm_params"][
"metadata"
].get("user_api_key_end_user_id")
model_group = get_model_group_from_litellm_kwargs(kwargs)
# Get total tokens from response

View file

@ -17,6 +17,7 @@ from litellm.proxy._types import (
SpecialHeaders,
TeamCallbackMetadata,
UserAPIKeyAuth,
LitellmUserRoles,
)
from litellm.proxy.auth.route_checks import RouteChecks
from litellm.router import Router
@ -335,6 +336,22 @@ class LiteLLMProxyRequestSetup:
return value
return None
@staticmethod
def add_internal_user_from_user_mapping(general_settings: Optional[Dict], user_api_key_dict: UserAPIKeyAuth, headers: dict) -> UserAPIKeyAuth:
if general_settings is None:
return user_api_key_dict
user_header_mapping = general_settings.get("user_header_mappings")
if not user_header_mapping:
return user_api_key_dict
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(user_header_mapping)
if not header_name:
return user_api_key_dict
header_value = LiteLLMProxyRequestSetup._get_case_insensitive_header(headers, header_name)
if header_value:
user_api_key_dict.user_id = header_value
return user_api_key_dict
return user_api_key_dict
@staticmethod
def get_user_from_headers(
headers: dict, general_settings: Optional[Dict] = None
@ -428,6 +445,26 @@ class LiteLLMProxyRequestSetup:
data["headers"] = _headers
return data
@staticmethod
def get_internal_user_header_from_mapping(user_header_mapping) -> Optional[str]:
if not user_header_mapping:
return None
items = (
user_header_mapping
if isinstance(user_header_mapping, list)
else [user_header_mapping]
)
for item in items:
if not isinstance(item, dict):
continue
role = item.get("litellm_user_role")
header_name = item.get("header_name")
if role is None or not header_name:
continue
if str(role).lower() == str(LitellmUserRoles.INTERNAL_USER).lower():
return header_name
return None
@staticmethod
def add_litellm_data_for_backend_llm_call(
*,
@ -726,6 +763,8 @@ async def add_litellm_data_to_request( # noqa: PLR0915
data=data, headers=_headers, user_api_key_dict=user_api_key_dict
)
user_api_key_dict = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(general_settings, user_api_key_dict, _headers)
# Parse user info from headers
user = LiteLLMProxyRequestSetup.get_user_from_headers(_headers, general_settings)
if user is not None:

View file

@ -346,6 +346,7 @@ def handle_key_type(data: GenerateKeyRequest, data_json: dict) -> dict:
data_json["allowed_routes"] = ["info_routes"]
return data_json
async def validate_team_id_used_in_service_account_request(
team_id: Optional[str],
prisma_client: Optional[PrismaClient],
@ -358,13 +359,13 @@ async def validate_team_id_used_in_service_account_request(
status_code=400,
detail="team_id is required for service account keys. Please specify `team_id` in the request body.",
)
if prisma_client is None:
raise HTTPException(
status_code=400,
detail="prisma_client is required for service account keys. Please specify `prisma_client` in the request body.",
)
# check if team_id exists in the database
team = await prisma_client.db.litellm_teamtable.find_unique(
where={"team_id": team_id},
@ -376,6 +377,7 @@ async def validate_team_id_used_in_service_account_request(
)
return True
async def _common_key_generation_helper( # noqa: PLR0915
data: GenerateKeyRequest,
user_api_key_dict: UserAPIKeyAuth,
@ -557,7 +559,7 @@ async def _common_key_generation_helper( # noqa: PLR0915
status_code=400,
detail={
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {data.key}"
}
},
)
response = await generate_key_helper_fn(
@ -2885,7 +2887,10 @@ async def unblock_key(
param="key",
code=status.HTTP_400_BAD_REQUEST,
)
hashed_token = hash_token(token=data.key)
if data.key.startswith("sk-"):
hashed_token = hash_token(token=data.key)
else:
hashed_token = data.key
if litellm.store_audit_logs is True:
# make an audit log for key update

View file

@ -1,18 +1,13 @@
model_list:
- model_name: db-openai-endpoint
- model_name: bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0
litellm_params:
model: openai/*
api_base: https://exampleopenaiendpoint-production-0ee2.up.railway.app/
- model_name: bedrock/*
litellm_params:
model: bedrock/*
- model_name: openai/*
litellm_params:
model: openai/*
- model_name: gemini/*
litellm_params:
model: gemini/*
litellm_settings:
callbacks: ["cloudzero"]
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
#########################################################
########## batch specific params ########################
s3_bucket_name: litellm-proxy
s3_region_name: us-west-2
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
model_info:
mode: batch

View file

@ -85,6 +85,7 @@ async def route_request(
"""
team_id = get_team_id_from_data(data)
router_model_names = llm_router.model_names if llm_router is not None else []
if "api_key" in data or "api_base" in data:
if llm_router is not None:
return getattr(llm_router, f"{route_type}")(**data)
@ -123,24 +124,20 @@ async def route_request(
data["model"] in router_model_names
or data["model"] in llm_router.get_model_ids()
):
return getattr(llm_router, f"{route_type}")(**data)
elif (
llm_router.model_group_alias is not None
and data["model"] in llm_router.model_group_alias
):
return getattr(llm_router, f"{route_type}")(**data)
elif data["model"] in llm_router.deployment_names:
return getattr(llm_router, f"{route_type}")(
**data, specific_deployment=True
)
elif data["model"] not in router_model_names:
if llm_router.router_general_settings.pass_through_all_models:
return getattr(litellm, f"{route_type}")(**data)
elif (
@ -162,7 +159,6 @@ async def route_request(
elif user_model is not None:
return getattr(litellm, f"{route_type}")(**data)
elif route_type == "allm_passthrough_route":
return getattr(litellm, f"{route_type}")(**data)
# if no route found then it's a bad request

View file

@ -10,6 +10,7 @@ from fastapi import APIRouter, Depends, HTTPException, status
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.router_strategy.budget_limiter import RouterBudgetLimiting
from litellm.proxy._types import *
from litellm.proxy._types import ProviderBudgetResponse, ProviderBudgetResponseObject
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
@ -2765,16 +2766,23 @@ async def provider_budgets() -> ProviderBudgetResponse:
provider_budget_response_dict: Dict[str, ProviderBudgetResponseObject] = {}
for _provider, _budget_info in provider_budget_config.items():
if llm_router.router_budget_logger is None:
router_budget_logger = next(
(
cb
for cb in (llm_router.optional_callbacks or [])
if isinstance(cb, RouterBudgetLimiting)
),
None,
)
if router_budget_logger is None:
raise ValueError("No router budget logger found")
_provider_spend = (
await llm_router.router_budget_logger._get_current_provider_spend(
await router_budget_logger._get_current_provider_spend(_provider) or 0.0
)
_provider_budget_ttl = (
await router_budget_logger._get_current_provider_budget_reset_at(
_provider
)
or 0.0
)
_provider_budget_ttl = await llm_router.router_budget_logger._get_current_provider_budget_reset_at(
_provider
)
provider_budget_response_object = ProviderBudgetResponseObject(
budget_limit=_budget_info.max_budget,

View file

@ -70,7 +70,8 @@ async def create_mcp_list_tools_events(
mcp_tools_dict = []
for tool in filtered_mcp_tools:
if hasattr(tool, 'model_dump') and callable(getattr(tool, 'model_dump')):
mcp_tools_dict.append(tool.model_dump())
# Type cast to help mypy understand this is safe after hasattr check
mcp_tools_dict.append(cast(Any, tool).model_dump())
elif hasattr(tool, '__dict__'):
mcp_tools_dict.append(tool.__dict__)
else:

View file

@ -3046,7 +3046,7 @@ class Router:
from litellm.router_utils.common_utils import add_model_file_id_mappings
verbose_router_logger.debug(
f"Inside _atext_completion()- model: {model}; kwargs: {kwargs}"
f"Inside _acreate_file()- model: {model}; kwargs: {kwargs}"
)
parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs)
healthy_deployments = await self.async_get_healthy_deployments(

View file

@ -162,6 +162,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
SearchContextCostPerQuery
] # Cost for using web search tool
citation_cost_per_token: Optional[float] # Cost per citation token for Perplexity
tiered_pricing: Optional[List[Dict[str, Any]]] # Tiered pricing structure for models like Dashscope
litellm_provider: Required[str]
mode: Required[
Literal[
@ -1995,7 +1996,7 @@ class StandardLoggingGuardrailInformation(TypedDict, total=False):
]
guardrail_request: Optional[dict]
guardrail_response: Optional[Union[dict, str, List[dict]]]
guardrail_status: Literal["success", "failure","blocked"]
guardrail_status: Literal["success", "failure", "blocked"]
start_time: Optional[float]
end_time: Optional[float]
duration: Optional[float]
@ -2123,6 +2124,7 @@ all_litellm_params = [
"metadata",
"litellm_metadata",
"litellm_trace_id",
"litellm_request_debug",
"guardrails",
"tags",
"acompletion",

View file

@ -2501,6 +2501,23 @@ def get_optional_params_transcription(
return optional_params
def _map_openai_size_to_vertex_ai_aspect_ratio(size: Optional[str]) -> str:
"""Map OpenAI size parameter to Vertex AI aspectRatio."""
if size is None:
return "1:1"
# Map OpenAI size strings to Vertex AI aspect ratio strings
# Vertex AI accepts: "1:1", "9:16", "16:9", "4:3", "3:4"
size_to_aspect_ratio = {
"256x256": "1:1", # Square
"512x512": "1:1", # Square
"1024x1024": "1:1", # Square (default)
"1792x1024": "16:9", # Landscape
"1024x1792": "9:16", # Portrait
}
return size_to_aspect_ratio.get(size, "1:1") # Default to square if size not recognized
def get_optional_params_image_gen(
model: Optional[str] = None,
n: Optional[int] = None,
@ -2614,19 +2631,7 @@ def get_optional_params_image_gen(
# Map OpenAI size parameter to Vertex AI aspectRatio
if size is not None:
# Map OpenAI size strings to Vertex AI aspect ratio strings
# Vertex AI accepts: "1:1", "9:16", "16:9", "4:3", "3:4"
size_to_aspect_ratio = {
"256x256": "1:1", # Square
"512x512": "1:1", # Square
"1024x1024": "1:1", # Square (default)
"1792x1024": "16:9", # Landscape
"1024x1792": "9:16", # Portrait
}
aspect_ratio = size_to_aspect_ratio.get(
size, "1:1"
) # Default to square if size not recognized
optional_params["aspectRatio"] = aspect_ratio
optional_params["aspectRatio"] = _map_openai_size_to_vertex_ai_aspect_ratio(size)
openai_params: list[str] = list(default_params.keys())
if provider_config is not None:
@ -2642,6 +2647,12 @@ def get_optional_params_image_gen(
openai_params=openai_params,
additional_drop_params=additional_drop_params,
)
# remove keys with None or empty dict/list values to avoid sending empty payloads
optional_params = {
k: v
for k, v in optional_params.items()
if v is not None and (not isinstance(v, (dict, list)) or len(v) > 0)
}
return optional_params
@ -4902,6 +4913,7 @@ def _get_model_info_helper( # noqa: PLR0915
citation_cost_per_token=_model_info.get(
"citation_cost_per_token", None
),
tiered_pricing=_model_info.get("tiered_pricing", None),
litellm_provider=_model_info.get(
"litellm_provider", custom_llm_provider
),

View file

@ -6,6 +6,7 @@
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
"output_cost_per_reasoning_token": 0.0,
"input_cost_per_audio_token": 0.0,
"litellm_provider": "one of https://docs.litellm.ai/docs/providers",
"mode": "one of: chat, embedding, completion, image_generation, audio_transcription, audio_speech, image_generation, moderation, rerank",
"supports_function_calling": true,
@ -19045,34 +19046,43 @@
"max_tokens": 32768,
"max_input_tokens": 30720,
"max_output_tokens": 8192,
"input_cost_per_token": 1.6e-06,
"output_cost_per_token": 6.4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-latest": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo-latest": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 16384,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"output_cost_per_reasoning_token": 5e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-30b-a3b": {
"max_tokens": 131072,
@ -19083,7 +19093,272 @@
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-max-preview": {
"max_tokens": 262144,
"max_input_tokens": 258048,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 6e-06},
{"range": [32e3, 128e3], "input_cost_per_token": 2.4e-06, "output_cost_per_token": 1.2e-05},
{"range": [128e3, 252e3], "input_cost_per_token": 3.0e-06, "output_cost_per_token": 1.5e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-flash": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-coder": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 16384,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.5e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-plus": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06, "cache_read_input_token_cost": 1e-07},
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06, "cache_read_input_token_cost": 1.8e-07},
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "cache_read_input_token_cost": 3e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05, "cache_read_input_token_cost": 6e-07}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-plus-2025-07-22": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06},
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06},
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05},
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-flash": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06, "cache_read_input_token_cost": 8e-08},
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06, "cache_read_input_token_cost": 1.2e-07},
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06, "cache_read_input_token_cost": 2e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06, "cache_read_input_token_cost": 4e-07}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen3-coder-flash-2025-07-28": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 65536,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06},
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06},
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-09-11": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-07-28": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-07-14": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"output_cost_per_reasoning_token": 4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-04-28": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"output_cost_per_reasoning_token": 4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-plus-2025-01-25": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 8192,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-flash-2025-07-28": {
"max_tokens": 1000000,
"max_input_tokens": 997952,
"max_output_tokens": 32768,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"tiered_pricing": [
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
],
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"output_cost_per_reasoning_token": 5e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo-2025-04-28": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 16384,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"output_cost_per_reasoning_token": 5e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwen-turbo-2024-11-01": {
"max_tokens": 1000000,
"max_input_tokens": 1000000,
"max_output_tokens": 8192,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 2e-07,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"dashscope/qwq-plus": {
"max_tokens": 131072,
"max_input_tokens": 98304,
"max_output_tokens": 8192,
"input_cost_per_token": 8e-07,
"output_cost_per_token": 2.4e-06,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
},
"moonshot/moonshot-v1-8k": {
"max_tokens": 8192,

View file

@ -1,3 +1,128 @@
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}

View file

@ -18,6 +18,8 @@ sys.path.insert(
import pytest
from typing import Optional
import litellm
from unittest.mock import patch, MagicMock
import httpx
@pytest.mark.asyncio()
@ -64,7 +66,55 @@ async def test_async_file_and_batch():
input_file_id=file_obj.id,
metadata={"key1": "value1", "key2": "value2"},
custom_llm_provider="bedrock",
#########################################################
# bedrock specific params
#########################################################
model="us.anthropic.claude-3-5-sonnet-20240620-v1:0",
aws_batch_role_arn="arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV"
)
print("CREATED BATCH RESPONSE=", create_batch_response)
@pytest.mark.asyncio()
async def test_mock_bedrock_file_url_mapping():
"""
Simple test to capture PUT URL and validate mapping to file ID.
"""
print("Testing Bedrock file URL mapping")
captured_put_url = None
async def mock_async_create_file(transformed_request, **kwargs):
nonlocal captured_put_url
# Capture PUT URL from transformed request
if isinstance(transformed_request, dict) and "url" in transformed_request:
captured_put_url = transformed_request["url"]
# Call the real method to get actual response
from litellm.files.main import base_llm_http_handler
return await base_llm_http_handler.__class__.async_create_file(
base_llm_http_handler, transformed_request, **kwargs
)
with patch('litellm.files.main.base_llm_http_handler.async_create_file', side_effect=mock_async_create_file):
file_obj = await litellm.acreate_file(
file=open(os.path.join(os.path.dirname(__file__), "bedrock_batch_completions.jsonl"), "rb"),
purpose="batch",
custom_llm_provider="bedrock",
s3_bucket_name="litellm-proxy",
)
print(f"PUT URL: {captured_put_url}")
print(f"File ID: {file_obj.id}")
# Validate URL was captured and response is correct
assert captured_put_url is not None
assert file_obj.id.startswith("s3://")
# Verify mapping
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
bedrock_config = BedrockFilesConfig()
expected_s3_uri, _ = bedrock_config._convert_https_url_to_s3_uri(captured_put_url)
assert file_obj.id == expected_s3_uri

View file

@ -0,0 +1,4 @@
{
"model": "gpt-image-1",
"prompt": "test prompt"
}

View file

@ -5,6 +5,7 @@ import logging
import os
import sys
import traceback
from unittest.mock import AsyncMock, patch
sys.path.insert(
@ -329,3 +330,33 @@ async def test_aiml_image_generation_with_dynamic_api_key():
assert captured_json_data is not None
assert captured_json_data["prompt"] == "A cute baby sea otter"
assert captured_json_data["model"] == "flux-pro/v1.1"
@pytest.mark.asyncio
async def test_azure_image_generation_request_body():
from litellm import aimage_generation
test_dir = os.path.dirname(__file__)
expected_path = os.path.join(
test_dir, "request_payloads", "azure_gpt_image_1.json"
)
with open(expected_path, "r") as f:
expected_body = json.load(f)
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new_callable=AsyncMock,
) as mock_post:
mock_post.side_effect = Exception("test")
with pytest.raises(Exception):
await aimage_generation(
model="azure/gpt-image-1",
prompt="test prompt",
api_base="https://example.azure.com",
api_key="test-key",
api_version="2025-04-01-preview",
)
mock_post.assert_called_once()
call_args = mock_post.call_args
request_json = call_args.kwargs.get("json", {})
assert request_json == expected_body

View file

@ -259,3 +259,55 @@ def test_get_model_from_request(request_data, expected_model):
model = get_model_from_request(request_data, "/v1/files")
assert model == ["gpt-3.5-turbo", "gpt-4o-mini-general-deployment"]
def test_get_customer_user_header_from_mapping_returns_customer_header():
from litellm.proxy.auth.auth_utils import get_customer_user_header_from_mapping
mappings = [
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
]
result = get_customer_user_header_from_mapping(mappings)
assert result == "X-OpenWebUI-User-Email"
def test_get_customer_user_header_from_mapping_no_customer_returns_none():
from litellm.proxy.auth.auth_utils import get_customer_user_header_from_mapping
mappings = [
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"}
]
result = get_customer_user_header_from_mapping(mappings)
assert result is None
# Also support a single mapping dict
single_mapping = {"header_name": "X-Only-Internal", "litellm_user_role": "internal_user"}
result = get_customer_user_header_from_mapping(single_mapping)
assert result is None
def test_get_internal_user_header_from_mapping_returns_internal_header():
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
mappings = [
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
]
result = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
assert result == "X-OpenWebUI-User-Id"
def test_get_internal_user_header_from_mapping_no_internal_returns_none():
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
mappings = [
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"}
]
result = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
assert result is None
# Also support single mapping dict
single_mapping = {"header_name": "X-Only-Customer", "litellm_user_role": "customer"}
result = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(single_mapping)
assert result is None

View file

@ -1,5 +1,7 @@
import os
import sys
import uuid
from functools import partial
from typing import Optional
import pytest
@ -146,24 +148,36 @@ async def test_pass_through_endpoint_rerank(client):
@pytest.mark.parametrize(
"auth, rpm_limit, expected_error_code",
[(True, 0, 429), (True, 1, 200), (False, 0, 200)],
"auth, rpm_limit, requests_to_make, expected_status_codes, num_users",
[
# Single user tests
(True, 0, 1, [429], 1),
(True, 1, 1, [200], 1),
(True, 1, 2, [200, 429], 1),
(True, 2, 4, [200, 200, 429, 429], 1),
(True, 3, 4, [200, 200, 200, 429], 1),
(True, 4, 4, [200, 200, 200, 200], 1),
(False, 0, 1, [200], 1),
(False, 0, 4, [200, 200, 200, 200], 1),
# Multiple user tests (same parameters as single user)
(True, 0, 1, [429], 2),
(True, 1, 1, [200], 2),
(True, 1, 2, [200, 429], 2),
(True, 2, 4, [200, 200, 429, 429], 2),
(True, 3, 4, [200, 200, 200, 429], 2),
(True, 4, 4, [200, 200, 200, 200], 2),
(False, 0, 1, [200], 2),
(False, 0, 4, [200, 200, 200, 200], 2),
],
)
@pytest.mark.asyncio
async def test_pass_through_endpoint_rpm_limit(
client, auth, expected_error_code, rpm_limit
client, auth, rpm_limit, requests_to_make, expected_status_codes, num_users
):
import litellm
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.proxy_server import ProxyLogging, hash_token, user_api_key_cache
mock_api_key = "sk-my-test-key"
cache_value = UserAPIKeyAuth(token=hash_token(mock_api_key), rpm_limit=rpm_limit)
_cohere_api_key = os.environ.get("COHERE_API_KEY")
user_api_key_cache.set_cache(key=hash_token(mock_api_key), value=cache_value)
proxy_logging_obj = ProxyLogging(user_api_key_cache=user_api_key_cache)
proxy_logging_obj._init_litellm_callbacks()
@ -173,6 +187,7 @@ async def test_pass_through_endpoint_rpm_limit(
setattr(litellm.proxy.proxy_server, "proxy_logging_obj", proxy_logging_obj)
# Define a pass-through endpoint
_cohere_api_key = os.environ.get("COHERE_API_KEY")
pass_through_endpoints = [
{
"path": "/v1/rerank",
@ -190,6 +205,13 @@ async def test_pass_through_endpoint_rpm_limit(
general_settings.update({"pass_through_endpoints": pass_through_endpoints})
setattr(litellm.proxy.proxy_server, "general_settings", general_settings)
# Setup API keys and cache
mock_api_keys = [f"sk-test-{uuid.uuid4().hex}" for _ in range(num_users)]
for mock_api_key in mock_api_keys:
cache_value = UserAPIKeyAuth(token=hash_token(mock_api_key), rpm_limit=rpm_limit)
user_api_key_cache.set_cache(key=hash_token(mock_api_key), value=cache_value)
_json_data = {
"model": "rerank-english-v3.0",
"query": "What is the capital of the United States?",
@ -200,16 +222,134 @@ async def test_pass_through_endpoint_rpm_limit(
}
# Make a request to the pass-through endpoint
response = client.post(
"/v1/rerank",
json=_json_data,
headers={"Authorization": "Bearer {}".format(mock_api_key)},
)
tasks = []
for mock_api_key in mock_api_keys:
for _ in range(requests_to_make):
task = asyncio.get_running_loop().run_in_executor(
None,
partial(
client.post,
"/v1/rerank",
json=_json_data,
headers={"Authorization": "Bearer {}".format(mock_api_key)},
),
)
tasks.append(task)
responses = await asyncio.gather(*tasks)
if num_users == 1:
status_codes = sorted([response.status_code for response in responses])
assert status_codes == sorted(expected_status_codes)
else:
first_user_responses = responses[requests_to_make:]
second_user_responses = responses[:requests_to_make]
first_user_status_codes = sorted([response.status_code for response in first_user_responses])
second_user_status_codes = sorted([response.status_code for response in second_user_responses])
expected_status_codes.sort()
assert first_user_status_codes == expected_status_codes
assert second_user_status_codes == expected_status_codes
print("JSON response: ", _json_data)
# Assert the response
assert response.status_code == expected_error_code
@pytest.mark.parametrize(
"auth, rpm_limit, requests_to_make, expected_status_codes",
[
# Multiple user tests (same parameters as single user)
(True, 0, 1, [429]),
(True, 1, 1, [200]),
(True, 1, 2, [200, 429]),
(True, 2, 4, [200, 200, 429, 429]),
(True, 3, 4, [200, 200, 200, 429]),
(True, 4, 4, [200, 200, 200, 200]),
(False, 0, 1, [200]),
(False, 0, 4, [200, 200, 200, 200]),
],
)
@pytest.mark.asyncio
async def test_pass_through_endpoint_sequential_rpm_limit(
client, auth, rpm_limit, requests_to_make, expected_status_codes
):
import litellm
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.proxy_server import ProxyLogging, hash_token, user_api_key_cache
proxy_logging_obj = ProxyLogging(user_api_key_cache=user_api_key_cache)
proxy_logging_obj._init_litellm_callbacks()
setattr(litellm.proxy.proxy_server, "user_api_key_cache", user_api_key_cache)
setattr(litellm.proxy.proxy_server, "master_key", "sk-1234")
setattr(litellm.proxy.proxy_server, "prisma_client", "FAKE-VAR")
setattr(litellm.proxy.proxy_server, "proxy_logging_obj", proxy_logging_obj)
# Define a pass-through endpoint
_cohere_api_key = os.environ.get("COHERE_API_KEY")
pass_through_endpoints = [
{
"path": "/v1/rerank",
"target": "https://api.cohere.com/v1/rerank",
"auth": auth,
"headers": {"Authorization": f"bearer {_cohere_api_key}"},
}
]
# Initialize the pass-through endpoint
await initialize_pass_through_endpoints(pass_through_endpoints)
general_settings: Optional[dict] = (
getattr(litellm.proxy.proxy_server, "general_settings", {}) or {}
)
general_settings.update({"pass_through_endpoints": pass_through_endpoints})
setattr(litellm.proxy.proxy_server, "general_settings", general_settings)
# Setup API keys and cache
mock_api_keys = [f"sk-test-{uuid.uuid4().hex}" for _ in range(2)]
for mock_api_key in mock_api_keys:
cache_value = UserAPIKeyAuth(token=hash_token(mock_api_key), rpm_limit=rpm_limit)
user_api_key_cache.set_cache(key=hash_token(mock_api_key), value=cache_value)
_json_data = {
"model": "rerank-english-v3.0",
"query": "What is the capital of the United States?",
"top_n": 3,
"documents": [
"Carson City is the capital city of the American state of Nevada."
],
}
# Make a request to the pass-through endpoint
first_user_responses = []
second_user_responses = []
for _ in range(requests_to_make):
requests = []
for mock_api_key in mock_api_keys:
task = asyncio.get_running_loop().run_in_executor(
None,
partial(
client.post,
"/v1/rerank",
json=_json_data,
headers={"Authorization": "Bearer {}".format(mock_api_key)},
),
)
requests.append(task)
first_user_response, second_user_response = await asyncio.gather(*requests)
first_user_responses.append(first_user_response)
second_user_responses.append(second_user_response)
first_user_status_codes = sorted([response.status_code for response in first_user_responses])
second_user_status_codes = sorted([response.status_code for response in second_user_responses])
expected_status_codes.sort()
assert first_user_status_codes == expected_status_codes
assert second_user_status_codes == expected_status_codes
print("JSON response: ", _json_data)
@pytest.mark.parametrize(

View file

@ -0,0 +1,2 @@
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}

View file

@ -0,0 +1,37 @@
from openai import OpenAI
import pytest
client = OpenAI(
base_url="http://0.0.0.0:4000",
api_key="sk-1234",
)
BEDROCK_BATCH_MODEL = "bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0"
@pytest.mark.asyncio
async def test_bedrock_batches_api():
"""
Test bedrock batches api
E2E Test Creating a File and a Batch on Bedrock
"""
# Upload file
batch_input_file = client.files.create(
file=open("tests/openai_endpoints_tests/bedrock_batch_completions.jsonl", "rb"),
purpose="batch",
extra_body={"target_model_names": BEDROCK_BATCH_MODEL}
)
print(batch_input_file)
# Create batch
batch = client.batches.create(
input_file_id=batch_input_file.id,
endpoint="/v1/chat/completions",
completion_window="24h",
metadata={"description": "Test batch job"},
)
print(batch)
assert batch.id is not None

View file

@ -0,0 +1,350 @@
"""
Test case for Google Gemini API proxy request handling.
This test verifies that when a request comes to the proxy endpoint:
http://localhost:4000/v1beta/models/gemini-2.5-flash:generateContent
The request payload is correctly processed and forwarded to the httpx client.
"""
import json
import os
import sys
import unittest.mock
from typing import Optional
from unittest.mock import AsyncMock, MagicMock, patch
import httpx
import pytest
# Add the parent directory to the system path
sys.path.insert(0, os.path.abspath("../.."))
import litellm
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.google_endpoints.endpoints import google_generate_content
from litellm.proxy.proxy_server import ProxyConfig
from litellm.proxy.utils import ProxyLogging
from fastapi import Request, Response
from fastapi.datastructures import Headers
@pytest.fixture
def sample_request_payload():
"""Sample request payload as provided in the user query."""
return {
"contents": [
{
"parts": [
{
"text": "You are an interactive CLI agent specializing in software engineering tasks. Your primary goal is to help users safely and efficiently, adhering strictly to the following instructions and utilizing your available tools"
}
],
"role": "user"
},
{
"parts": [{"text": "Got it. Thanks for the context!"}],
"role": "model"
},
{
"parts": [{"text": "Hello how are you"}],
"role": "user"
},
{
"parts": [{"text": "I'm doing well, thank you! How can I help you today?\n"}],
"role": "model"
},
{
"parts": [
{
"text": "Analyze *only* the content and structure of your immediately preceding response (your last turn in the conversation history)."
}
],
"role": "user"
}
],
"systemInstruction": {
"parts": [
{
"text": "You are an interactive CLI agent specializing in software engineering tasks. Your primary goal is to help users safely and efficiently, adhering strictly to the following instructions and utilizing your available tools"
}
],
"role": "user"
},
"generationConfig": {
"temperature": 0,
"topP": 1,
"responseMimeType": "application/json",
"responseJsonSchema": {
"type": "object",
"properties": {
"reasoning": {
"type": "string",
"description": "Brief explanation justifying the 'next_speaker' choice based *strictly* on the applicable rule and the content/structure of the preceding turn."
},
"next_speaker": {
"type": "string",
"enum": ["user", "model"],
"description": "Who should speak next based *only* on the preceding turn and the decision rules"
}
},
"required": ["reasoning", "next_speaker"]
}
}
}
@pytest.fixture
def mock_user_api_key_dict():
"""Mock user API key dictionary."""
return UserAPIKeyAuth(
api_key="test_api_key",
user_id="test_user_id",
user_email="test@example.com",
team_id="test_team_id",
max_budget=100.0,
spend=0.0,
user_role="internal_user",
allowed_cache_controls=[],
metadata={},
tpm_limit=None,
rpm_limit=None,
)
@pytest.fixture
def mock_request(sample_request_payload):
"""Create a mock FastAPI request with the sample payload."""
mock_request = MagicMock(spec=Request)
mock_request.headers = Headers({"content-type": "application/json"})
mock_request.method = "POST"
mock_request.url.path = "/v1beta/models/gemini-2.5-flash:generateContent"
# Mock the request body reading
async def mock_body():
return json.dumps(sample_request_payload).encode('utf-8')
mock_request.body = mock_body
return mock_request
@pytest.fixture
def mock_response():
"""Create a mock FastAPI response."""
return MagicMock(spec=Response)
@pytest.mark.asyncio
async def test_google_gemini_httpx_request_direct():
"""
Test that the Google Gemini generate_content_handler correctly processes the request
and forwards it to the httpx client with the correct parameters.
This test directly calls the HTTP handler to verify the httpx integration.
"""
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
# Sample request payload
sample_payload = {
"contents": [
{
"parts": [
{
"text": "You are an interactive CLI agent specializing in software engineering tasks."
}
],
"role": "user"
},
{
"parts": [{"text": "Got it. Thanks for the context!"}],
"role": "model"
},
{
"parts": [{"text": "Hello how are you"}],
"role": "user"
}
],
"systemInstruction": {
"parts": [
{
"text": "You are an interactive CLI agent specializing in software engineering tasks."
}
],
"role": "user"
},
"config": { # Note: already transformed from generationConfig
"temperature": 0,
"topP": 1,
"responseMimeType": "application/json",
"responseJsonSchema": {
"type": "object",
"properties": {
"reasoning": {"type": "string"},
"next_speaker": {"type": "string", "enum": ["user", "model"]}
},
"required": ["reasoning", "next_speaker"]
}
}
}
# Mock the HTTP handler to capture the request
with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post") as mock_post:
# Create mock response
mock_http_response = MagicMock()
mock_http_response.status_code = 200
mock_http_response.json.return_value = {
"candidates": [
{
"content": {
"parts": [
{
"text": '{"reasoning": "The preceding response was a helpful greeting asking how to assist.", "next_speaker": "user"}'
}
],
"role": "model"
}
}
]
}
mock_post.return_value = mock_http_response
# Create the HTTP handler and provider config
from litellm.types.router import GenericLiteLLMParams
http_handler = BaseLLMHTTPHandler()
provider_config = GoogleGenAIConfig()
# Create proper litellm params
litellm_params = GenericLiteLLMParams(
api_base="https://generativelanguage.googleapis.com",
api_key="test_api_key"
)
logging_obj = LiteLLMLoggingObj(
model="gemini/gemini-2.5-flash",
messages=[],
stream=False,
call_type="agenerate_content",
start_time=None,
litellm_call_id="test_call_id",
function_id="test_function_id"
)
try:
# Call the generate_content_handler directly
response = http_handler.generate_content_handler(
model="gemini/gemini-2.5-flash",
contents=sample_payload["contents"],
generate_content_provider_config=provider_config,
generate_content_config_dict=sample_payload["config"],
tools=None,
custom_llm_provider="gemini",
litellm_params=litellm_params,
logging_obj=logging_obj,
extra_headers=None,
extra_body=None,
timeout=30.0,
_is_async=False,
client=None,
stream=False,
litellm_metadata={}
)
# Verify that the HTTP post was called
assert mock_post.called, "Expected HTTP POST to be called"
# Get the call arguments
call_args, call_kwargs = mock_post.call_args
print(f"POST call args: {call_args}")
print(f"POST call kwargs: {call_kwargs}")
# Validate that the request data includes the expected fields
request_data = call_kwargs.get('json')
if request_data:
assert 'contents' in request_data, "Expected 'contents' in request data"
# The config should be included in the request as generationConfig
if 'generationConfig' in request_data:
config = request_data['generationConfig']
assert config['temperature'] == 0, "Expected temperature to be 0"
assert config['topP'] == 1, "Expected topP to be 1"
assert config['responseMimeType'] == "application/json", "Expected responseMimeType to be application/json"
assert 'responseJsonSchema' in config, "Expected responseJsonSchema in config"
# Validate the responseJsonSchema structure
schema = config['responseJsonSchema']
assert schema['type'] == 'object', "Expected schema type to be object"
assert 'properties' in schema, "Expected properties in schema"
assert 'reasoning' in schema['properties'], "Expected reasoning property in schema"
assert 'next_speaker' in schema['properties'], "Expected next_speaker property in schema"
print("✅ Request data validation passed")
print(f"Request data: {json.dumps(request_data, indent=2)}")
# Validate URL contains the correct endpoint
if call_args:
url = call_args[0] if len(call_args) > 0 else call_kwargs.get('url')
assert url is not None, "Expected URL to be provided"
print(f"✅ URL validation passed: {url}")
except Exception as e:
print(f"Exception occurred: {e}")
# Check if the HTTP handler was called despite the exception
if mock_post.called:
call_args, call_kwargs = mock_post.call_args
print(f"HTTP POST was called with args: {call_args}")
print(f"HTTP POST was called with kwargs: {call_kwargs}")
# Even with an exception, we can validate the request structure
request_data = call_kwargs.get('json')
if request_data:
assert 'contents' in request_data, "Expected 'contents' in request data"
if 'generationConfig' in request_data:
config = request_data['generationConfig']
assert config['temperature'] == 0, "Expected temperature to be 0"
assert config['responseMimeType'] == "application/json", "Expected responseMimeType to be application/json"
print("✅ Request structure validation passed despite exception")
else:
# If no HTTP call was made, re-raise the exception for debugging
raise
@pytest.mark.asyncio
async def test_generationconfig_to_config_mapping(sample_request_payload):
"""
Test that generationConfig is correctly mapped to config parameter
for Google GenAI compatibility in the main functions.
"""
from litellm.google_genai.main import agenerate_content
# Create a copy of the payload to avoid modifying the fixture
test_data = sample_request_payload.copy()
# Test that agenerate_content can handle generationConfig parameter
# This should not raise an error about parameter handling
try:
# This will fail due to missing API key, but should not fail due to parameter handling
await agenerate_content(
model="gemini/gemini-2.5-flash",
contents=test_data["contents"],
generationConfig=test_data["generationConfig"], # Pass as generationConfig
custom_llm_provider="gemini"
)
except Exception as e:
# Should not fail due to parameter handling issues
error_msg = str(e).lower()
if "generationconfig" in error_msg or "config" in error_msg or "parameter" in error_msg:
pytest.fail(f"Parameter handling failed: {e}")
# Other errors (like API key missing) are expected
print(f"✅ Parameter handling worked (API error expected): {type(e).__name__}")
print("✅ generationConfig to config mapping test passed")
if __name__ == "__main__":
# Run the tests
pytest.main([__file__, "-v"])

View file

@ -467,3 +467,73 @@ async def test_e2e_generate_cold_storage_object_key_not_configured():
assert result is None
@pytest.mark.asyncio
async def test_logging_opentelemetry_context_propagation():
"""
Test that OpenTelemtry context propagation works with async completion.
"""
import asyncio
import litellm
from litellm.integrations.custom_logger import CustomLogger
from opentelemetry import trace
from opentelemetry.sdk.trace import TracerProvider
from opentelemetry.sdk.trace.export import SimpleSpanProcessor
from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanExporter
provider = TracerProvider()
exporter = InMemorySpanExporter()
provider.add_span_processor(SimpleSpanProcessor(exporter))
trace.set_tracer_provider(provider)
tracer = trace.get_tracer(__name__)
class MockOpenTelemetryLogger(CustomLogger):
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
span = tracer.start_span(start_time=start_time.timestamp() * 1e9, name="async_log_success_event")
span.end(end_time=end_time)
mock_logging_obj = MockOpenTelemetryLogger()
litellm.callbacks = [mock_logging_obj]
with tracer.start_as_current_span("span_1") as span:
span_1_id = span.get_span_context().span_id
await litellm.acompletion(
max_tokens=100,
messages=[{"role": "user", "content": "Hey"}],
model="openai/codex-mini-latest",
mock_response="Hello, world!",
)
with tracer.start_as_current_span("span_2") as span:
span_2_id = span.get_span_context().span_id
await litellm.acompletion(
max_tokens=100,
messages=[{"role": "user", "content": "Hey"}],
model="openai/codex-mini-latest",
mock_response="Hello, world!",
)
await asyncio.sleep(1)
spans = exporter.get_finished_spans()
assert len(spans) == 4
assert span_1_id != span_2_id
sorted_spans = sorted(list(spans), key=lambda x: x.start_time or 0)
assert sorted_spans[0].name == "span_1"
assert sorted_spans[1].name == "async_log_success_event"
assert sorted_spans[2].name == "span_2"
assert sorted_spans[3].name == "async_log_success_event"
first_span_context = sorted_spans[0].get_span_context()
assert first_span_context is not None and first_span_context.span_id == span_1_id
second_span_context = sorted_spans[2].get_span_context()
assert second_span_context is not None and second_span_context.span_id == span_2_id
first_completion_span_parent = sorted_spans[1].parent
assert first_completion_span_parent is not None and first_completion_span_parent.span_id == span_1_id
# This check would fail without the proper context propagation, and span[3] would end up with span_1_id as the parent
second_completion_span_parent = sorted_spans[3].parent
assert second_completion_span_parent is not None and second_completion_span_parent.span_id == span_2_id

View file

@ -2,6 +2,7 @@
Tests for the LoggingWorker class to ensure graceful shutdown handling.
"""
import asyncio
import contextvars
import pytest
from unittest.mock import AsyncMock, patch
@ -139,3 +140,81 @@ class TestLoggingWorker:
# Should have logged queue full exceptions
exception_calls = [call for call in mock_logger.exception.call_args_list if "queue is full" in str(call)]
assert len(exception_calls) > 0
@pytest.mark.asyncio
async def test_context_propagation(self, logging_worker):
"""Test that enqueued tasks execute in their original context."""
# Create a context variable for testing
test_context_var: contextvars.ContextVar[str] = contextvars.ContextVar('test_context_var')
# Track results from multiple tasks
task_results = []
async def test_task(task_id: str):
"""A test coroutine that checks if it can access the context variable."""
# Sleep a bit to simulate real work and ensure context persists
await asyncio.sleep(0.1)
try:
# Try to get the context variable value
value = test_context_var.get()
task_results.append({
'task_id': task_id,
'context_value': value,
'context_accessible': True
})
except LookupError:
# Context variable not found
task_results.append({
'task_id': task_id,
'context_accessible': False,
'context_value': None
})
# Start the logging worker
logging_worker.start()
# Create two separate contexts and enqueue tasks from each
# Context 1: Set context var to "context_1"
ctx1 = contextvars.copy_context()
ctx1.run(test_context_var.set, "context_1")
ctx1.run(logging_worker.enqueue, test_task("task_1"))
# Context 2: Set context var to "context_2"
ctx2 = contextvars.copy_context()
ctx2.run(test_context_var.set, "context_2")
ctx2.run(logging_worker.enqueue, test_task("task_2"))
# Context 3: No context variable set (should get LookupError)
ctx3 = contextvars.copy_context()
ctx3.run(logging_worker.enqueue, test_task("task_3"))
# Wait for all tasks to be processed
await asyncio.sleep(0.5)
# Stop the worker
await logging_worker.stop()
# Sort results by task_id for consistent testing
task_results.sort(key=lambda x: x['task_id'])
# Verify that each task saw its own context
assert len(task_results) == 3, f"Expected 3 results, got {len(task_results)}"
# Task 1 should see "context_1"
task1_result = next((r for r in task_results if r['task_id'] == 'task_1'), None)
assert task1_result is not None, "Task 1 result not found"
assert task1_result['context_accessible'] is True, "Task 1 should have access to context variable"
assert task1_result['context_value'] == "context_1", f"Task 1 should see 'context_1', got: {task1_result['context_value']}"
# Task 2 should see "context_2"
task2_result = next((r for r in task_results if r['task_id'] == 'task_2'), None)
assert task2_result is not None, "Task 2 result not found"
assert task2_result['context_accessible'] is True, "Task 2 should have access to context variable"
assert task2_result['context_value'] == "context_2", f"Task 2 should see 'context_2', got: {task2_result['context_value']}"
# Task 3 should not have access to the context variable
task3_result = next((r for r in task_results if r['task_id'] == 'task_3'), None)
assert task3_result is not None, "Task 3 result not found"
assert task3_result['context_accessible'] is False, "Task 3 should not have access to context variable"

View file

@ -0,0 +1,343 @@
"""
Test for AnthropicStreamWrapper handling content blocks that exist after message_delta with stop_reason and usage.
This tests the scenario where a streaming response includes:
1. Initial content blocks
2. A message_delta chunk with stop_reason and usage
3. Additional content blocks after the stop_reason
The wrapper should properly handle this by:
- Holding the stop_reason chunk until usage is available
- Merging usage into the stop_reason chunk
- Properly managing content_block_stop/start events for subsequent content
"""
import os
import sys
from typing import List
import pytest
sys.path.insert(0, os.path.abspath("../../../../.."))
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
AnthropicStreamWrapper,
)
from litellm.types.utils import Delta, ModelResponse, StreamingChoices, Usage
class MockCompletionStreamWithContentAfterStopReason:
"""Mock stream that simulates content blocks existing after message_delta with stop_reason and usage."""
def __init__(self):
self.responses = [
# Initial text content
ModelResponse(
stream=True,
choices=[
StreamingChoices(
delta=Delta(content="Hello"), index=0, finish_reason=None
)
],
),
ModelResponse(
stream=True,
choices=[
StreamingChoices(
delta=Delta(content=" world"), index=0, finish_reason=None
)
],
),
# Message delta with stop_reason AND usage (this is how it actually comes from the API)
ModelResponse(
stream=True,
choices=[
StreamingChoices(
delta=Delta(content=""), index=0, finish_reason="stop"
)
],
usage=Usage(prompt_tokens=230, completion_tokens=65, total_tokens=295),
),
# Additional content after the stop_reason - this simulates the scenario
# where there might be additional content blocks after the main response
ModelResponse(
stream=True,
choices=[
StreamingChoices(
delta=Delta(content=" Additional content"),
index=0,
finish_reason=None,
)
],
),
]
self.index = 0
def __iter__(self):
return self
def __next__(self):
if self.index >= len(self.responses):
raise StopIteration
response = self.responses[self.index]
self.index += 1
return response
def __aiter__(self):
return self
async def __anext__(self):
if self.index >= len(self.responses):
raise StopAsyncIteration
response = self.responses[self.index]
self.index += 1
return response
def test_anthropic_stream_wrapper_content_after_stop_reason():
"""Test that AnthropicStreamWrapper properly handles content blocks after message_delta with stop_reason."""
wrapper = AnthropicStreamWrapper(
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
model="claude-3",
)
chunks = []
chunk_types = []
# Collect all chunks
for chunk in wrapper:
chunks.append(chunk)
chunk_types.append(chunk.get("type"))
# Verify the expected sequence of chunk types
expected_types = [
"message_start", # Initial message start
"content_block_start", # Start of first content block
"content_block_delta", # "Hello"
"content_block_delta", # " world"
"content_block_stop", # End of first content block due to stop_reason
"message_delta", # Stop reason with merged usage
"message_stop", # Final message stop
]
print(f"Actual chunk types: {chunk_types}")
print(f"Expected chunk types: {expected_types}")
# Verify we have the expected number of chunks
assert len(chunk_types) >= len(
expected_types
), f"Expected at least {len(expected_types)} chunks, got {len(chunk_types)}"
# Verify key chunk types are present
assert "message_start" in chunk_types
assert "content_block_start" in chunk_types
assert "content_block_delta" in chunk_types
assert "content_block_stop" in chunk_types
assert "message_delta" in chunk_types
assert "message_stop" in chunk_types
# Find the message_delta chunk with stop_reason
message_delta_chunk = None
for chunk in chunks:
if chunk.get("type") == "message_delta":
message_delta_chunk = chunk
break
assert message_delta_chunk is not None, "message_delta chunk not found"
# Verify that the message_delta chunk has both stop_reason and usage
delta = message_delta_chunk.get("delta", {})
usage = message_delta_chunk.get("usage", {})
assert (
delta.get("stop_reason") == "end_turn"
), f"Expected stop_reason 'end_turn', got {delta.get('stop_reason')}"
assert (
usage.get("input_tokens") == 230
), f"Expected input_tokens 230, got {usage.get('input_tokens')}"
assert (
usage.get("output_tokens") == 65
), f"Expected output_tokens 65, got {usage.get('output_tokens')}"
# Verify content_block_stop comes before message_delta
content_block_stop_index = None
message_delta_index = None
for i, chunk_type in enumerate(chunk_types):
if chunk_type == "content_block_stop" and content_block_stop_index is None:
content_block_stop_index = i
elif chunk_type == "message_delta":
message_delta_index = i
assert content_block_stop_index is not None, "content_block_stop not found"
assert message_delta_index is not None, "message_delta not found"
assert (
content_block_stop_index < message_delta_index
), "content_block_stop should come before message_delta"
@pytest.mark.asyncio
async def test_async_anthropic_stream_wrapper_content_after_stop_reason():
"""Test async version of AnthropicStreamWrapper handling content blocks after message_delta with stop_reason."""
wrapper = AnthropicStreamWrapper(
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
model="claude-3",
)
chunks = []
chunk_types = []
# Collect all chunks asynchronously
async for chunk in wrapper:
chunks.append(chunk)
chunk_types.append(chunk.get("type"))
print(f"Async - Actual chunk types: {chunk_types}")
# Verify key chunk types are present
assert "message_start" in chunk_types
assert "content_block_start" in chunk_types
assert "content_block_delta" in chunk_types
assert "content_block_stop" in chunk_types
assert "message_delta" in chunk_types
assert "message_stop" in chunk_types
# Find the message_delta chunk with stop_reason
message_delta_chunk = None
for chunk in chunks:
if chunk.get("type") == "message_delta":
message_delta_chunk = chunk
break
assert message_delta_chunk is not None, "message_delta chunk not found"
# Verify that the message_delta chunk has both stop_reason and usage
delta = message_delta_chunk.get("delta", {})
usage = message_delta_chunk.get("usage", {})
assert (
delta.get("stop_reason") == "end_turn"
), f"Expected stop_reason 'end_turn', got {delta.get('stop_reason')}"
assert (
usage.get("input_tokens") == 230
), f"Expected input_tokens 230, got {usage.get('input_tokens')}"
assert (
usage.get("output_tokens") == 65
), f"Expected output_tokens 65, got {usage.get('output_tokens')}"
def test_usage_merging_behavior():
"""Test that usage information is properly merged with stop_reason chunk."""
wrapper = AnthropicStreamWrapper(
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
model="claude-3",
)
# Process chunks and look specifically for the usage merging behavior
chunks = []
for chunk in wrapper:
chunks.append(chunk)
# If this is a message_delta with stop_reason, verify it has usage
if (
chunk.get("type") == "message_delta"
and chunk.get("delta", {}).get("stop_reason") is not None
):
usage = chunk.get("usage", {})
assert (
usage.get("input_tokens") is not None
), "Usage should be merged with stop_reason chunk"
assert (
usage.get("output_tokens") is not None
), "Usage should be merged with stop_reason chunk"
break
def test_sse_wrapper_with_content_after_stop_reason():
"""Test SSE wrapper formatting for the content after stop_reason scenario."""
wrapper = AnthropicStreamWrapper(
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
model="claude-3",
)
# Get SSE formatted chunks
sse_chunks = []
for chunk in wrapper.anthropic_sse_wrapper():
sse_chunks.append(chunk)
if len(sse_chunks) >= 10: # Limit to avoid infinite loops in tests
break
# Verify all chunks are properly formatted as bytes
for chunk in sse_chunks:
assert isinstance(chunk, bytes), "SSE chunks should be bytes"
# Decode and verify SSE format
chunk_str = chunk.decode("utf-8")
lines = chunk_str.split("\n")
# Should have event and data lines
assert any(
line.startswith("event: ") for line in lines
), f"Missing event line in: {chunk_str}"
assert any(
line.startswith("data: ") for line in lines
), f"Missing data line in: {chunk_str}"
@pytest.mark.asyncio
async def test_async_sse_wrapper_with_content_after_stop_reason():
"""Test async SSE wrapper formatting for the content after stop_reason scenario."""
wrapper = AnthropicStreamWrapper(
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
model="claude-3",
)
# Get SSE formatted chunks asynchronously
sse_chunks = []
async for chunk in wrapper.async_anthropic_sse_wrapper():
sse_chunks.append(chunk)
if len(sse_chunks) >= 10: # Limit to avoid infinite loops in tests
break
# Verify all chunks are properly formatted as bytes
for chunk in sse_chunks:
assert isinstance(chunk, bytes), "Async SSE chunks should be bytes"
# Decode and verify SSE format
chunk_str = chunk.decode("utf-8")
lines = chunk_str.split("\n")
# Should have event and data lines
assert any(
line.startswith("event: ") for line in lines
), f"Missing event line in: {chunk_str}"
assert any(
line.startswith("data: ") for line in lines
), f"Missing data line in: {chunk_str}"
if __name__ == "__main__":
# Run a quick test
test_anthropic_stream_wrapper_content_after_stop_reason()
print("✅ Sync test passed")
import asyncio
asyncio.run(test_async_anthropic_stream_wrapper_content_after_stop_reason())
print("✅ Async test passed")
test_usage_merging_behavior()
print("✅ Usage merging test passed")
test_sse_wrapper_with_content_after_stop_reason()
print("✅ SSE wrapper test passed")
asyncio.run(test_async_sse_wrapper_with_content_after_stop_reason())
print("✅ Async SSE wrapper test passed")
print("🎉 All tests passed!")

View file

@ -0,0 +1,2 @@
{"recordId": "request-1", "modelInput": {"messages": [{"role": "user", "content": [{"type": "text", "text": "Hello world!"}]}], "max_tokens": 10, "system": [{"type": "text", "text": "You are a helpful assistant."}], "anthropic_version": "bedrock-2023-05-31"}}
{"recordId": "request-2", "modelInput": {"messages": [{"role": "user", "content": [{"type": "text", "text": "Hello world!"}]}], "max_tokens": 10, "system": [{"type": "text", "text": "You are an unhelpful assistant."}], "anthropic_version": "bedrock-2023-05-31"}}

View file

@ -0,0 +1,2 @@
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}

View file

@ -0,0 +1,90 @@
"""
Test bedrock files transformation functionality
"""
import json
import os
from typing import Any, Dict, List
import pytest
from litellm.llms.bedrock.files.transformation import BedrockJsonlFilesTransformation
class TestBedrockFilesTransformation:
"""Test bedrock files transformation"""
def test_transform_openai_jsonl_content_to_bedrock_jsonl_content(self):
"""
Test transformation of OpenAI JSONL format to Bedrock batch format.
Validates that the transformation correctly converts OpenAI batch completion
format to Bedrock's expected batch format with proper recordId and modelInput structure.
"""
# Initialize the transformation class
transformation = BedrockJsonlFilesTransformation()
# Load input JSONL file
input_file_path = os.path.join(
os.path.dirname(__file__),
"input_batch_completions.jsonl"
)
# Read and parse the JSONL content
openai_jsonl_content = []
with open(input_file_path, 'r') as f:
for line in f:
if line.strip():
openai_jsonl_content.append(json.loads(line))
# Transform the content
bedrock_jsonl_content = transformation._transform_openai_jsonl_content_to_bedrock_jsonl_content(
openai_jsonl_content=openai_jsonl_content
)
# Print the transformation results for validation
print("\n=== INPUT (OpenAI format) ===")
for i, content in enumerate(openai_jsonl_content):
print(f"Record {i+1}:")
print(json.dumps(content, indent=2))
print()
print("\n=== OUTPUT (Bedrock format) ===")
for i, content in enumerate(bedrock_jsonl_content):
print(f"Record {i+1}:")
print(json.dumps(content, indent=2))
print()
# Basic validation
assert len(bedrock_jsonl_content) == len(openai_jsonl_content), "Should have same number of records"
# Check structure of transformed records
for i, record in enumerate(bedrock_jsonl_content):
assert "recordId" in record, f"Record {i+1} should have recordId"
assert "modelInput" in record, f"Record {i+1} should have modelInput"
# Check recordId matches custom_id from input
expected_custom_id = openai_jsonl_content[i].get("custom_id")
assert record["recordId"] == expected_custom_id, f"Record {i+1} recordId should match custom_id"
# Check modelInput has expected structure
model_input = record["modelInput"]
assert isinstance(model_input, dict), f"Record {i+1} modelInput should be a dictionary"
# For Anthropic models, should have anthropic_version and messages
if "anthropic.claude" in openai_jsonl_content[i]["body"]["model"]:
assert "anthropic_version" in model_input, f"Record {i+1} should have anthropic_version"
assert "messages" in model_input, f"Record {i+1} should have messages"
assert "max_tokens" in model_input, f"Record {i+1} should have max_tokens"
# Write expected output to file for reference
expected_output_path = os.path.join(
os.path.dirname(__file__),
"expected_bedrock_batch_completions.jsonl"
)
with open(expected_output_path, 'w') as f:
for record in bedrock_jsonl_content:
f.write(json.dumps(record) + '\n')
print(f"\n=== Expected output written to: {expected_output_path} ===")

View file

@ -0,0 +1,183 @@
"""
Test suite for Dashscope cost calculation functionality.
Tests the cost calculation for Dashscope models including:
- Tiered pricing based on input token ranges
- Caching discounts
- Reasoning tokens
- Standard flat pricing fallback
"""
import json
import math
import os
import sys
import pytest
# Add the project root to Python path
sys.path.insert(0, os.path.abspath("../../../.."))
import litellm
from litellm.llms.dashscope.cost_calculator import (
cost_per_token as dashscope_cost_per_token,
)
from litellm.types.utils import (
CompletionTokensDetailsWrapper,
PromptTokensDetailsWrapper,
Usage,
)
class TestDashscopeCostCalculator:
"""Test suite for Dashscope cost calculation functionality."""
@pytest.fixture(autouse=True)
def setup_model_cost_map(self):
"""Set up the model cost map for testing."""
# Ensure we use local model cost map for consistent testing
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
# Find the project root directory and load model cost map
current_dir = os.path.dirname(os.path.abspath(__file__))
project_root = current_dir
while not os.path.exists(os.path.join(project_root, "model_prices_and_context_window.json")):
parent = os.path.dirname(project_root)
if parent == project_root: # Reached filesystem root
break
project_root = parent
model_cost_path = os.path.join(project_root, "model_prices_and_context_window.json")
with open(model_cost_path, "r") as f:
model_cost_map = json.load(f)
litellm.model_cost = model_cost_map
def test_flat_pricing_basic_cost_calculation(self):
"""Test basic cost calculation for flat pricing models (qwen-max)."""
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500
)
prompt_cost, completion_cost = dashscope_cost_per_token(
model="qwen-max",
usage=usage
)
# Expected costs for qwen-max:
# Input: 1000 tokens * $1.6e-6 = $0.0016
# Output: 500 tokens * $6.4e-6 = $0.0032
expected_prompt_cost = 1000 * 1.6e-6
expected_completion_cost = 500 * 6.4e-6
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_tiered_pricing_single_tier(self):
"""Test tiered pricing when all tokens fall within first tier."""
usage = Usage(
prompt_tokens=20000, # Within first tier (0-32K)
completion_tokens=1000,
total_tokens=21000
)
prompt_cost, completion_cost = dashscope_cost_per_token(
model="qwen3-coder-plus",
usage=usage
)
# Expected costs for qwen3-coder-plus (tier 1):
# Input: 20,000 tokens * $1e-6 = $0.02
# Output: 1,000 tokens * $5e-6 = $0.005
expected_prompt_cost = 20000 * 1e-6
expected_completion_cost = 1000 * 5e-6
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_tiered_pricing_higher_tier(self):
"""Test tiered pricing when tokens fall in higher tier (tier 3)."""
usage = Usage(
prompt_tokens=150000, # Falls in tier 3 (128K-256K)
completion_tokens=2000,
total_tokens=152000
)
prompt_cost, completion_cost = dashscope_cost_per_token(
model="qwen3-coder-plus",
usage=usage
)
# Expected input cost calculation:
# 150,000 tokens falls in tier 3 (128K-256K), so all tokens are charged at tier 3 rate
# Input: 150,000 tokens * $3e-6 = $0.45
# Output: 2,000 tokens falls in tier 1 (0-32K), so charged at tier 1 rate
# Output: 2,000 tokens * $5e-6 = $0.01
expected_prompt_cost = 150000 * 3e-6 # All tokens at tier 3 rate
expected_completion_cost = 2000 * 5e-6 # All tokens at tier 1 rate
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_tiered_pricing_with_caching(self):
"""Test tiered pricing with cached tokens."""
prompt_tokens_details = PromptTokensDetailsWrapper(
cached_tokens=10000 # 10K cached tokens
)
usage = Usage(
prompt_tokens=50000, # 40K regular + 10K cached = 50K total
completion_tokens=1000,
total_tokens=51000,
prompt_tokens_details=prompt_tokens_details
)
prompt_cost, completion_cost = dashscope_cost_per_token(
model="qwen3-coder-plus",
usage=usage
)
# Expected cost calculation:
# Regular tokens: 40,000 falls in tier 2 (32K-128K), so all charged at tier 2 rate
# - Regular: 40,000 * $1.8e-6 = $0.072
# Cached tokens: 10,000 falls in tier 1 (0-32K), so charged at tier 1 cached rate
# - Cached: 10,000 * $1e-7 = $0.001
# Total input cost = $0.072 + $0.001 = $0.073
regular_tokens = 40000
cached_tokens = 10000
expected_regular_cost = regular_tokens * 1.8e-6 # Tier 2 rate
expected_cached_cost = cached_tokens * 1e-7 # Tier 1 cached rate
expected_prompt_cost = expected_regular_cost + expected_cached_cost
expected_completion_cost = 1000 * 5e-6 # Tier 1 rate
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_tiered_pricing_highest_tier(self):
"""Test tiered pricing when tokens exceed highest tier range."""
usage = Usage(
prompt_tokens=2000000, # Exceeds tier 4 max (1M), should use tier 4 rate
completion_tokens=5000,
total_tokens=2005000
)
prompt_cost, completion_cost = dashscope_cost_per_token(
model="qwen3-coder-plus",
usage=usage
)
# Expected cost calculation:
# 2,000,000 tokens exceeds tier 4 (256K-1M), so use tier 4 rate for all tokens
# Input: 2,000,000 tokens * $6e-6 = $12.0
# Output: 5,000 tokens falls in tier 1 (0-32K), so charged at tier 1 rate
# Output: 5,000 tokens * $5e-6 = $0.025
expected_prompt_cost = 2000000 * 6e-6 # Tier 4 rate (highest tier)
expected_completion_cost = 5000 * 5e-6 # Tier 1 rate
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)

View file

@ -158,3 +158,81 @@ class TestGoogleAIStudioTokenCounter:
model=model_to_use,
contents=contents
)
def test_clean_contents_for_gemini_api_removes_id_field(self):
"""Test that _clean_contents_for_gemini_api removes unsupported 'id' field from function responses"""
from litellm.llms.gemini.count_tokens.handler import GoogleAIStudioTokenCounter
token_counter = GoogleAIStudioTokenCounter()
# Test contents with function response containing 'id' field (camelCase)
contents_with_id = [
{
"parts": [
{
"text": "Hello world"
}
],
"role": "user"
},
{
"parts": [
{
"functionResponse": {
"id": "read_many_files-1757526647518-730a691aac11c", # This should be removed
"name": "read_many_files",
"response": {
"output": "No files matching the criteria were found or all were skipped."
}
}
}
],
"role": "user"
}
]
# Clean the contents
cleaned_contents = token_counter._clean_contents_for_gemini_api(contents_with_id)
# Verify the 'id' field was removed
function_response = cleaned_contents[1]["parts"][0]["functionResponse"]
assert "id" not in function_response
assert "name" in function_response
assert "response" in function_response
assert function_response["name"] == "read_many_files"
assert function_response["response"]["output"] == "No files matching the criteria were found or all were skipped."
def test_clean_contents_for_gemini_api_preserves_other_fields(self):
"""Test that _clean_contents_for_gemini_api preserves other fields and structure"""
from litellm.llms.gemini.count_tokens.handler import GoogleAIStudioTokenCounter
token_counter = GoogleAIStudioTokenCounter()
# Test contents without function responses
contents_without_function_response = [
{
"parts": [
{
"text": "This is a regular message"
}
],
"role": "user"
},
{
"parts": [
{
"text": "This is a model response"
}
],
"role": "model"
}
]
# Clean the contents
cleaned_contents = token_counter._clean_contents_for_gemini_api(contents_without_function_response)
# Verify the contents are unchanged
assert cleaned_contents == contents_without_function_response

View file

@ -183,7 +183,9 @@ async def test_budget_reset_and_expires_at_first_of_month(monkeypatch):
assert (
response_date.month == expected_month
), f"Expected month {expected_month}, got {response_date.month} for {key}"
assert response_date.day == 1, f"Expected day 1, got {response_date.day} for {key}"
assert (
response_date.day == 1
), f"Expected day 1, got {response_date.day} for {key}"
@pytest.mark.asyncio
@ -507,7 +509,6 @@ def test_get_new_token_with_invalid_key():
assert "New key must start with 'sk-'" in str(exc_info.value.detail)
@pytest.mark.asyncio
async def test_generate_service_account_requires_team_id():
with pytest.raises(HTTPException):
@ -529,11 +530,12 @@ async def test_generate_service_account_works_with_team_id():
from unittest.mock import patch
# Mock the database and router dependencies from proxy_server
with patch('litellm.proxy.proxy_server.prisma_client') as mock_prisma, \
patch('litellm.proxy.proxy_server.llm_router') as mock_router, \
patch('litellm.proxy.proxy_server.premium_user', False), \
patch('litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn') as mock_generate_key:
with patch("litellm.proxy.proxy_server.prisma_client") as mock_prisma, patch(
"litellm.proxy.proxy_server.llm_router"
) as mock_router, patch("litellm.proxy.proxy_server.premium_user", False), patch(
"litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn"
) as mock_generate_key:
# Configure mocks
mock_prisma.return_value = AsyncMock()
mock_router.return_value = None
@ -542,9 +544,9 @@ async def test_generate_service_account_works_with_team_id():
"key": "sk-test-key",
"expires": None,
"user_id": "test-user",
"team_id": "IJ"
"team_id": "IJ",
}
# This should not raise an exception since team_id is provided
await _common_key_generation_helper(
data=GenerateKeyRequest(
@ -559,7 +561,6 @@ async def test_generate_service_account_works_with_team_id():
)
@pytest.mark.asyncio
async def test_update_service_account_requires_team_id():
data = UpdateKeyRequest(key="sk-1", metadata={"service_account_id": "sa"})
@ -571,7 +572,9 @@ async def test_update_service_account_requires_team_id():
@pytest.mark.asyncio
async def test_update_service_account_works_with_team_id():
data = UpdateKeyRequest(key="sk-1", metadata={"service_account_id": "sa"}, team_id="IJ")
data = UpdateKeyRequest(
key="sk-1", metadata={"service_account_id": "sa"}, team_id="IJ"
)
existing_key = LiteLLM_VerificationToken(token="hashed")
await prepare_key_update_data(data=data, existing_key_row=existing_key)
@ -580,22 +583,22 @@ async def test_update_service_account_works_with_team_id():
@pytest.mark.asyncio
async def test_validate_team_id_used_in_service_account_request_requires_team_id():
"""
Test that validate_team_id_used_in_service_account_request raises HTTPException
Test that validate_team_id_used_in_service_account_request raises HTTPException
when team_id is None for service account key generation.
"""
from litellm.proxy.management_endpoints.key_management_endpoints import (
validate_team_id_used_in_service_account_request,
)
mock_prisma_client = AsyncMock()
# Test that HTTPException is raised when team_id is None
with pytest.raises(HTTPException) as exc_info:
await validate_team_id_used_in_service_account_request(
team_id=None,
prisma_client=mock_prisma_client,
)
assert exc_info.value.status_code == 400
assert "team_id is required for service account keys" in str(exc_info.value.detail)
@ -603,7 +606,7 @@ async def test_validate_team_id_used_in_service_account_request_requires_team_id
@pytest.mark.asyncio
async def test_validate_team_id_used_in_service_account_request_requires_prisma_client():
"""
Test that validate_team_id_used_in_service_account_request raises HTTPException
Test that validate_team_id_used_in_service_account_request raises HTTPException
when prisma_client is None for service account key generation.
"""
from litellm.proxy.management_endpoints.key_management_endpoints import (
@ -616,78 +619,76 @@ async def test_validate_team_id_used_in_service_account_request_requires_prisma_
team_id="test-team-id",
prisma_client=None,
)
assert exc_info.value.status_code == 400
assert "prisma_client is required for service account keys" in str(exc_info.value.detail)
assert "prisma_client is required for service account keys" in str(
exc_info.value.detail
)
@pytest.mark.asyncio
async def test_validate_team_id_used_in_service_account_request_checks_team_exists():
"""
Test that validate_team_id_used_in_service_account_request validates that
Test that validate_team_id_used_in_service_account_request validates that
the team_id exists in the database for service account key generation.
"""
from litellm.proxy.management_endpoints.key_management_endpoints import (
validate_team_id_used_in_service_account_request,
)
mock_prisma_client = AsyncMock()
# Mock the database query to return None (team doesn't exist)
mock_find_unique = AsyncMock(return_value=None)
mock_prisma_client.db.litellm_teamtable.find_unique = mock_find_unique
# Test that HTTPException is raised when team doesn't exist in DB
with pytest.raises(HTTPException) as exc_info:
await validate_team_id_used_in_service_account_request(
team_id="non-existent-team-id",
prisma_client=mock_prisma_client,
)
assert exc_info.value.status_code == 400
assert "team_id does not exist in the database" in str(exc_info.value.detail)
# Verify the database was queried with the correct parameters
mock_find_unique.assert_called_once_with(
where={"team_id": "non-existent-team-id"}
)
mock_find_unique.assert_called_once_with(where={"team_id": "non-existent-team-id"})
@pytest.mark.asyncio
async def test_validate_team_id_used_in_service_account_request_success():
"""
Test that validate_team_id_used_in_service_account_request returns True
Test that validate_team_id_used_in_service_account_request returns True
when team_id exists in the database for service account key generation.
"""
from litellm.proxy.management_endpoints.key_management_endpoints import (
validate_team_id_used_in_service_account_request,
)
mock_prisma_client = AsyncMock()
# Mock the database query to return a team object (team exists)
mock_team = {"team_id": "existing-team-id", "team_name": "Test Team"}
mock_find_unique = AsyncMock(return_value=mock_team)
mock_prisma_client.db.litellm_teamtable.find_unique = mock_find_unique
# Test that function returns True when team exists
result = await validate_team_id_used_in_service_account_request(
team_id="existing-team-id",
prisma_client=mock_prisma_client,
)
assert result is True
# Verify the database was queried with the correct parameters
mock_find_unique.assert_called_once_with(
where={"team_id": "existing-team-id"}
)
mock_find_unique.assert_called_once_with(where={"team_id": "existing-team-id"})
@pytest.mark.asyncio
async def test_generate_service_account_key_endpoint_validation():
"""
Test that the /key/service-account/generate endpoint properly validates
Test that the /key/service-account/generate endpoint properly validates
team_id requirement and team existence in database.
"""
from unittest.mock import patch
@ -705,16 +706,16 @@ async def test_generate_service_account_key_endpoint_validation():
),
litellm_changed_by=None,
)
assert exc_info.value.status_code == 400
assert "team_id is required for service account keys" in str(exc_info.value.detail)
# Test case 2: Team doesn't exist in database
with patch('litellm.proxy.proxy_server.prisma_client') as mock_prisma:
# Test case 2: Team doesn't exist in database
with patch("litellm.proxy.proxy_server.prisma_client") as mock_prisma:
# Mock team not found
mock_find_unique = AsyncMock(return_value=None)
mock_prisma.db.litellm_teamtable.find_unique = mock_find_unique
with pytest.raises(HTTPException) as exc_info:
await generate_service_account_key_fn(
data=GenerateKeyRequest(team_id="non-existent-team"),
@ -723,7 +724,165 @@ async def test_generate_service_account_key_endpoint_validation():
),
litellm_changed_by=None,
)
assert exc_info.value.status_code == 400
assert "team_id does not exist in the database" in str(exc_info.value.detail)
@pytest.mark.asyncio
async def test_unblock_key_supports_both_sk_and_hashed_tokens(monkeypatch):
"""
Test that the unblock_key endpoint correctly handles both sk- prefixed tokens
and hashed tokens by properly converting sk- tokens to hashed format before
database operations.
"""
from unittest.mock import AsyncMock, MagicMock
from litellm.proxy._types import BlockKeyRequest
from litellm.proxy.management_endpoints.key_management_endpoints import unblock_key
# Mock dependencies
mock_prisma_client = AsyncMock()
mock_user_api_key_cache = MagicMock()
mock_proxy_logging_obj = MagicMock()
# Use a proper 64-character hex hash for testing
test_hashed_token = (
"a1b2c3d4e5f6789012345678901234567890123456789012345678901234abcd"
)
# Mock the key record that will be returned from database
mock_key_record = MagicMock()
mock_key_record.token = test_hashed_token
mock_key_record.blocked = False
mock_key_record.model_dump_json.return_value = (
f'{{"token": "{test_hashed_token}", "blocked": false}}'
)
# Mock database operations
mock_prisma_client.db.litellm_verificationtoken.find_unique = AsyncMock(
return_value=mock_key_record
)
mock_prisma_client.db.litellm_verificationtoken.update = AsyncMock(
return_value=mock_key_record
)
# Mock get_key_object and _cache_key_object functions
mock_key_object = MagicMock()
mock_key_object.blocked = True # Initially blocked
# Mock hash_token function
def mock_hash_token(token):
if token == "sk-test123456789":
return test_hashed_token
return token
# Apply monkeypatch
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
monkeypatch.setattr(
"litellm.proxy.proxy_server.user_api_key_cache", mock_user_api_key_cache
)
monkeypatch.setattr(
"litellm.proxy.proxy_server.proxy_logging_obj", mock_proxy_logging_obj
)
monkeypatch.setattr("litellm.proxy.proxy_server.hash_token", mock_hash_token)
monkeypatch.setattr(
"litellm.store_audit_logs", False
) # Disable audit logs for simpler test
# Mock get_key_object and _cache_key_object
async def mock_get_key_object(**kwargs):
return mock_key_object
async def mock_cache_key_object(**kwargs):
pass
monkeypatch.setattr(
"litellm.proxy.management_endpoints.key_management_endpoints.get_key_object",
mock_get_key_object,
)
monkeypatch.setattr(
"litellm.proxy.management_endpoints.key_management_endpoints._cache_key_object",
mock_cache_key_object,
)
# Create mock request and user auth
mock_request = MagicMock()
user_api_key_dict = UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-admin", user_id="admin_user"
)
# Test Case 1: Using sk- prefixed token
sk_token_request = BlockKeyRequest(key="sk-test123456789")
result = await unblock_key(
data=sk_token_request,
http_request=mock_request,
user_api_key_dict=user_api_key_dict,
litellm_changed_by=None,
)
# Verify that the database update was called with hashed token
mock_prisma_client.db.litellm_verificationtoken.update.assert_called_with(
where={"token": test_hashed_token}, data={"blocked": False}
)
assert result == mock_key_record
assert mock_key_object.blocked == False # Should be updated to unblocked
# Reset mocks for second test
mock_prisma_client.db.litellm_verificationtoken.update.reset_mock()
mock_key_object.blocked = True # Reset to blocked state
# Test Case 2: Using already hashed token
hashed_token_request = BlockKeyRequest(key=test_hashed_token)
result = await unblock_key(
data=hashed_token_request,
http_request=mock_request,
user_api_key_dict=user_api_key_dict,
litellm_changed_by=None,
)
# Verify that the database update was called with the same hashed token
mock_prisma_client.db.litellm_verificationtoken.update.assert_called_with(
where={"token": test_hashed_token}, data={"blocked": False}
)
assert result == mock_key_record
assert mock_key_object.blocked == False # Should be updated to unblocked
@pytest.mark.asyncio
async def test_unblock_key_invalid_key_format(monkeypatch):
"""
Test that unblock_key properly validates key format and raises appropriate errors
for invalid keys.
"""
from litellm.proxy._types import BlockKeyRequest
from litellm.proxy.management_endpoints.key_management_endpoints import unblock_key
from litellm.proxy.utils import ProxyException
# Mock prisma_client to avoid DB connection error
mock_prisma_client = AsyncMock()
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
# Mock request and user auth
mock_request = MagicMock()
user_api_key_dict = UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-admin", user_id="admin_user"
)
# Test with invalid key format
invalid_key_request = BlockKeyRequest(key="invalid-key-format")
with pytest.raises(ProxyException) as exc_info:
await unblock_key(
data=invalid_key_request,
http_request=mock_request,
user_api_key_dict=user_api_key_dict,
litellm_changed_by=None,
)
assert exc_info.value.code == "400"
assert "Invalid key format" in str(exc_info.value.message)

View file

@ -18,7 +18,9 @@ import litellm
from litellm.proxy._types import SpendLogsPayload
from litellm.proxy.hooks.proxy_track_cost_callback import _ProxyDBLogger
from litellm.proxy.proxy_server import app, prisma_client
from litellm.proxy.spend_tracking import spend_management_endpoints
from litellm.router import Router
from litellm.types.utils import BudgetConfig
ignored_keys = [
"request_id",
@ -32,6 +34,18 @@ ignored_keys = [
"metadata.cold_storage_object_key",
]
MODEL_LIST = [
{
"model_name": "azure-gpt-4o",
"litellm_params": {
"model": "azure/gpt-4o-mini",
"mock_response": "Hello, world!",
"tags": ["default"],
"base_model": "gpt-4o-mini",
},
},
]
@pytest.fixture
def client():
@ -43,6 +57,19 @@ def add_anthropic_api_key_to_env(monkeypatch):
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-api03-1234567890")
@pytest.fixture
def disable_budget_sync(monkeypatch):
"""Disable periodic sync during tests"""
async def noop(*a, **k):
return None
monkeypatch.setattr(
"litellm.router_strategy.budget_limiter.RouterBudgetLimiting.periodic_sync_in_memory_spend_with_redis",
noop,
)
@pytest.mark.asyncio
async def test_ui_view_spend_logs_with_user_id(client, monkeypatch):
# Mock data for the test
@ -1318,3 +1345,70 @@ async def test_view_spend_tags_no_database(client, monkeypatch):
# Check the actual error message structure
assert "error" in data
assert "Database not connected" in data["error"]["message"]
@pytest.mark.asyncio
async def test_provider_budget_under(disable_budget_sync):
"""Test that router allows completion when under budget"""
provider_budget_config = {
"azure": BudgetConfig(max_budget=0.01, budget_duration="10d")
}
router = Router(
enable_pre_call_checks=True,
provider_budget_config=provider_budget_config,
model_list=MODEL_LIST,
)
response = await router.acompletion(
model="azure-gpt-4o",
messages=[{"role": "user", "content": "Hello, world!"}],
)
assert response is not None
@pytest.mark.asyncio
async def test_provider_budget_over(disable_budget_sync):
"""Test that router allows completion when over budget"""
provider_budget_config = {
"azure": BudgetConfig(max_budget=-0.01, budget_duration="10d")
}
router = Router(
num_retries=0,
enable_pre_call_checks=True,
provider_budget_config=provider_budget_config,
model_list=MODEL_LIST,
)
with pytest.raises(Exception) as e:
response = await router.acompletion(
model="azure-gpt-4o",
messages=[{"role": "user", "content": "Hello, world!"}],
)
assert "Exceeded budget for provider" in str(e.value)
@pytest.mark.asyncio
async def test_provider_budget_provider_budgets(disable_budget_sync):
"""Test that provider_budgets() returns correct values"""
provider = "azure"
max_budget = -0.01
budget_duration = "10d"
provider_budget_config = {
provider: BudgetConfig(max_budget=max_budget, budget_duration=budget_duration)
}
router = Router(
num_retries=0,
enable_pre_call_checks=True,
provider_budget_config=provider_budget_config,
model_list=MODEL_LIST,
)
with patch("litellm.proxy.proxy_server.llm_router", router):
response = await spend_management_endpoints.provider_budgets()
provider_budget_response = response.providers[provider]
assert provider_budget_response.budget_limit == max_budget
assert provider_budget_response.time_period == budget_duration

View file

@ -1059,3 +1059,63 @@ async def test_add_litellm_metadata_from_request_headers():
assert SPEND_LOGS_METADATA == dict(json.loads(headers["x-litellm-spend-logs-metadata"])), "spend_logs_metadata should be the same as the headers"
def test_get_internal_user_header_from_mapping_returns_expected_header():
mappings = [
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
]
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
assert header_name == "X-OpenWebUI-User-Id"
def test_get_internal_user_header_from_mapping_none_when_absent():
mappings = [
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"}
]
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
assert header_name is None
single = {"header_name": "X-Only-Customer", "litellm_user_role": "customer"}
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(single)
assert header_name is None
def test_add_internal_user_from_user_mapping_sets_user_id_when_header_present():
user_api_key_dict = UserAPIKeyAuth(api_key="test-key")
headers = {"X-OpenWebUI-User-Id": "internal-user-123"}
general_settings = {
"user_header_mappings": [
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
]
}
result = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(
general_settings, user_api_key_dict, headers
)
assert result is user_api_key_dict
assert user_api_key_dict.user_id == "internal-user-123"
def test_add_internal_user_from_user_mapping_no_header_or_mapping_returns_unchanged():
user_api_key_dict = UserAPIKeyAuth(api_key="test-key")
result = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(
None, user_api_key_dict, {"X-OpenWebUI-User-Id": "abc"}
)
assert result is user_api_key_dict
assert user_api_key_dict.user_id is None
general_settings = {
"user_header_mappings": [
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"}
]
}
result = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(
general_settings, user_api_key_dict, {"Other": "value"}
)
assert result is user_api_key_dict
assert user_api_key_dict.user_id is None

View file

@ -84,6 +84,15 @@ def test_get_optional_params_image_gen_vertex_ai_size():
assert optional_params["sampleCount"] == 1
def test_get_optional_params_image_gen_filters_empty_values():
optional_params = get_optional_params_image_gen(
model="gpt-image-1",
custom_llm_provider="openai",
extra_body={},
)
assert optional_params == {}
def test_all_model_configs():
from litellm.llms.vertex_ai.vertex_ai_partner_models.ai21.transformation import (
VertexAIAi21Config,
@ -643,6 +652,26 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
},
},
"supports_native_streaming": {"type": "boolean"},
"tiered_pricing": {
"type": "array",
"items": {
"type": "object",
"properties": {
"range": {
"type": "array",
"items": {"type": "number"},
"minItems": 2,
"maxItems": 2
},
"input_cost_per_token": {"type": "number"},
"output_cost_per_token": {"type": "number"},
"cache_read_input_token_cost": {"type": "number"},
"output_cost_per_reasoning_token": {"type": "number"}
},
"required": ["range"],
"additionalProperties": False
}
},
},
"additionalProperties": False,
},

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

Binary file not shown.

After

Width:  |  Height:  |  Size: 48 KiB

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[75832,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","220","static/chunks/220-1c8d82f7ce7658c4.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-8dc8d9524a1f3965.js"],"default",1]
3:I[30628,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","220","static/chunks/220-5061c4cea850d728.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-127adcf8da2b5294.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

Some files were not shown because too many files have changed in this diff Show more