mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge branch 'main' into litellm_dev_09_11_2025_p1
This commit is contained in:
commit
a11f50d8ba
104 changed files with 3960 additions and 453 deletions
|
|
@ -671,6 +671,7 @@ jobs:
|
|||
pip install mypy
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install "google-genai==1.22.0"
|
||||
pip install pyarrow
|
||||
pip install "boto3==1.36.0"
|
||||
pip install "aioboto3==13.4.0"
|
||||
|
|
|
|||
25
cookbook/litellm_proxy_server/batch_api/bedrock/bedrock.py
Normal file
25
cookbook/litellm_proxy_server/batch_api/bedrock/bedrock.py
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
BEDROCK_BATCH_MODEL = "bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0"
|
||||
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("./bedrock_batch_completions.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": BEDROCK_BATCH_MODEL}
|
||||
)
|
||||
print(batch_input_file)
|
||||
|
||||
# Create batch
|
||||
batch = client.batches.create(
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
|
|
@ -0,0 +1,128 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
|
|
@ -7,7 +7,7 @@ Covers Batches, Files
|
|||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Supported Providers | OpenAI, Azure, Vertex | - |
|
||||
| Supported Providers | OpenAI, Azure, Vertex, Bedrock | - |
|
||||
| ✨ Cost Tracking | ✅ | LiteLLM Enterprise only |
|
||||
| Logging | ✅ | Works across all logging integrations |
|
||||
|
||||
|
|
@ -178,6 +178,7 @@ print("list_batches_response=", list_batches_response)
|
|||
### [Azure OpenAI](./providers/azure#azure-batches-api)
|
||||
### [OpenAI](#quick-start)
|
||||
### [Vertex AI](./providers/vertex#batch-apis)
|
||||
### [Bedrock](./providers/bedrock_batches)
|
||||
|
||||
|
||||
## How Cost Tracking for Batches API Works
|
||||
|
|
|
|||
180
docs/my-website/docs/providers/bedrock_batches.md
Normal file
180
docs/my-website/docs/providers/bedrock_batches.md
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Bedrock Batches
|
||||
|
||||
Use Amazon Bedrock Batch Inference API through LiteLLM.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Amazon Bedrock Batch Inference allows you to run inference on large datasets asynchronously |
|
||||
| Provider Doc | [AWS Bedrock Batch Inference ↗](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html) |
|
||||
|
||||
## Overview
|
||||
|
||||
Use this to:
|
||||
|
||||
- Run batch inference on large datasets with Bedrock models
|
||||
- Control batch model access by key/user/team (same as chat completion models)
|
||||
- Manage S3 storage for batch input/output files
|
||||
|
||||
## (Proxy Admin) Usage
|
||||
|
||||
Here's how to give developers access to your Bedrock Batch models.
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
- Specify `mode: batch` for each model: Allows developers to know this is a batch model
|
||||
- Configure S3 bucket and AWS credentials for batch operations
|
||||
|
||||
```yaml showLineNumbers title="litellm_config.yaml"
|
||||
model_list:
|
||||
- model_name: "bedrock-batch-claude"
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
#########################################################
|
||||
########## batch specific params ########################
|
||||
s3_bucket_name: litellm-proxy
|
||||
s3_region_name: us-west-2
|
||||
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
|
||||
model_info:
|
||||
mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model
|
||||
```
|
||||
|
||||
**Required Parameters:**
|
||||
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `s3_bucket_name` | S3 bucket for batch input/output files |
|
||||
| `s3_region_name` | AWS region for S3 bucket |
|
||||
| `s3_access_key_id` | AWS access key for S3 bucket |
|
||||
| `s3_secret_access_key` | AWS secret key for S3 bucket |
|
||||
| `aws_batch_role_arn` | IAM role ARN for Bedrock batch operations. Bedrock Batch APIs require an IAM role ARN to be set. |
|
||||
| `mode: batch` | Indicates to LiteLLM this is a batch model |
|
||||
|
||||
### 2. Create Virtual Key
|
||||
|
||||
```bash showLineNumbers title="create_virtual_key.sh"
|
||||
curl -L -X POST 'https://{PROXY_BASE_URL}/key/generate' \
|
||||
-H 'Authorization: Bearer ${PROXY_API_KEY}' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"models": ["bedrock-batch-claude"]}'
|
||||
```
|
||||
|
||||
You can now use the virtual key to access the batch models (See Developer flow).
|
||||
|
||||
## (Developer) Usage
|
||||
|
||||
Here's how to create a LiteLLM managed file and execute Bedrock Batch CRUD operations with the file.
|
||||
|
||||
### 1. Create request.jsonl
|
||||
|
||||
- Check models available via `/model_group/info`
|
||||
- See all models with `mode: batch`
|
||||
- Set `model` in .jsonl to the model from `/model_group/info`
|
||||
|
||||
```json showLineNumbers title="bedrock_batch_completions.jsonl"
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock-batch-claude", "messages": [{"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello world!"}], "max_tokens": 1000}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock-batch-claude", "messages": [{"role": "system", "content": "You are an unhelpful assistant."}, {"role": "user", "content": "Hello world!"}], "max_tokens": 1000}}
|
||||
```
|
||||
|
||||
Expectation:
|
||||
|
||||
- LiteLLM translates this to the bedrock deployment specific value (e.g. `bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0`)
|
||||
|
||||
### 2. Upload File
|
||||
|
||||
Specify `target_model_names: "<model-name>"` to enable LiteLLM managed files and request validation.
|
||||
|
||||
model-name should be the same as the model-name in the request.jsonl
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="bedrock_batch.py"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("./bedrock_batch_completions.jsonl", "rb"), # {"model": "bedrock-batch-claude"} <-> {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0"}
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": "bedrock-batch-claude"}
|
||||
)
|
||||
print(batch_input_file)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Upload File"
|
||||
curl http://localhost:4000/v1/files \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-F purpose="batch" \
|
||||
-F file="@bedrock_batch_completions.jsonl" \
|
||||
-F extra_body='{"target_model_names": "bedrock-batch-claude"}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Where is the file written?**:
|
||||
|
||||
The file is written to S3 bucket specified in your config and prepared for Bedrock batch inference.
|
||||
|
||||
### 3. Create the batch
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="bedrock_batch.py"
|
||||
...
|
||||
# Create batch
|
||||
batch = client.batches.create(
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Create Batch Request"
|
||||
curl http://localhost:4000/v1/batches \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"input_file_id": "file-abc123",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"completion_window": "24h",
|
||||
"metadata": {"description": "Test batch job"}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## FAQ
|
||||
|
||||
### Where are my files written?
|
||||
|
||||
When a `target_model_names` is specified, the file is written to the S3 bucket configured in your Bedrock batch model configuration.
|
||||
|
||||
### What models are supported?
|
||||
|
||||
LiteLLM only supports Bedrock Anthropic Models for Batch API. If you want other bedrock models file an issue [here](https://github.com/BerriAI/litellm/issues/new/choose).
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [AWS Bedrock Batch Inference Documentation](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html)
|
||||
- [LiteLLM Managed Batches](../proxy/managed_batches)
|
||||
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
# Dashscope
|
||||
# Dashscope (Qwen API)
|
||||
https://dashscope.console.aliyun.com/
|
||||
|
||||
**We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests**
|
||||
|
|
|
|||
|
|
@ -11,13 +11,13 @@ The proxy also supports json logs. [See here](#json-logs)
|
|||
|
||||
**via cli**
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
$ litellm --debug
|
||||
```
|
||||
|
||||
**via env**
|
||||
|
||||
```python
|
||||
```python showLineNumbers
|
||||
os.environ["LITELLM_LOG"] = "INFO"
|
||||
```
|
||||
|
||||
|
|
@ -25,25 +25,25 @@ os.environ["LITELLM_LOG"] = "INFO"
|
|||
|
||||
**via cli**
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
$ litellm --detailed_debug
|
||||
```
|
||||
|
||||
**via env**
|
||||
|
||||
```python
|
||||
```python showLineNumbers
|
||||
os.environ["LITELLM_LOG"] = "DEBUG"
|
||||
```
|
||||
|
||||
### Debug Logs
|
||||
|
||||
Run the proxy with `--detailed_debug` to view detailed debug logs
|
||||
```shell
|
||||
```shell showLineNumbers
|
||||
litellm --config /path/to/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
When making requests you should see the POST request sent by LiteLLM to the LLM on the Terminal output
|
||||
```shell
|
||||
```shell showLineNumbers
|
||||
POST Request Sent from LiteLLM:
|
||||
curl -X POST \
|
||||
https://api.openai.com/v1/chat/completions \
|
||||
|
|
@ -51,25 +51,63 @@ https://api.openai.com/v1/chat/completions \
|
|||
-d '{"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}'
|
||||
```
|
||||
|
||||
## Debug single request
|
||||
|
||||
Pass in `litellm_request_debug=True` in the request body
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model":"fake-openai-endpoint",
|
||||
"messages": [{"role": "user","content": "How many r in the word strawberry?"}],
|
||||
"litellm_request_debug": true
|
||||
}'
|
||||
```
|
||||
|
||||
This will emit the raw request sent by LiteLLM to the API Provider and raw response received from the API Provider for **just** this request in the logs.
|
||||
|
||||
|
||||
```bash showLineNumbers
|
||||
INFO: Uvicorn running on http://0.0.0.0:4000 (Press CTRL+C to quit)
|
||||
20:14:06 - LiteLLM:WARNING: litellm_logging.py:938 -
|
||||
|
||||
POST Request Sent from LiteLLM:
|
||||
curl -X POST \
|
||||
https://exampleopenaiendpoint-production.up.railway.app/chat/completions \
|
||||
-H 'Authorization: Be****ey' -H 'Content-Type: application/json' \
|
||||
-d '{'model': 'fake', 'messages': [{'role': 'user', 'content': 'How many r in the word strawberry?'}], 'stream': False}'
|
||||
|
||||
|
||||
20:14:06 - LiteLLM:WARNING: litellm_logging.py:1015 - RAW RESPONSE:
|
||||
{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-3.5-turbo-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}}
|
||||
|
||||
|
||||
INFO: 127.0.0.1:56155 - "POST /chat/completions HTTP/1.1" 200 OK
|
||||
|
||||
```
|
||||
|
||||
|
||||
## JSON LOGS
|
||||
|
||||
Set `JSON_LOGS="True"` in your env:
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
export JSON_LOGS="True"
|
||||
```
|
||||
**OR**
|
||||
|
||||
Set `json_logs: true` in your yaml:
|
||||
|
||||
```yaml
|
||||
```yaml showLineNumbers
|
||||
litellm_settings:
|
||||
json_logs: true
|
||||
```
|
||||
|
||||
Start proxy
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
$ litellm
|
||||
```
|
||||
|
||||
|
|
@ -80,7 +118,7 @@ The proxy will now all logs in json format.
|
|||
Turn off fastapi's default 'INFO' logs
|
||||
|
||||
1. Turn on 'json logs'
|
||||
```yaml
|
||||
```yaml showLineNumbers
|
||||
litellm_settings:
|
||||
json_logs: true
|
||||
```
|
||||
|
|
@ -89,20 +127,20 @@ litellm_settings:
|
|||
|
||||
Only get logs if an error occurs.
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
LITELLM_LOG="ERROR"
|
||||
```
|
||||
|
||||
3. Start proxy
|
||||
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
$ litellm
|
||||
```
|
||||
|
||||
Expected Output:
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
# no info statements
|
||||
```
|
||||
|
||||
|
|
@ -119,14 +157,14 @@ This can be caused due to all your models hitting rate limit errors, causing the
|
|||
How to control this?
|
||||
- Adjust the cooldown time
|
||||
|
||||
```yaml
|
||||
```yaml showLineNumbers
|
||||
router_settings:
|
||||
cooldown_time: 0 # 👈 KEY CHANGE
|
||||
```
|
||||
|
||||
- Disable Cooldowns [NOT RECOMMENDED]
|
||||
|
||||
```yaml
|
||||
```yaml showLineNumbers
|
||||
router_settings:
|
||||
disable_cooldowns: True
|
||||
```
|
||||
|
|
|
|||
|
|
@ -410,6 +410,7 @@ const sidebars = {
|
|||
items: [
|
||||
"providers/bedrock",
|
||||
"providers/bedrock_agents",
|
||||
"providers/bedrock_batches",
|
||||
"providers/bedrock_vector_store",
|
||||
]
|
||||
},
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast
|
|||
import httpx
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.azure.batches.handler import AzureBatchesAPI
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
|
|
@ -38,6 +39,7 @@ from litellm.utils import (
|
|||
ProviderConfigManager,
|
||||
client,
|
||||
get_litellm_params,
|
||||
get_llm_provider,
|
||||
supports_httpx_timeout,
|
||||
)
|
||||
|
||||
|
|
@ -49,6 +51,45 @@ base_llm_http_handler = BaseLLMHTTPHandler()
|
|||
#################################################
|
||||
|
||||
|
||||
def _resolve_timeout(
|
||||
optional_params: GenericLiteLLMParams,
|
||||
kwargs: Dict[str, Any],
|
||||
custom_llm_provider: str,
|
||||
default_timeout: float = 600.0,
|
||||
) -> float:
|
||||
"""
|
||||
Resolve timeout value from various sources and handle httpx.Timeout objects.
|
||||
|
||||
Args:
|
||||
optional_params: GenericLiteLLMParams object containing timeout
|
||||
kwargs: Additional kwargs that may contain request_timeout
|
||||
custom_llm_provider: Provider name for httpx timeout support check
|
||||
default_timeout: Default timeout value to use
|
||||
|
||||
Returns:
|
||||
Resolved timeout as float
|
||||
"""
|
||||
timeout = optional_params.timeout or kwargs.get("request_timeout", default_timeout) or default_timeout
|
||||
|
||||
# Handle httpx.Timeout objects
|
||||
if isinstance(timeout, httpx.Timeout):
|
||||
if supports_httpx_timeout(custom_llm_provider) is False:
|
||||
# Extract read timeout for providers that don't support httpx.Timeout
|
||||
read_timeout = timeout.read or default_timeout
|
||||
return float(read_timeout)
|
||||
else:
|
||||
# For providers that support httpx.Timeout, we still need to return a float
|
||||
# This case might need to be handled differently based on the actual use case
|
||||
return float(timeout.read or default_timeout)
|
||||
|
||||
# Handle None case
|
||||
if timeout is None:
|
||||
return float(default_timeout)
|
||||
|
||||
# Handle numeric values (int, float, string representations)
|
||||
return float(timeout)
|
||||
|
||||
|
||||
@client
|
||||
async def acreate_batch(
|
||||
completion_window: Literal["24h"],
|
||||
|
|
@ -118,13 +159,23 @@ def create_batch(
|
|||
litellm_call_id = kwargs.get("litellm_call_id", None)
|
||||
proxy_server_request = kwargs.get("proxy_server_request", None)
|
||||
model_info = kwargs.get("model_info", None)
|
||||
model: Optional[str] = kwargs.get("model", None)
|
||||
try:
|
||||
if model is not None:
|
||||
model, _, _, _ = get_llm_provider(
|
||||
model=model,
|
||||
custom_llm_provider=None,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"litellm.batches.main.py::create_batch() - Error inferring custom_llm_provider - {str(e)}")
|
||||
|
||||
_is_async = kwargs.pop("acreate_batch", False) is True
|
||||
litellm_params = dict(GenericLiteLLMParams(**kwargs))
|
||||
litellm_logging_obj: LiteLLMLoggingObj = cast(LiteLLMLoggingObj, kwargs.get("litellm_logging_obj", None))
|
||||
### TIMEOUT LOGIC ###
|
||||
timeout = optional_params.timeout or kwargs.get("request_timeout", 600) or 600
|
||||
timeout = _resolve_timeout(optional_params, kwargs, custom_llm_provider)
|
||||
litellm_logging_obj.update_environment_variables(
|
||||
model=None,
|
||||
model=model,
|
||||
user=None,
|
||||
optional_params=optional_params.model_dump(),
|
||||
litellm_params={
|
||||
|
|
@ -138,18 +189,6 @@ def create_batch(
|
|||
},
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
if (
|
||||
timeout is not None
|
||||
and isinstance(timeout, httpx.Timeout)
|
||||
and supports_httpx_timeout(custom_llm_provider) is False
|
||||
):
|
||||
read_timeout = timeout.read or 600
|
||||
timeout = read_timeout # default 10 min timeout
|
||||
elif timeout is not None and not isinstance(timeout, httpx.Timeout):
|
||||
timeout = float(timeout) # type: ignore
|
||||
elif timeout is None:
|
||||
timeout = 600.0
|
||||
|
||||
|
||||
_create_batch_request = CreateBatchRequest(
|
||||
|
|
@ -160,10 +199,13 @@ def create_batch(
|
|||
extra_headers=extra_headers,
|
||||
extra_body=extra_body,
|
||||
)
|
||||
provider_config = ProviderConfigManager.get_provider_batches_config(
|
||||
model="",
|
||||
provider=LlmProviders(custom_llm_provider),
|
||||
)
|
||||
if model is not None:
|
||||
provider_config = ProviderConfigManager.get_provider_batches_config(
|
||||
model=model,
|
||||
provider=LlmProviders(custom_llm_provider),
|
||||
)
|
||||
else:
|
||||
provider_config = None
|
||||
if provider_config is not None:
|
||||
response = base_llm_http_handler.create_batch(
|
||||
provider_config=provider_config,
|
||||
|
|
@ -179,6 +221,7 @@ def create_batch(
|
|||
and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
|
||||
else None,
|
||||
timeout=timeout,
|
||||
model=model,
|
||||
)
|
||||
return response
|
||||
api_base: Optional[str] = None
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ DEFAULT_SQS_FLUSH_INTERVAL_SECONDS = int(
|
|||
os.getenv("DEFAULT_SQS_FLUSH_INTERVAL_SECONDS", 10)
|
||||
)
|
||||
DEFAULT_NUM_WORKERS_LITELLM_PROXY = int(
|
||||
os.getenv("DEFAULT_NUM_WORKERS_LITELLM_PROXY", os.cpu_count() or 4)
|
||||
os.getenv("DEFAULT_NUM_WORKERS_LITELLM_PROXY", 1)
|
||||
)
|
||||
DEFAULT_SQS_BATCH_SIZE = int(os.getenv("DEFAULT_SQS_BATCH_SIZE", 512))
|
||||
SQS_SEND_MESSAGE_ACTION = "SendMessage"
|
||||
|
|
@ -60,7 +60,9 @@ DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO = int(
|
|||
os.getenv("DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO", 128)
|
||||
)
|
||||
DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE = int(
|
||||
os.getenv("DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE", 512)
|
||||
os.getenv(
|
||||
"DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE", 512
|
||||
)
|
||||
)
|
||||
|
||||
# Generic fallback for unknown models
|
||||
|
|
@ -949,7 +951,9 @@ LITELLM_CLI_SESSION_TOKEN_PREFIX = "litellm-session-token"
|
|||
DB_SPEND_UPDATE_JOB_NAME = "db_spend_update_job"
|
||||
PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME = "prometheus_emit_budget_metrics"
|
||||
CLOUDZERO_EXPORT_USAGE_DATA_JOB_NAME = "cloudzero_export_usage_data"
|
||||
CLOUDZERO_MAX_FETCHED_DATA_RECORDS = int(os.getenv("CLOUDZERO_MAX_FETCHED_DATA_RECORDS", 50000))
|
||||
CLOUDZERO_MAX_FETCHED_DATA_RECORDS = int(
|
||||
os.getenv("CLOUDZERO_MAX_FETCHED_DATA_RECORDS", 50000)
|
||||
)
|
||||
SPEND_LOG_CLEANUP_JOB_NAME = "spend_log_cleanup"
|
||||
SPEND_LOG_RUN_LOOPS = int(os.getenv("SPEND_LOG_RUN_LOOPS", 500))
|
||||
SPEND_LOG_CLEANUP_BATCH_SIZE = int(os.getenv("SPEND_LOG_CLEANUP_BATCH_SIZE", 1000))
|
||||
|
|
|
|||
|
|
@ -344,6 +344,11 @@ def cost_per_token( # noqa: PLR0915
|
|||
return perplexity_cost_per_token(model=model, usage=usage_block)
|
||||
elif custom_llm_provider == "xai":
|
||||
return xai_cost_per_token(model=model, usage=usage_block)
|
||||
elif custom_llm_provider == "dashscope":
|
||||
from litellm.llms.dashscope.cost_calculator import (
|
||||
cost_per_token as dashscope_cost_per_token,
|
||||
)
|
||||
return dashscope_cost_per_token(model=model, usage=usage_block)
|
||||
else:
|
||||
model_info = _cached_get_model_info_helper(
|
||||
model=model, custom_llm_provider=custom_llm_provider
|
||||
|
|
|
|||
27
litellm/files/utils.py
Normal file
27
litellm/files/utils.py
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
from typing import Optional
|
||||
|
||||
from litellm.types.llms.openai import CreateFileRequest
|
||||
from litellm.types.utils import ExtractedFileData
|
||||
|
||||
|
||||
class FilesAPIUtils:
|
||||
"""
|
||||
Utils for files API interface on litellm
|
||||
"""
|
||||
@staticmethod
|
||||
def is_batch_jsonl_file(create_file_data: CreateFileRequest, extracted_file_data: ExtractedFileData) -> bool:
|
||||
"""
|
||||
Check if the file is a batch jsonl file
|
||||
"""
|
||||
return (
|
||||
create_file_data.get("purpose") == "batch"
|
||||
and FilesAPIUtils.valid_content_type(extracted_file_data.get("content_type"))
|
||||
and extracted_file_data.get("content") is not None
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def valid_content_type(content_type: Optional[str]) -> bool:
|
||||
"""
|
||||
Check if the content type is valid
|
||||
"""
|
||||
return content_type in set(["application/jsonl", "application/octet-stream"])
|
||||
|
|
@ -224,6 +224,9 @@ async def agenerate_content(
|
|||
loop = asyncio.get_event_loop()
|
||||
kwargs["agenerate_content"] = True
|
||||
|
||||
# Handle generationConfig parameter from kwargs for backward compatibility
|
||||
if "generationConfig" in kwargs and config is None:
|
||||
config = kwargs.pop("generationConfig")
|
||||
# get custom llm provider so we can use this for mapping exceptions
|
||||
if custom_llm_provider is None:
|
||||
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
|
|
@ -288,6 +291,9 @@ def generate_content(
|
|||
try:
|
||||
_is_async = kwargs.pop("agenerate_content", False) is True
|
||||
|
||||
# Handle generationConfig parameter from kwargs for backward compatibility
|
||||
if "generationConfig" in kwargs and config is None:
|
||||
config = kwargs.pop("generationConfig")
|
||||
# Check for mock response first
|
||||
litellm_params = GenericLiteLLMParams(**kwargs)
|
||||
if litellm_params.mock_response and isinstance(
|
||||
|
|
@ -374,6 +380,9 @@ async def agenerate_content_stream(
|
|||
try:
|
||||
kwargs["agenerate_content_stream"] = True
|
||||
|
||||
# Handle generationConfig parameter from kwargs for backward compatibility
|
||||
if "generationConfig" in kwargs and config is None:
|
||||
config = kwargs.pop("generationConfig")
|
||||
# get custom llm provider so we can use this for mapping exceptions
|
||||
if custom_llm_provider is None:
|
||||
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
|
|
@ -461,6 +470,9 @@ def generate_content_stream(
|
|||
# Remove any async-related flags since this is the sync function
|
||||
_is_async = kwargs.pop("agenerate_content_stream", False)
|
||||
|
||||
# Handle generationConfig parameter from kwargs for backward compatibility
|
||||
if "generationConfig" in kwargs and config is None:
|
||||
config = kwargs.pop("generationConfig")
|
||||
# Setup the call
|
||||
setup_result = GenerateContentHelper.setup_generate_content_call(
|
||||
model=model,
|
||||
|
|
|
|||
|
|
@ -62,6 +62,7 @@ def get_litellm_params(
|
|||
use_litellm_proxy: Optional[bool] = None,
|
||||
api_version: Optional[str] = None,
|
||||
max_retries: Optional[int] = None,
|
||||
litellm_request_debug: Optional[bool] = None,
|
||||
**kwargs,
|
||||
) -> dict:
|
||||
litellm_params = {
|
||||
|
|
@ -118,5 +119,6 @@ def get_litellm_params(
|
|||
"vertex_credentials": kwargs.get("vertex_credentials"),
|
||||
"vertex_project": kwargs.get("vertex_project"),
|
||||
"use_litellm_proxy": use_litellm_proxy,
|
||||
"litellm_request_debug": litellm_request_debug,
|
||||
}
|
||||
return litellm_params
|
||||
|
|
|
|||
|
|
@ -245,6 +245,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
global supabaseClient, promptLayerLogger, weightsBiasesLogger, logfireLogger, capture_exception, add_breadcrumb, lunaryLogger, logfireLogger, prometheusLogger, slack_app
|
||||
custom_pricing: bool = False
|
||||
stream_options = None
|
||||
litellm_request_debug: bool = False
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
|
|
@ -470,6 +471,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
**self.litellm_params,
|
||||
**scrub_sensitive_keys_in_metadata(litellm_params),
|
||||
}
|
||||
self.litellm_request_debug = litellm_params.get("litellm_request_debug", False)
|
||||
self.logger_fn = litellm_params.get("logger_fn", None)
|
||||
verbose_logger.debug(f"self.optional_params: {self.optional_params}")
|
||||
|
||||
|
|
@ -907,13 +909,19 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
|
||||
Prints the RAW curl command sent from LiteLLM
|
||||
"""
|
||||
if _is_debugging_on():
|
||||
if _is_debugging_on() or self.litellm_request_debug:
|
||||
if json_logs:
|
||||
masked_headers = self._get_masked_headers(headers)
|
||||
verbose_logger.debug(
|
||||
"POST Request Sent from LiteLLM",
|
||||
extra={"api_base": {api_base}, **masked_headers},
|
||||
)
|
||||
if self.litellm_request_debug:
|
||||
verbose_logger.warning( # .warning ensures this shows up in all environments
|
||||
"POST Request Sent from LiteLLM",
|
||||
extra={"api_base": {api_base}, **masked_headers},
|
||||
)
|
||||
else:
|
||||
verbose_logger.debug(
|
||||
"POST Request Sent from LiteLLM",
|
||||
extra={"api_base": {api_base}, **masked_headers},
|
||||
)
|
||||
else:
|
||||
headers = additional_args.get("headers", {})
|
||||
if headers is None:
|
||||
|
|
@ -926,7 +934,12 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
additional_args=additional_args,
|
||||
data=data,
|
||||
)
|
||||
verbose_logger.debug(f"\033[92m{curl_command}\033[0m\n")
|
||||
if self.litellm_request_debug:
|
||||
verbose_logger.warning(
|
||||
f"\033[92m{curl_command}\033[0m\n"
|
||||
) # .warning ensures this shows up in all environments
|
||||
else:
|
||||
verbose_logger.debug(f"\033[92m{curl_command}\033[0m\n")
|
||||
|
||||
def _get_request_body(self, data: dict) -> str:
|
||||
return str(data)
|
||||
|
|
@ -983,8 +996,14 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
self.model_call_details["additional_args"] = additional_args
|
||||
self.model_call_details["log_event_type"] = "post_api_call"
|
||||
|
||||
if self.litellm_request_debug:
|
||||
attr = "warning"
|
||||
else:
|
||||
attr = "debug"
|
||||
|
||||
if json_logs:
|
||||
verbose_logger.debug(
|
||||
callattr = getattr(verbose_logger, attr)
|
||||
callattr(
|
||||
"RAW RESPONSE:\n{}\n\n".format(
|
||||
self.model_call_details.get(
|
||||
"original_response", self.model_call_details
|
||||
|
|
@ -992,7 +1011,8 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
),
|
||||
)
|
||||
else:
|
||||
print_verbose(
|
||||
callattr = getattr(verbose_logger, attr)
|
||||
callattr(
|
||||
"RAW RESPONSE:\n{}\n\n".format(
|
||||
self.model_call_details.get(
|
||||
"original_response", self.model_call_details
|
||||
|
|
@ -1714,12 +1734,16 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
response_obj=result,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
litellm_call_id=current_call_id
|
||||
if (
|
||||
current_call_id := litellm_params.get("litellm_call_id")
|
||||
)
|
||||
is not None
|
||||
else str(uuid.uuid4()),
|
||||
litellm_call_id=(
|
||||
current_call_id
|
||||
if (
|
||||
current_call_id := litellm_params.get(
|
||||
"litellm_call_id"
|
||||
)
|
||||
)
|
||||
is not None
|
||||
else str(uuid.uuid4())
|
||||
),
|
||||
print_verbose=print_verbose,
|
||||
)
|
||||
if callback == "wandb" and weightsBiasesLogger is not None:
|
||||
|
|
@ -3367,6 +3391,7 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915
|
|||
return galileo_logger # type: ignore
|
||||
elif logging_integration == "cloudzero":
|
||||
from litellm.integrations.cloudzero.cloudzero import CloudZeroLogger
|
||||
|
||||
for callback in _in_memory_loggers:
|
||||
if isinstance(callback, CloudZeroLogger):
|
||||
return callback # type: ignore
|
||||
|
|
@ -3594,6 +3619,7 @@ def get_custom_logger_compatible_class( # noqa: PLR0915
|
|||
return callback
|
||||
elif logging_integration == "cloudzero":
|
||||
from litellm.integrations.cloudzero.cloudzero import CloudZeroLogger
|
||||
|
||||
for callback in _in_memory_loggers:
|
||||
if isinstance(callback, CloudZeroLogger):
|
||||
return callback
|
||||
|
|
@ -4504,7 +4530,7 @@ def get_standard_logging_object_payload(
|
|||
|
||||
def emit_standard_logging_payload(payload: StandardLoggingPayload):
|
||||
if os.getenv("LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD"):
|
||||
print(json.dumps(payload, indent=4)) # noqa
|
||||
print(json.dumps(payload, indent=4)) # noqa
|
||||
|
||||
|
||||
def get_standard_logging_metadata(
|
||||
|
|
|
|||
|
|
@ -1,10 +1,21 @@
|
|||
import asyncio
|
||||
import contextlib
|
||||
from typing import Coroutine, Optional
|
||||
import contextvars
|
||||
from typing import Coroutine, Optional, TypedDict
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
|
||||
|
||||
class LoggingTask(TypedDict):
|
||||
"""
|
||||
A logging task with its associated context to ensure logging is executed in
|
||||
the original task's context.
|
||||
"""
|
||||
|
||||
coroutine: Coroutine
|
||||
context: contextvars.Context
|
||||
|
||||
|
||||
class LoggingWorker:
|
||||
"""
|
||||
A simple, async logging worker that processes log coroutines in the background.
|
||||
|
|
@ -13,77 +24,84 @@ class LoggingWorker:
|
|||
This leads to a +200 RPS performance improvement when using LiteLLM Python SDK or Proxy Server.
|
||||
- Use this to queue coroutine tasks that are not critical to the main flow of the application. e.g Success/Error callbacks, logging, etc.
|
||||
"""
|
||||
|
||||
LOGGING_WORKER_MAX_QUEUE_SIZE = 50_000
|
||||
LOGGING_WORKER_MAX_TIME_PER_COROUTINE = 20.0
|
||||
|
||||
MAX_ITERATIONS_TO_CLEAR_QUEUE = 200
|
||||
MAX_TIME_TO_CLEAR_QUEUE = 5.0
|
||||
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
timeout: float = LOGGING_WORKER_MAX_TIME_PER_COROUTINE,
|
||||
self,
|
||||
timeout: float = LOGGING_WORKER_MAX_TIME_PER_COROUTINE,
|
||||
max_queue_size: int = LOGGING_WORKER_MAX_QUEUE_SIZE,
|
||||
):
|
||||
self.timeout = timeout
|
||||
self.max_queue_size = max_queue_size
|
||||
self._queue: Optional[asyncio.Queue] = None
|
||||
self._queue: Optional[asyncio.Queue[LoggingTask]] = None
|
||||
self._worker_task: Optional[asyncio.Task] = None
|
||||
|
||||
|
||||
def _ensure_queue(self) -> None:
|
||||
"""Initialize the queue if it doesn't exist."""
|
||||
if self._queue is None:
|
||||
self._queue = asyncio.Queue(maxsize=self.max_queue_size)
|
||||
|
||||
|
||||
def start(self) -> None:
|
||||
"""Start the logging worker. Idempotent - safe to call multiple times."""
|
||||
self._ensure_queue()
|
||||
if self._worker_task is None or self._worker_task.done():
|
||||
self._worker_task = asyncio.create_task(self._worker_loop())
|
||||
|
||||
|
||||
async def _worker_loop(self) -> None:
|
||||
"""Main worker loop that processes log coroutines sequentially."""
|
||||
try:
|
||||
if self._queue is None:
|
||||
return
|
||||
|
||||
|
||||
while True:
|
||||
# Process one coroutine at a time to keep event loop load predictable
|
||||
coroutine = await self._queue.get()
|
||||
task = await self._queue.get()
|
||||
try:
|
||||
await asyncio.wait_for(coroutine, timeout=self.timeout)
|
||||
# Run the coroutine in its original context
|
||||
await asyncio.wait_for(
|
||||
task["context"].run(asyncio.create_task, task["coroutine"]),
|
||||
timeout=self.timeout,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"LoggingWorker error: {e}")
|
||||
pass
|
||||
finally:
|
||||
self._queue.task_done()
|
||||
|
||||
|
||||
except asyncio.CancelledError:
|
||||
verbose_logger.debug("LoggingWorker cancelled during shutdown")
|
||||
# Attempt to clear remaining items to prevent "never awaited" warnings
|
||||
await self.clear_queue()
|
||||
|
||||
|
||||
def enqueue(self, coroutine: Coroutine) -> None:
|
||||
"""
|
||||
Add a coroutine to the logging queue.
|
||||
Add a coroutine to the logging queue.
|
||||
Hot path: never blocks, drops logs if queue is full.
|
||||
"""
|
||||
if self._queue is None:
|
||||
return
|
||||
|
||||
|
||||
try:
|
||||
self._queue.put_nowait(coroutine)
|
||||
# Capture the current context when enqueueing
|
||||
task = LoggingTask(coroutine=coroutine, context=contextvars.copy_context())
|
||||
self._queue.put_nowait(task)
|
||||
except asyncio.QueueFull as e:
|
||||
verbose_logger.exception(f"LoggingWorker queue is full: {e}")
|
||||
# Drop logs on overload to protect request throughput
|
||||
pass
|
||||
|
||||
|
||||
def ensure_initialized_and_enqueue(self, async_coroutine: Coroutine):
|
||||
"""
|
||||
Ensure the logging worker is initialized and enqueue the coroutine.
|
||||
"""
|
||||
self.start()
|
||||
self.enqueue(async_coroutine)
|
||||
|
||||
|
||||
async def stop(self) -> None:
|
||||
"""Stop the logging worker and clean up resources."""
|
||||
if self._worker_task:
|
||||
|
|
@ -91,34 +109,42 @@ class LoggingWorker:
|
|||
with contextlib.suppress(Exception):
|
||||
await self._worker_task
|
||||
self._worker_task = None
|
||||
|
||||
|
||||
async def flush(self) -> None:
|
||||
"""Flush the logging queue."""
|
||||
if self._queue is None:
|
||||
return
|
||||
while not self._queue.empty():
|
||||
await self._queue.join()
|
||||
|
||||
|
||||
async def clear_queue(self):
|
||||
"""
|
||||
Clear the queue with a maximum time limit.
|
||||
"""
|
||||
if self._queue is None:
|
||||
return
|
||||
|
||||
|
||||
start_time = asyncio.get_event_loop().time()
|
||||
|
||||
|
||||
for _ in range(self.MAX_ITERATIONS_TO_CLEAR_QUEUE):
|
||||
# Check if we've exceeded the maximum time
|
||||
if asyncio.get_event_loop().time() - start_time >= self.MAX_TIME_TO_CLEAR_QUEUE:
|
||||
verbose_logger.warning(f"clear_queue exceeded max_time of {self.MAX_TIME_TO_CLEAR_QUEUE}s, stopping early")
|
||||
if (
|
||||
asyncio.get_event_loop().time() - start_time
|
||||
>= self.MAX_TIME_TO_CLEAR_QUEUE
|
||||
):
|
||||
verbose_logger.warning(
|
||||
f"clear_queue exceeded max_time of {self.MAX_TIME_TO_CLEAR_QUEUE}s, stopping early"
|
||||
)
|
||||
break
|
||||
|
||||
|
||||
try:
|
||||
coroutine = self._queue.get_nowait()
|
||||
task = self._queue.get_nowait()
|
||||
# Await the coroutine to properly execute and avoid "never awaited" warnings
|
||||
try:
|
||||
await asyncio.wait_for(coroutine, timeout=self.timeout)
|
||||
await asyncio.wait_for(
|
||||
task["context"].run(asyncio.create_task, task["coroutine"]),
|
||||
timeout=self.timeout,
|
||||
)
|
||||
except Exception:
|
||||
# Suppress errors during cleanup
|
||||
pass
|
||||
|
|
@ -129,4 +155,3 @@ class LoggingWorker:
|
|||
|
||||
# Global instance for backward compatibility
|
||||
GLOBAL_LOGGING_WORKER = LoggingWorker()
|
||||
|
||||
|
|
|
|||
|
|
@ -28,10 +28,6 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
TextBlock,
|
||||
)
|
||||
|
||||
def __init__(self, completion_stream: Any, model: str):
|
||||
super().__init__(completion_stream)
|
||||
self.model = model
|
||||
|
||||
sent_first_chunk: bool = False
|
||||
sent_content_block_start: bool = False
|
||||
sent_content_block_finish: bool = False
|
||||
|
|
@ -39,6 +35,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
sent_last_message: bool = False
|
||||
holding_chunk: Optional[Any] = None
|
||||
holding_stop_reason_chunk: Optional[Any] = None
|
||||
queued_usage_chunk: bool = False
|
||||
current_content_block_index: int = 0
|
||||
current_content_block_start: ContentBlockContentBlockDict = TextBlock(
|
||||
type="text",
|
||||
|
|
@ -47,6 +44,10 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
pending_new_content_block: bool = False
|
||||
chunk_queue: deque = deque() # Queue for buffering multiple chunks
|
||||
|
||||
def __init__(self, completion_stream: Any, model: str):
|
||||
super().__init__(completion_stream)
|
||||
self.model = model
|
||||
|
||||
def __next__(self):
|
||||
from .transformation import LiteLLMAnthropicMessagesAdapter
|
||||
|
||||
|
|
@ -217,77 +218,83 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
|
||||
# Queue the merged chunk and reset
|
||||
self.chunk_queue.append(merged_chunk)
|
||||
self.queued_usage_chunk = True
|
||||
self.holding_stop_reason_chunk = None
|
||||
return self.chunk_queue.popleft()
|
||||
|
||||
# Check if this processed chunk has a stop_reason - hold it for next chunk
|
||||
|
||||
if should_start_new_block and not self.sent_content_block_finish:
|
||||
# Queue the sequence: content_block_stop -> content_block_start -> current_chunk
|
||||
if not self.queued_usage_chunk:
|
||||
if should_start_new_block and not self.sent_content_block_finish:
|
||||
# Queue the sequence: content_block_stop -> content_block_start -> current_chunk
|
||||
|
||||
# 1. Stop current content block
|
||||
self.chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_stop",
|
||||
"index": max(self.current_content_block_index - 1, 0),
|
||||
}
|
||||
)
|
||||
# 1. Stop current content block
|
||||
self.chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_stop",
|
||||
"index": max(self.current_content_block_index - 1, 0),
|
||||
}
|
||||
)
|
||||
|
||||
# 2. Start new content block
|
||||
self.chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": self.current_content_block_index,
|
||||
"content_block": self.current_content_block_start,
|
||||
}
|
||||
)
|
||||
# 2. Start new content block
|
||||
self.chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": self.current_content_block_index,
|
||||
"content_block": self.current_content_block_start,
|
||||
}
|
||||
)
|
||||
|
||||
# 3. Queue the current chunk (don't lose it!)
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
|
||||
# Reset state for new block
|
||||
self.sent_content_block_finish = False
|
||||
|
||||
# Return the first queued item
|
||||
return self.chunk_queue.popleft()
|
||||
|
||||
if (
|
||||
processed_chunk["type"] == "message_delta"
|
||||
and self.sent_content_block_finish is False
|
||||
):
|
||||
# Queue both the content_block_stop and the holding chunk
|
||||
self.chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_stop",
|
||||
"index": self.current_content_block_index,
|
||||
}
|
||||
)
|
||||
self.sent_content_block_finish = True
|
||||
if processed_chunk.get("delta", {}).get("stop_reason") is not None:
|
||||
|
||||
self.holding_stop_reason_chunk = processed_chunk
|
||||
else:
|
||||
# 3. Queue the current chunk (don't lose it!)
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
return self.chunk_queue.popleft()
|
||||
elif self.holding_chunk is not None:
|
||||
# Queue both chunks
|
||||
self.chunk_queue.append(self.holding_chunk)
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
self.holding_chunk = None
|
||||
return self.chunk_queue.popleft()
|
||||
else:
|
||||
# Queue the current chunk
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
return self.chunk_queue.popleft()
|
||||
|
||||
# Reset state for new block
|
||||
self.sent_content_block_finish = False
|
||||
|
||||
# Return the first queued item
|
||||
return self.chunk_queue.popleft()
|
||||
|
||||
if (
|
||||
processed_chunk["type"] == "message_delta"
|
||||
and self.sent_content_block_finish is False
|
||||
):
|
||||
# Queue both the content_block_stop and the holding chunk
|
||||
self.chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_stop",
|
||||
"index": self.current_content_block_index,
|
||||
}
|
||||
)
|
||||
self.sent_content_block_finish = True
|
||||
if (
|
||||
processed_chunk.get("delta", {}).get("stop_reason")
|
||||
is not None
|
||||
):
|
||||
|
||||
self.holding_stop_reason_chunk = processed_chunk
|
||||
else:
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
return self.chunk_queue.popleft()
|
||||
elif self.holding_chunk is not None:
|
||||
# Queue both chunks
|
||||
self.chunk_queue.append(self.holding_chunk)
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
self.holding_chunk = None
|
||||
return self.chunk_queue.popleft()
|
||||
else:
|
||||
# Queue the current chunk
|
||||
self.chunk_queue.append(processed_chunk)
|
||||
return self.chunk_queue.popleft()
|
||||
|
||||
# Handle any remaining held chunks after stream ends
|
||||
if self.holding_stop_reason_chunk is not None:
|
||||
self.chunk_queue.append(self.holding_stop_reason_chunk)
|
||||
self.holding_stop_reason_chunk = None
|
||||
if not self.queued_usage_chunk:
|
||||
if self.holding_stop_reason_chunk is not None:
|
||||
self.chunk_queue.append(self.holding_stop_reason_chunk)
|
||||
self.holding_stop_reason_chunk = None
|
||||
|
||||
if self.holding_chunk is not None:
|
||||
self.chunk_queue.append(self.holding_chunk)
|
||||
self.holding_chunk = None
|
||||
if self.holding_chunk is not None:
|
||||
self.chunk_queue.append(self.holding_chunk)
|
||||
self.holding_chunk = None
|
||||
|
||||
if not self.sent_last_message:
|
||||
self.sent_last_message = True
|
||||
|
|
|
|||
|
|
@ -124,15 +124,13 @@ class BedrockBatchesConfig(BaseAWSLLM, BaseBatchesConfig):
|
|||
"AWS IAM role ARN is required for Bedrock batch jobs. "
|
||||
"Set 'aws_batch_role_arn' in litellm_params or AWS_BATCH_ROLE_ARN env var"
|
||||
)
|
||||
|
||||
|
||||
# Get the actual Bedrock model ID using common utility
|
||||
bedrock_model_id = self.common_utils.extract_model_from_s3_file_path(input_file_id, optional_params)
|
||||
|
||||
if not bedrock_model_id:
|
||||
raise ValueError("Could not determine Bedrock model ID. Ensure the model is specified in the input file or passed as a parameter.")
|
||||
if not model:
|
||||
raise ValueError("Could not determine Bedrock model ID. Please pass `model` in your request body.")
|
||||
|
||||
# Generate job name with the correct model ID using common utility
|
||||
job_name = self.common_utils.generate_unique_job_name(bedrock_model_id, prefix="litellm")
|
||||
job_name = self.common_utils.generate_unique_job_name(model, prefix="litellm")
|
||||
output_key = f"litellm-batch-outputs/{job_name}/"
|
||||
|
||||
# Build input data config
|
||||
|
|
@ -151,7 +149,7 @@ class BedrockBatchesConfig(BaseAWSLLM, BaseBatchesConfig):
|
|||
|
||||
# Create Bedrock batch request with proper typing
|
||||
bedrock_request: BedrockCreateBatchRequest = {
|
||||
"modelId": bedrock_model_id,
|
||||
"modelId": model,
|
||||
"jobName": job_name,
|
||||
"inputDataConfig": input_data_config,
|
||||
"outputDataConfig": output_data_config,
|
||||
|
|
|
|||
|
|
@ -6,6 +6,8 @@ from typing import Any, Dict, List, Optional, Tuple, Union
|
|||
|
||||
from httpx import Headers, Response
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.files.utils import FilesAPIUtils
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.files.transformation import (
|
||||
|
|
@ -21,6 +23,7 @@ from litellm.types.llms.openai import (
|
|||
PathLike,
|
||||
)
|
||||
from litellm.types.utils import ExtractedFileData, LlmProviders
|
||||
from litellm.utils import get_llm_provider
|
||||
|
||||
from ..base_aws_llm import BaseAWSLLM
|
||||
from ..common_utils import BedrockError
|
||||
|
|
@ -111,6 +114,10 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
# Remove bedrock/ prefix if present
|
||||
if _model.startswith("bedrock/"):
|
||||
_model = _model[8:]
|
||||
|
||||
# Replace colons with hyphens for Bedrock S3 URI compliance
|
||||
_model = _model.replace(":", "-")
|
||||
|
||||
object_name = f"litellm-bedrock-files-{_model}-{uuid.uuid4()}.jsonl"
|
||||
return object_name
|
||||
|
||||
|
|
@ -191,24 +198,6 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
) -> dict:
|
||||
return optional_params
|
||||
|
||||
def _get_bedrock_provider_from_model(self, model: str) -> Optional[str]:
|
||||
"""
|
||||
Extract provider from Bedrock model name
|
||||
"""
|
||||
if model.startswith("anthropic."):
|
||||
return "anthropic"
|
||||
elif model.startswith("cohere."):
|
||||
return "cohere"
|
||||
elif model.startswith("meta.") or model.startswith("llama"):
|
||||
return "meta"
|
||||
elif model.startswith("mistral."):
|
||||
return "mistral"
|
||||
elif model.startswith("ai21."):
|
||||
return "ai21"
|
||||
elif model.startswith("amazon."):
|
||||
return "amazon"
|
||||
else:
|
||||
return None
|
||||
|
||||
def _map_openai_to_bedrock_params(
|
||||
self,
|
||||
|
|
@ -218,11 +207,12 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
"""
|
||||
Transform OpenAI request body to Bedrock-compatible modelInput parameters using existing transformation logic
|
||||
"""
|
||||
from litellm.types.utils import LlmProviders
|
||||
_model = openai_request_body.get("model", "")
|
||||
messages = openai_request_body.get("messages", [])
|
||||
|
||||
# Use existing Anthropic transformation logic for Anthropic models
|
||||
if provider == "anthropic":
|
||||
if provider == LlmProviders.ANTHROPIC:
|
||||
from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import (
|
||||
AmazonAnthropicClaudeConfig,
|
||||
)
|
||||
|
|
@ -231,16 +221,22 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
|
||||
# Extract optional params (everything except model and messages)
|
||||
optional_params = {k: v for k, v in openai_request_body.items() if k not in ["model", "messages"]}
|
||||
mapped_params = anthropic_config.map_openai_params(
|
||||
non_default_params={},
|
||||
optional_params=optional_params,
|
||||
model=_model,
|
||||
drop_params=False
|
||||
)
|
||||
|
||||
# Transform using existing Anthropic logic
|
||||
bedrock_params = anthropic_config.transform_request(
|
||||
model=_model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
optional_params=mapped_params,
|
||||
litellm_params={},
|
||||
headers={}
|
||||
)
|
||||
|
||||
|
||||
return bedrock_params
|
||||
else:
|
||||
# For other providers, use basic mapping
|
||||
|
|
@ -278,9 +274,17 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
# Extract the request body from OpenAI format
|
||||
openai_body = _openai_jsonl_content.get("body", {})
|
||||
model = openai_body.get("model", "")
|
||||
|
||||
try:
|
||||
model, _, _, _ = get_llm_provider(
|
||||
model=model,
|
||||
custom_llm_provider=None,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"litellm.llms.bedrock.files.transformation.py::_transform_openai_jsonl_content_to_bedrock_jsonl_content() - Error inferring custom_llm_provider - {str(e)}")
|
||||
|
||||
# Determine provider from model name
|
||||
provider = self._get_bedrock_provider_from_model(model)
|
||||
provider = self.get_bedrock_invoke_provider(model)
|
||||
|
||||
# Transform to Bedrock modelInput format
|
||||
model_input = self._map_openai_to_bedrock_params(
|
||||
|
|
@ -315,11 +319,13 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
extracted_file_data = extract_file_data(file_data)
|
||||
extracted_file_data_content = extracted_file_data.get("content")
|
||||
|
||||
if extracted_file_data_content is None:
|
||||
raise ValueError("file content is required")
|
||||
|
||||
# Get and transform the file content
|
||||
if (
|
||||
create_file_data.get("purpose") == "batch"
|
||||
and extracted_file_data.get("content_type") == "application/jsonl"
|
||||
and extracted_file_data_content is not None
|
||||
if FilesAPIUtils.is_batch_jsonl_file(
|
||||
create_file_data=create_file_data,
|
||||
extracted_file_data=extracted_file_data,
|
||||
):
|
||||
## Transform JSONL content to Bedrock format
|
||||
original_file_content = self._get_content_from_openai_file(
|
||||
|
|
@ -357,6 +363,8 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
api_base=api_base,
|
||||
optional_params=optional_params,
|
||||
)
|
||||
|
||||
litellm_params["upload_url"] = api_base
|
||||
|
||||
# Return a dict that tells the HTTP handler exactly what to do
|
||||
return {
|
||||
|
|
@ -440,6 +448,56 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
|
||||
return dict(aws_request.headers), signed_body
|
||||
|
||||
def _convert_https_url_to_s3_uri(self, https_url: str) -> tuple[str, str]:
|
||||
"""
|
||||
Convert HTTPS S3 URL to s3:// URI format.
|
||||
|
||||
Args:
|
||||
https_url: HTTPS S3 URL (e.g., "https://s3.us-west-2.amazonaws.com/bucket/key")
|
||||
|
||||
Returns:
|
||||
Tuple of (s3_uri, filename)
|
||||
|
||||
Example:
|
||||
Input: "https://s3.us-west-2.amazonaws.com/litellm-proxy/file.jsonl"
|
||||
Output: ("s3://litellm-proxy/file.jsonl", "file.jsonl")
|
||||
"""
|
||||
import re
|
||||
|
||||
# Match HTTPS S3 URL patterns
|
||||
# Pattern 1: https://s3.region.amazonaws.com/bucket/key
|
||||
# Pattern 2: https://bucket.s3.region.amazonaws.com/key
|
||||
|
||||
pattern1 = r"https://s3\.([^.]+)\.amazonaws\.com/([^/]+)/(.+)"
|
||||
pattern2 = r"https://([^.]+)\.s3\.([^.]+)\.amazonaws\.com/(.+)"
|
||||
|
||||
match1 = re.match(pattern1, https_url)
|
||||
match2 = re.match(pattern2, https_url)
|
||||
|
||||
if match1:
|
||||
# Pattern: https://s3.region.amazonaws.com/bucket/key
|
||||
region, bucket, key = match1.groups()
|
||||
s3_uri = f"s3://{bucket}/{key}"
|
||||
elif match2:
|
||||
# Pattern: https://bucket.s3.region.amazonaws.com/key
|
||||
bucket, region, key = match2.groups()
|
||||
s3_uri = f"s3://{bucket}/{key}"
|
||||
else:
|
||||
# Fallback: try to extract bucket and key from URL path
|
||||
from urllib.parse import urlparse
|
||||
parsed = urlparse(https_url)
|
||||
path_parts = parsed.path.lstrip('/').split('/', 1)
|
||||
if len(path_parts) >= 2:
|
||||
bucket, key = path_parts[0], path_parts[1]
|
||||
s3_uri = f"s3://{bucket}/{key}"
|
||||
else:
|
||||
raise ValueError(f"Unable to parse S3 URL: {https_url}")
|
||||
|
||||
# Extract filename from key
|
||||
filename = key.split("/")[-1] if "/" in key else key
|
||||
|
||||
return s3_uri, filename
|
||||
|
||||
def transform_create_file_response(
|
||||
self,
|
||||
model: Optional[str],
|
||||
|
|
@ -452,21 +510,18 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
"""
|
||||
# For S3 uploads, we typically get an ETag and other metadata
|
||||
response_headers = raw_response.headers
|
||||
|
||||
# Extract S3 object information from the response
|
||||
# S3 PUT object returns ETag and other metadata in headers
|
||||
content_length = response_headers.get("Content-Length", "0")
|
||||
|
||||
# Extract bucket and key from the request URL or litellm_params
|
||||
bucket_name = litellm_params.get("s3_bucket_name") or os.getenv("AWS_S3_BUCKET_NAME")
|
||||
|
||||
# Generate file ID in S3 format
|
||||
object_key = getattr(logging_obj, 'object_key', None) or f"file-{int(time.time())}"
|
||||
file_id = f"s3://{bucket_name}/{object_key}"
|
||||
|
||||
# Extract filename from object key
|
||||
filename = object_key.split("/")[-1] if "/" in object_key else object_key
|
||||
|
||||
# Use the actual upload URL that was used for the S3 upload
|
||||
upload_url = litellm_params.get("upload_url")
|
||||
file_id: str = ""
|
||||
filename: str = ""
|
||||
if upload_url:
|
||||
# Convert HTTPS S3 URL to s3:// URI format
|
||||
file_id, filename = self._convert_https_url_to_s3_uri(upload_url)
|
||||
|
||||
return OpenAIFileObject(
|
||||
purpose="batch", # Default purpose for Bedrock files
|
||||
id=file_id,
|
||||
|
|
|
|||
|
|
@ -2201,7 +2201,6 @@ class BaseLLMHTTPHandler:
|
|||
litellm_params=litellm_params,
|
||||
optional_params={},
|
||||
)
|
||||
|
||||
if _is_async:
|
||||
return self.async_create_file(
|
||||
transformed_request=transformed_request,
|
||||
|
|
@ -2218,6 +2217,7 @@ class BaseLLMHTTPHandler:
|
|||
sync_httpx_client = _get_httpx_client()
|
||||
else:
|
||||
sync_httpx_client = client
|
||||
|
||||
|
||||
if isinstance(transformed_request, dict) and "method" in transformed_request:
|
||||
# Handle pre-signed requests (e.g., from Bedrock S3 uploads)
|
||||
|
|
@ -2283,11 +2283,15 @@ class BaseLLMHTTPHandler:
|
|||
provider_config=provider_config,
|
||||
)
|
||||
|
||||
# Store the upload URL in litellm_params for the transformation method
|
||||
litellm_params_with_url = dict(litellm_params)
|
||||
litellm_params_with_url["upload_url"] = api_base
|
||||
|
||||
return provider_config.transform_create_file_response(
|
||||
model=None,
|
||||
raw_response=upload_response,
|
||||
logging_obj=logging_obj,
|
||||
litellm_params=litellm_params,
|
||||
litellm_params=litellm_params_with_url,
|
||||
)
|
||||
|
||||
async def async_create_file(
|
||||
|
|
@ -2408,15 +2412,19 @@ class BaseLLMHTTPHandler:
|
|||
_is_async: bool = False,
|
||||
client: Optional[Union["HTTPHandler", "AsyncHTTPHandler"]] = None,
|
||||
timeout: Optional[Union[float, httpx.Timeout]] = None,
|
||||
model: Optional[str] = None,
|
||||
) -> Union["LiteLLMBatch", Coroutine[Any, Any, "LiteLLMBatch"]]:
|
||||
"""
|
||||
Creates a batch using provider-specific batch creation process
|
||||
"""
|
||||
# get config from model, custom llm provider
|
||||
if model is None:
|
||||
raise ValueError("model is required for create_batch")
|
||||
|
||||
headers = provider_config.validate_environment(
|
||||
api_key=api_key,
|
||||
headers=headers,
|
||||
model="",
|
||||
model=model,
|
||||
messages=[],
|
||||
optional_params={},
|
||||
litellm_params=litellm_params,
|
||||
|
|
@ -2425,7 +2433,7 @@ class BaseLLMHTTPHandler:
|
|||
api_base = provider_config.get_complete_batch_url(
|
||||
api_base=api_base,
|
||||
api_key=api_key,
|
||||
model="",
|
||||
model=model,
|
||||
optional_params={},
|
||||
litellm_params=litellm_params,
|
||||
data=create_batch_data,
|
||||
|
|
@ -2435,7 +2443,7 @@ class BaseLLMHTTPHandler:
|
|||
|
||||
# Get the transformed request data
|
||||
transformed_request = provider_config.transform_create_batch_request(
|
||||
model="",
|
||||
model=model,
|
||||
create_batch_data=create_batch_data,
|
||||
litellm_params=litellm_params,
|
||||
optional_params={},
|
||||
|
|
@ -2495,7 +2503,7 @@ class BaseLLMHTTPHandler:
|
|||
litellm_params_with_request = {**litellm_params, "original_batch_request": create_batch_data}
|
||||
|
||||
return provider_config.transform_create_batch_response(
|
||||
model=None,
|
||||
model=model,
|
||||
raw_response=batch_response,
|
||||
logging_obj=logging_obj,
|
||||
litellm_params=litellm_params_with_request,
|
||||
|
|
@ -2512,6 +2520,7 @@ class BaseLLMHTTPHandler:
|
|||
client: Optional[Union["HTTPHandler", "AsyncHTTPHandler"]] = None,
|
||||
timeout: Optional[Union[float, httpx.Timeout]] = None,
|
||||
create_batch_data: Optional["CreateBatchRequest"] = None,
|
||||
model: Optional[str] = None,
|
||||
):
|
||||
"""
|
||||
Async version of create_batch
|
||||
|
|
@ -2572,7 +2581,7 @@ class BaseLLMHTTPHandler:
|
|||
litellm_params_with_request = {**litellm_params, "original_batch_request": create_batch_data or {}}
|
||||
|
||||
return provider_config.transform_create_batch_response(
|
||||
model=None,
|
||||
model=model,
|
||||
raw_response=batch_response,
|
||||
logging_obj=logging_obj,
|
||||
litellm_params=litellm_params_with_request,
|
||||
|
|
|
|||
|
|
@ -1,21 +1,155 @@
|
|||
"""
|
||||
Cost calculator for DeepSeek Chat models.
|
||||
Cost calculator for Dashscope Chat models.
|
||||
|
||||
Handles prompt caching scenario.
|
||||
Handles tiered pricing and prompt caching scenarios.
|
||||
"""
|
||||
|
||||
from typing import Tuple
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
from litellm.types.utils import ModelInfo, Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
@dataclass
|
||||
class TokenBreakdown:
|
||||
"""Token breakdown for cost calculation."""
|
||||
text_tokens: int
|
||||
cached_tokens: int
|
||||
completion_tokens: int
|
||||
reasoning_tokens: int
|
||||
|
||||
|
||||
def _extract_token_breakdown(usage: Usage) -> TokenBreakdown:
|
||||
"""Extract token counts from usage, handling cached and reasoning tokens."""
|
||||
cached_tokens = 0
|
||||
if usage.prompt_tokens_details and hasattr(usage.prompt_tokens_details, "cached_tokens"):
|
||||
cached_tokens = usage.prompt_tokens_details.cached_tokens or 0
|
||||
|
||||
text_tokens = usage.prompt_tokens - cached_tokens
|
||||
|
||||
reasoning_tokens = 0
|
||||
if (hasattr(usage, "completion_tokens_details") and
|
||||
usage.completion_tokens_details and
|
||||
hasattr(usage.completion_tokens_details, "reasoning_tokens")):
|
||||
reasoning_tokens = usage.completion_tokens_details.reasoning_tokens or 0
|
||||
|
||||
completion_tokens = (usage.completion_tokens or 0) - reasoning_tokens
|
||||
|
||||
return TokenBreakdown(text_tokens, cached_tokens, completion_tokens, reasoning_tokens)
|
||||
|
||||
|
||||
def _calculate_tiered_cost(
|
||||
tokens: int,
|
||||
tiered_pricing: List[dict],
|
||||
cost_key: str,
|
||||
fallback_cost_key: Optional[str] = None
|
||||
) -> float:
|
||||
"""Calculate cost using tiered pricing structure.
|
||||
|
||||
Finds the appropriate tier based on token count and applies that tier's rate to all tokens.
|
||||
"""
|
||||
if not tiered_pricing or tokens <= 0:
|
||||
return 0.0
|
||||
|
||||
# Find the appropriate tier for the token count
|
||||
for tier in tiered_pricing:
|
||||
tier_range = tier.get("range", [])
|
||||
if len(tier_range) != 2:
|
||||
continue
|
||||
|
||||
range_start, range_end = tier_range
|
||||
|
||||
# Check if tokens fall within this tier's range
|
||||
if range_start <= tokens <= range_end:
|
||||
cost_per_token = tier.get(cost_key) or tier.get(fallback_cost_key, 0)
|
||||
return tokens * cost_per_token
|
||||
|
||||
# If no tier matches, use the last tier (highest tier)
|
||||
if tiered_pricing:
|
||||
last_tier = tiered_pricing[-1]
|
||||
cost_per_token = last_tier.get(cost_key) or last_tier.get(fallback_cost_key, 0)
|
||||
return tokens * cost_per_token
|
||||
|
||||
return 0.0
|
||||
|
||||
|
||||
def _calculate_flat_cost(tokens: int, cost_per_token: float) -> float:
|
||||
"""Calculate cost using flat pricing."""
|
||||
return tokens * cost_per_token
|
||||
|
||||
|
||||
def _calculate_prompt_cost(breakdown: TokenBreakdown, model_info: ModelInfo, tiered_pricing: Optional[List[dict]]) -> float:
|
||||
"""Calculate total prompt cost including cached tokens."""
|
||||
if tiered_pricing:
|
||||
text_cost = _calculate_tiered_cost(
|
||||
tokens=breakdown.text_tokens,
|
||||
tiered_pricing=tiered_pricing,
|
||||
cost_key="input_cost_per_token"
|
||||
)
|
||||
cache_cost = _calculate_tiered_cost(
|
||||
tokens=breakdown.cached_tokens,
|
||||
tiered_pricing=tiered_pricing,
|
||||
cost_key="cache_read_input_token_cost"
|
||||
)
|
||||
return text_cost + cache_cost
|
||||
|
||||
input_cost = model_info.get("input_cost_per_token", 0.0)
|
||||
cache_cost = model_info.get("cache_read_input_token_cost", input_cost) or input_cost
|
||||
|
||||
return (_calculate_flat_cost(tokens=breakdown.text_tokens, cost_per_token=input_cost) +
|
||||
_calculate_flat_cost(tokens=breakdown.cached_tokens, cost_per_token=cache_cost))
|
||||
|
||||
|
||||
def _calculate_completion_cost(breakdown: TokenBreakdown, model_info: ModelInfo, tiered_pricing: Optional[List[dict]]) -> float:
|
||||
"""Calculate total completion cost including reasoning tokens."""
|
||||
if tiered_pricing:
|
||||
completion_cost = _calculate_tiered_cost(
|
||||
tokens=breakdown.completion_tokens,
|
||||
tiered_pricing=tiered_pricing,
|
||||
cost_key="output_cost_per_token"
|
||||
)
|
||||
reasoning_cost = _calculate_tiered_cost(
|
||||
tokens=breakdown.reasoning_tokens,
|
||||
tiered_pricing=tiered_pricing,
|
||||
cost_key="output_cost_per_reasoning_token",
|
||||
fallback_cost_key="output_cost_per_token"
|
||||
)
|
||||
return completion_cost + reasoning_cost
|
||||
|
||||
output_cost = model_info.get("output_cost_per_token", 0.0)
|
||||
reasoning_cost = model_info.get("output_cost_per_reasoning_token", output_cost) or output_cost
|
||||
|
||||
return (_calculate_flat_cost(tokens=breakdown.completion_tokens, cost_per_token=output_cost) +
|
||||
_calculate_flat_cost(tokens=breakdown.reasoning_tokens, cost_per_token=reasoning_cost))
|
||||
|
||||
|
||||
def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]:
|
||||
"""
|
||||
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
|
||||
|
||||
Follows the same logic as Anthropic's cost per token calculation.
|
||||
Calculate cost per token for Dashscope models.
|
||||
|
||||
Supports both tiered and flat pricing with cached and reasoning tokens.
|
||||
|
||||
Args:
|
||||
model: Model name without provider prefix
|
||||
usage: LiteLLM Usage block
|
||||
|
||||
Returns:
|
||||
Tuple[float, float] - (prompt_cost_in_usd, completion_cost_in_usd)
|
||||
"""
|
||||
return generic_cost_per_token(
|
||||
model=model, usage=usage, custom_llm_provider="deepseek"
|
||||
model_info = get_model_info(model=model, custom_llm_provider="dashscope")
|
||||
breakdown = _extract_token_breakdown(usage)
|
||||
tiered_pricing = model_info.get("tiered_pricing") if isinstance(model_info.get("tiered_pricing"), list) else None
|
||||
|
||||
prompt_cost = _calculate_prompt_cost(
|
||||
breakdown=breakdown,
|
||||
model_info=model_info,
|
||||
tiered_pricing=tiered_pricing
|
||||
)
|
||||
completion_cost = _calculate_completion_cost(
|
||||
breakdown=breakdown,
|
||||
model_info=model_info,
|
||||
tiered_pricing=tiered_pricing
|
||||
)
|
||||
|
||||
return prompt_cost, completion_cost
|
||||
|
|
|
|||
|
|
@ -169,18 +169,20 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
|
|||
if tool is None:
|
||||
return None
|
||||
|
||||
kwags: dict = {
|
||||
# Build DatabricksFunction explicitly to avoid parameter conflicts
|
||||
function_params: DatabricksFunction = {
|
||||
"name": tool["name"],
|
||||
"parameters": cast(dict, tool.get("input_schema") or {})
|
||||
}
|
||||
|
||||
|
||||
# Only add description if it exists
|
||||
description = tool.get("description")
|
||||
if description is not None:
|
||||
kwags["description"] = cast(Union[dict, str], description)
|
||||
function_params["description"] = cast(Union[dict, str], description)
|
||||
|
||||
return DatabricksTool(
|
||||
type="function",
|
||||
function=DatabricksFunction(name=tool["name"], **kwags),
|
||||
function=function_params,
|
||||
)
|
||||
|
||||
def _map_openai_to_dbrx_tool(self, model: str, tools: List) -> List[DatabricksTool]:
|
||||
|
|
|
|||
|
|
@ -11,7 +11,39 @@ if TYPE_CHECKING:
|
|||
else:
|
||||
GenerateContentContentListUnionDict = Any
|
||||
|
||||
|
||||
class GoogleAIStudioTokenCounter:
|
||||
def _clean_contents_for_gemini_api(self, contents: Any) -> Any:
|
||||
"""
|
||||
Clean up contents to remove unsupported fields for the Gemini API.
|
||||
|
||||
The Google Gemini API doesn't recognize the 'id' field in function responses,
|
||||
so we need to remove it to prevent 400 Bad Request errors.
|
||||
|
||||
Args:
|
||||
contents: The contents to clean up
|
||||
|
||||
Returns:
|
||||
Cleaned contents with unsupported fields removed
|
||||
"""
|
||||
import copy
|
||||
|
||||
from google.genai.types import FunctionResponse
|
||||
|
||||
cleaned_contents = copy.deepcopy(contents)
|
||||
|
||||
for content in cleaned_contents:
|
||||
parts = content["parts"]
|
||||
for part in parts:
|
||||
if "functionResponse" in part:
|
||||
function_response_data = part["functionResponse"]
|
||||
function_response_part = FunctionResponse(**function_response_data)
|
||||
function_response_part.id = None
|
||||
part["functionResponse"] = function_response_part.model_dump(
|
||||
exclude_none=True
|
||||
)
|
||||
|
||||
return cleaned_contents
|
||||
|
||||
def _construct_url(self, model: str, api_base: Optional[str] = None) -> str:
|
||||
"""
|
||||
|
|
@ -20,7 +52,6 @@ class GoogleAIStudioTokenCounter:
|
|||
base_url = api_base or "https://generativelanguage.googleapis.com"
|
||||
return f"{base_url}/v1beta/models/{model}:countTokens"
|
||||
|
||||
|
||||
async def validate_environment(
|
||||
self,
|
||||
api_base: Optional[str] = None,
|
||||
|
|
@ -33,7 +64,8 @@ class GoogleAIStudioTokenCounter:
|
|||
Returns a Tuple of headers and url for the Google Gen AI Studio countTokens endpoint.
|
||||
"""
|
||||
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
|
||||
headers = GoogleGenAIConfig().validate_environment(
|
||||
|
||||
headers = GoogleGenAIConfig().validate_environment(
|
||||
api_key=api_key,
|
||||
headers=headers,
|
||||
model=model,
|
||||
|
|
@ -54,7 +86,7 @@ class GoogleAIStudioTokenCounter:
|
|||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Count tokens using Google Gen AI Studio countTokens endpoint.
|
||||
|
||||
|
||||
Args:
|
||||
contents: The content to count tokens for (Google Gen AI format)
|
||||
Example: [{"parts": [{"text": "Hello world"}]}]
|
||||
|
|
@ -63,7 +95,7 @@ class GoogleAIStudioTokenCounter:
|
|||
api_base: Optional API base URL (defaults to Google Gen AI Studio)
|
||||
timeout: Optional timeout for the request
|
||||
**kwargs: Additional parameters
|
||||
|
||||
|
||||
Returns:
|
||||
Dict containing token count information from Google Gen AI Studio API.
|
||||
Example response:
|
||||
|
|
@ -77,14 +109,13 @@ class GoogleAIStudioTokenCounter:
|
|||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
Raises:
|
||||
ValueError: If API key is missing
|
||||
litellm.APIError: If the API call fails
|
||||
litellm.APIConnectionError: If the connection fails
|
||||
Exception: For any other unexpected errors
|
||||
"""
|
||||
# Set up API base URL
|
||||
|
||||
# Prepare headers
|
||||
headers, url = await self.validate_environment(
|
||||
|
|
@ -94,46 +125,40 @@ class GoogleAIStudioTokenCounter:
|
|||
model=model,
|
||||
litellm_params=kwargs,
|
||||
)
|
||||
|
||||
# Prepare request body
|
||||
request_body = {
|
||||
"contents": contents
|
||||
}
|
||||
|
||||
|
||||
# Prepare request body - clean up contents to remove unsupported fields
|
||||
cleaned_contents = self._clean_contents_for_gemini_api(contents)
|
||||
request_body = {"contents": cleaned_contents}
|
||||
|
||||
async_httpx_client = get_async_httpx_client(
|
||||
llm_provider=LlmProviders.GEMINI,
|
||||
)
|
||||
|
||||
try:
|
||||
response = await async_httpx_client.post(
|
||||
url=url,
|
||||
headers=headers,
|
||||
json=request_body
|
||||
url=url, headers=headers, json=request_body
|
||||
)
|
||||
|
||||
|
||||
# Check for HTTP errors
|
||||
response.raise_for_status()
|
||||
|
||||
|
||||
# Parse response
|
||||
result = response.json()
|
||||
return result
|
||||
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
error_msg = f"Google Gen AI Studio API error: {e.response.status_code} - {e.response.text}"
|
||||
raise litellm.APIError(
|
||||
message=error_msg,
|
||||
llm_provider="gemini",
|
||||
model=model,
|
||||
status_code=e.response.status_code
|
||||
status_code=e.response.status_code,
|
||||
) from e
|
||||
except httpx.RequestError as e:
|
||||
error_msg = f"Request to Google Gen AI Studio failed: {str(e)}"
|
||||
raise litellm.APIConnectionError(
|
||||
message=error_msg,
|
||||
llm_provider="gemini",
|
||||
model=model
|
||||
message=error_msg, llm_provider="gemini", model=model
|
||||
) from e
|
||||
except Exception as e:
|
||||
error_msg = f"Unexpected error during token counting: {str(e)}"
|
||||
raise Exception(error_msg) from e
|
||||
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ from typing import Any, Dict, List, Optional, Tuple, Union
|
|||
|
||||
from httpx import Headers, Response
|
||||
|
||||
from litellm.files.utils import FilesAPIUtils
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.files.transformation import (
|
||||
|
|
@ -260,10 +261,13 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
|
|||
raise ValueError("file is required")
|
||||
extracted_file_data = extract_file_data(file_data)
|
||||
extracted_file_data_content = extracted_file_data.get("content")
|
||||
if (
|
||||
create_file_data.get("purpose") == "batch"
|
||||
and extracted_file_data.get("content_type") == "application/jsonl"
|
||||
and extracted_file_data_content is not None
|
||||
|
||||
if extracted_file_data_content is None:
|
||||
raise ValueError("file content is required")
|
||||
|
||||
if FilesAPIUtils.is_batch_jsonl_file(
|
||||
create_file_data=create_file_data,
|
||||
extracted_file_data=extracted_file_data,
|
||||
):
|
||||
## 1. If jsonl, check if there's a model name
|
||||
file_content = self._get_content_from_openai_file(
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""
|
||||
Transformation for Calling Google models in their native format.
|
||||
"""
|
||||
from typing import Literal, Optional, Union
|
||||
from typing import Dict, Literal, Optional, Union
|
||||
|
||||
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
|
@ -11,20 +11,20 @@ class VertexAIGoogleGenAIConfig(GoogleGenAIConfig):
|
|||
"""
|
||||
Configuration for calling Google models in their native format.
|
||||
"""
|
||||
|
||||
HEADER_NAME = "Authorization"
|
||||
BEARER_PREFIX = "Bearer"
|
||||
|
||||
|
||||
@property
|
||||
def custom_llm_provider(self) -> Literal["gemini", "vertex_ai"]:
|
||||
return "vertex_ai"
|
||||
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
self,
|
||||
api_key: Optional[str],
|
||||
headers: Optional[dict],
|
||||
model: str,
|
||||
litellm_params: Optional[Union[GenericLiteLLMParams, dict]]
|
||||
litellm_params: Optional[Union[GenericLiteLLMParams, dict]],
|
||||
) -> dict:
|
||||
default_headers = {
|
||||
"Content-Type": "application/json",
|
||||
|
|
@ -36,4 +36,65 @@ class VertexAIGoogleGenAIConfig(GoogleGenAIConfig):
|
|||
default_headers.update(headers)
|
||||
|
||||
return default_headers
|
||||
|
||||
|
||||
def _camel_to_snake(self, camel_str: str) -> str:
|
||||
"""Convert camelCase to snake_case"""
|
||||
import re
|
||||
|
||||
return re.sub(r"(?<!^)(?=[A-Z])", "_", camel_str).lower()
|
||||
|
||||
def map_generate_content_optional_params(
|
||||
self,
|
||||
generate_content_config_dict,
|
||||
model: str,
|
||||
):
|
||||
"""
|
||||
Map Google GenAI parameters to provider-specific format.
|
||||
|
||||
Args:
|
||||
generate_content_optional_params: Optional parameters for generate content
|
||||
model: The model name
|
||||
|
||||
Returns:
|
||||
Mapped parameters for the provider
|
||||
"""
|
||||
from litellm.types.google_genai.main import GenerateContentConfigDict
|
||||
|
||||
_generate_content_config_dict = GenerateContentConfigDict()
|
||||
|
||||
for param, value in generate_content_config_dict.items():
|
||||
camel_case_key = self._camel_to_snake(param)
|
||||
_generate_content_config_dict[camel_case_key] = value
|
||||
return dict(_generate_content_config_dict)
|
||||
|
||||
def transform_generate_content_request(
|
||||
self,
|
||||
model: str,
|
||||
contents: any,
|
||||
tools: Optional[any],
|
||||
generate_content_config_dict: Dict,
|
||||
system_instruction: Optional[any] = None,
|
||||
) -> dict:
|
||||
"""
|
||||
Transform the generate content request for Vertex AI.
|
||||
Since Vertex AI natively supports Google GenAI format, we can pass most fields directly.
|
||||
"""
|
||||
# Build the request in Google GenAI format that Vertex AI expects
|
||||
result = {
|
||||
"model": model,
|
||||
"contents": contents,
|
||||
}
|
||||
|
||||
# Add tools if provided
|
||||
if tools:
|
||||
result["tools"] = tools
|
||||
|
||||
# Add systemInstruction if provided
|
||||
if system_instruction:
|
||||
result["systemInstruction"] = system_instruction
|
||||
|
||||
# Handle generationConfig - Vertex AI expects it in the same format
|
||||
if generate_content_config_dict:
|
||||
result["generationConfig"] = generate_content_config_dict
|
||||
|
||||
return result
|
||||
|
|
|
|||
|
|
@ -150,9 +150,9 @@ from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
|
|||
from .llms.custom_llm import CustomLLM, custom_chat_llm_router
|
||||
from .llms.databricks.embed.handler import DatabricksEmbeddingHandler
|
||||
from .llms.deprecated_providers import aleph_alpha, palm
|
||||
from .llms.gemini.common_utils import get_api_key_from_env
|
||||
from .llms.groq.chat.handler import GroqChatCompletion
|
||||
from .llms.heroku.chat.transformation import HerokuChatConfig
|
||||
from .llms.gemini.common_utils import get_api_key_from_env
|
||||
from .llms.huggingface.embedding.handler import HuggingFaceEmbedding
|
||||
from .llms.nlp_cloud.chat.handler import completion as nlp_cloud_chat_completion
|
||||
from .llms.oci.chat.transformation import OCIChatConfig
|
||||
|
|
@ -358,7 +358,9 @@ async def acompletion(
|
|||
logprobs: Optional[bool] = None,
|
||||
top_logprobs: Optional[int] = None,
|
||||
deployment_id=None,
|
||||
reasoning_effort: Optional[Literal["none", "minimal", "low", "medium", "high", "default"]] = None,
|
||||
reasoning_effort: Optional[
|
||||
Literal["none", "minimal", "low", "medium", "high", "default"]
|
||||
] = None,
|
||||
safety_identifier: Optional[str] = None,
|
||||
# set api_base, api_version, api_key
|
||||
base_url: Optional[str] = None,
|
||||
|
|
@ -504,7 +506,9 @@ async def acompletion(
|
|||
}
|
||||
if custom_llm_provider is None:
|
||||
_, custom_llm_provider, _, _ = get_llm_provider(
|
||||
model=model, custom_llm_provider=custom_llm_provider, api_base=completion_kwargs.get("base_url", None)
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_base=completion_kwargs.get("base_url", None),
|
||||
)
|
||||
|
||||
fallbacks = fallbacks or litellm.model_fallbacks
|
||||
|
|
@ -899,7 +903,9 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
logit_bias: Optional[dict] = None,
|
||||
user: Optional[str] = None,
|
||||
# openai v1.0+ new params
|
||||
reasoning_effort: Optional[Literal["none", "minimal", "low", "medium", "high", "default"]] = None,
|
||||
reasoning_effort: Optional[
|
||||
Literal["none", "minimal", "low", "medium", "high", "default"]
|
||||
] = None,
|
||||
response_format: Optional[Union[dict, Type[BaseModel]]] = None,
|
||||
seed: Optional[int] = None,
|
||||
tools: Optional[List] = None,
|
||||
|
|
@ -1116,10 +1122,12 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
)
|
||||
|
||||
if provider_specific_header is not None:
|
||||
headers.update(ProviderSpecificHeaderUtils.get_provider_specific_headers(
|
||||
provider_specific_header=provider_specific_header,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
))
|
||||
headers.update(
|
||||
ProviderSpecificHeaderUtils.get_provider_specific_headers(
|
||||
provider_specific_header=provider_specific_header,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
)
|
||||
|
||||
if model_response is not None and hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
|
|
@ -1325,6 +1333,7 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
azure_scope=kwargs.get("azure_scope"),
|
||||
max_retries=max_retries,
|
||||
timeout=timeout,
|
||||
litellm_request_debug=kwargs.get("litellm_request_debug", False),
|
||||
)
|
||||
cast(LiteLLMLoggingObj, logging).update_environment_variables(
|
||||
model=model,
|
||||
|
|
@ -2712,9 +2721,7 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
)
|
||||
|
||||
api_key = (
|
||||
api_key
|
||||
or litellm.api_key
|
||||
or get_secret("VERCEL_AI_GATEWAY_API_KEY")
|
||||
api_key or litellm.api_key or get_secret("VERCEL_AI_GATEWAY_API_KEY")
|
||||
)
|
||||
|
||||
vercel_site_url = get_secret("VERCEL_SITE_URL") or "https://litellm.ai"
|
||||
|
|
@ -2730,7 +2737,7 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
vercel_headers.update(_headers)
|
||||
|
||||
headers = vercel_headers
|
||||
|
||||
|
||||
## Load Config
|
||||
config = litellm.VercelAIGatewayConfig.get_config()
|
||||
for k, v in config.items():
|
||||
|
|
@ -3712,7 +3719,9 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse:
|
|||
func_with_context = partial(ctx.run, func)
|
||||
|
||||
_, custom_llm_provider, _, _ = get_llm_provider(
|
||||
model=model, custom_llm_provider=custom_llm_provider, api_base=kwargs.get("api_base", None)
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_base=kwargs.get("api_base", None),
|
||||
)
|
||||
|
||||
# Await normally
|
||||
|
|
@ -5780,7 +5789,14 @@ async def ahealth_check(
|
|||
input=input or ["test"],
|
||||
),
|
||||
"audio_speech": lambda: litellm.aspeech(
|
||||
**{**_filter_model_params(model_params), **({"voice": "alloy"} if "voice" not in _filter_model_params(model_params) else {})},
|
||||
**{
|
||||
**_filter_model_params(model_params),
|
||||
**(
|
||||
{"voice": "alloy"}
|
||||
if "voice" not in _filter_model_params(model_params)
|
||||
else {}
|
||||
),
|
||||
},
|
||||
input=prompt or "test",
|
||||
),
|
||||
"audio_transcription": lambda: litellm.atranscription(
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@
|
|||
"input_cost_per_token": 0.0,
|
||||
"output_cost_per_token": 0.0,
|
||||
"output_cost_per_reasoning_token": 0.0,
|
||||
"input_cost_per_audio_token": 0.0,
|
||||
"litellm_provider": "one of https://docs.litellm.ai/docs/providers",
|
||||
"mode": "one of: chat, embedding, completion, image_generation, audio_transcription, audio_speech, image_generation, moderation, rerank",
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -19045,34 +19046,43 @@
|
|||
"max_tokens": 32768,
|
||||
"max_input_tokens": 30720,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 1.6e-06,
|
||||
"output_cost_per_token": 6.4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-latest": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo-latest": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"output_cost_per_reasoning_token": 5e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-30b-a3b": {
|
||||
"max_tokens": 131072,
|
||||
|
|
@ -19083,7 +19093,272 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-max-preview": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 258048,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 6e-06},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 2.4e-06, "output_cost_per_token": 1.2e-05},
|
||||
{"range": [128e3, 252e3], "input_cost_per_token": 3.0e-06, "output_cost_per_token": 1.5e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-flash": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-coder": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.5e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-plus": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06, "cache_read_input_token_cost": 1e-07},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06, "cache_read_input_token_cost": 1.8e-07},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "cache_read_input_token_cost": 3e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05, "cache_read_input_token_cost": 6e-07}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-plus-2025-07-22": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-flash": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06, "cache_read_input_token_cost": 8e-08},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06, "cache_read_input_token_cost": 1.2e-07},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06, "cache_read_input_token_cost": 2e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06, "cache_read_input_token_cost": 4e-07}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-flash-2025-07-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-09-11": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-07-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-07-14": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_reasoning_token": 4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-04-28": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_reasoning_token": 4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-01-25": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-flash-2025-07-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"output_cost_per_reasoning_token": 5e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo-2025-04-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"output_cost_per_reasoning_token": 5e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo-2024-11-01": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwq-plus": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 98304,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 8e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"moonshot/moonshot-v1-8k": {
|
||||
"max_tokens": 8192,
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
BIN
litellm/proxy/_experimental/out/assets/logos/qwen.png
Normal file
BIN
litellm/proxy/_experimental/out/assets/logos/qwen.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 48 KiB |
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[75832,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","220","static/chunks/220-1c8d82f7ce7658c4.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-8dc8d9524a1f3965.js"],"default",1]
|
||||
3:I[30628,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","220","static/chunks/220-5061c4cea850d728.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-127adcf8da2b5294.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
|
||||
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
|
||||
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
1
litellm/proxy/_experimental/out/onboarding.html
Normal file
1
litellm/proxy/_experimental/out/onboarding.html
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -2,6 +2,6 @@
|
|||
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","50","static/chunks/50-fe160ecfa8bc4059.js","154","static/chunks/154-fff436ed72b19a24.js","461","static/chunks/app/onboarding/page-3c5840c907b0a5c8.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
|
|
@ -1617,6 +1617,20 @@ class ConfigList(LiteLLMPydanticObjectBase):
|
|||
)
|
||||
|
||||
|
||||
class UserHeaderMapping(LiteLLMPydanticObjectBase):
|
||||
"""
|
||||
Map an incoming HTTP header to a LiteLLM user role.
|
||||
"""
|
||||
header_name: str
|
||||
litellm_user_role: Literal[
|
||||
LitellmUserRoles.INTERNAL_USER,
|
||||
LitellmUserRoles.CUSTOMER,
|
||||
]
|
||||
|
||||
model_config = {
|
||||
"extra": "forbid",
|
||||
}
|
||||
|
||||
class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
|
||||
"""
|
||||
Documents all the fields supported by `general_settings` in config.yaml
|
||||
|
|
@ -1721,6 +1735,11 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
|
|||
default=None,
|
||||
description="Set-up pass-through endpoints for provider-specific endpoints. Docs - https://docs.litellm.ai/docs/proxy/pass_through",
|
||||
)
|
||||
user_header_name: Optional[str] = Field(
|
||||
None,
|
||||
description="[DEPRECATED] Use 'user_header_mappings' instead. When set, the header value is treated as the end user id unless overridden by user_header_mappings.",
|
||||
)
|
||||
user_header_mappings: Optional[List[UserHeaderMapping]] = None
|
||||
|
||||
|
||||
class ConfigYAML(LiteLLMPydanticObjectBase):
|
||||
|
|
|
|||
|
|
@ -473,6 +473,22 @@ def _has_user_setup_sso():
|
|||
|
||||
return sso_setup
|
||||
|
||||
def get_customer_user_header_from_mapping(user_id_mapping) -> Optional[str]:
|
||||
"""Return the header_name mapped to CUSTOMER role, if any (dict-based)."""
|
||||
if not user_id_mapping:
|
||||
return None
|
||||
items = user_id_mapping if isinstance(user_id_mapping, list) else [user_id_mapping]
|
||||
for item in items:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
role = item.get("litellm_user_role")
|
||||
header_name = item.get("header_name")
|
||||
if role is None or not header_name:
|
||||
continue
|
||||
if str(role).lower() == str(LitellmUserRoles.CUSTOMER).lower():
|
||||
return header_name
|
||||
return None
|
||||
|
||||
|
||||
def get_end_user_id_from_request_body(
|
||||
request_body: dict, request_headers: Optional[dict] = None
|
||||
|
|
@ -481,20 +497,34 @@ def get_end_user_id_from_request_body(
|
|||
# and to ensure it's fetched at runtime.
|
||||
from litellm.proxy.proxy_server import general_settings
|
||||
|
||||
# Check 1: Custom Header from general_settings.user_header_name (only if request_headers is provided)
|
||||
# Check 1 : Follow the user header mappings feature, if not found, then check for deprecated user_header_name (only if request_headers is provided)
|
||||
# User query: "system not respecting user_header_name property"
|
||||
# This implies the key in general_settings is 'user_header_name'.
|
||||
if request_headers is not None:
|
||||
user_id_header_config_key = "user_header_name"
|
||||
custom_header_name_to_check: Optional[str] = None
|
||||
|
||||
custom_header_name_to_check = general_settings.get(user_id_header_config_key)
|
||||
# Prefer user mappings (new behavior)
|
||||
user_id_mapping = general_settings.get("user_header_mappings", None)
|
||||
if user_id_mapping:
|
||||
custom_header_name_to_check = get_customer_user_header_from_mapping(
|
||||
user_id_mapping
|
||||
)
|
||||
|
||||
if custom_header_name_to_check and isinstance(custom_header_name_to_check, str):
|
||||
# Fallback to deprecated user_header_name if mapping did not specify
|
||||
if not custom_header_name_to_check:
|
||||
user_id_header_config_key = "user_header_name"
|
||||
value = general_settings.get(user_id_header_config_key)
|
||||
if isinstance(value, str) and value.strip() != "":
|
||||
custom_header_name_to_check = value
|
||||
|
||||
# If we have a header name to check, try to read it from request headers
|
||||
if isinstance(custom_header_name_to_check, str):
|
||||
for header_name, header_value in request_headers.items():
|
||||
if header_name.lower() == custom_header_name_to_check.lower():
|
||||
user_id_from_header = header_value
|
||||
if user_id_from_header.strip():
|
||||
return str(user_id_from_header)
|
||||
user_id_str = str(user_id_from_header) if user_id_from_header is not None else ""
|
||||
if user_id_str.strip():
|
||||
return user_id_str
|
||||
|
||||
# Check 2: 'user' field in request_body (commonly OpenAI)
|
||||
if "user" in request_body and request_body["user"] is not None:
|
||||
|
|
|
|||
|
|
@ -18,6 +18,19 @@ model_list:
|
|||
litellm_params:
|
||||
model: "groq/*"
|
||||
api_key: os.environ/GROQ_API_KEY
|
||||
- model_name: bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
#########################################################
|
||||
########## batch specific params ########################
|
||||
s3_bucket_name: litellm-proxy
|
||||
s3_region_name: us-west-2
|
||||
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
|
||||
model_info:
|
||||
mode: batch
|
||||
|
||||
litellm_settings:
|
||||
# set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production
|
||||
drop_params: True
|
||||
|
|
|
|||
|
|
@ -17,13 +17,13 @@ except ImportError:
|
|||
# List of all available hooks that can be enabled
|
||||
PROXY_HOOKS = {
|
||||
"max_budget_limiter": _PROXY_MaxBudgetLimiter,
|
||||
"parallel_request_limiter": _PROXY_MaxParallelRequestsHandler,
|
||||
"parallel_request_limiter": _PROXY_MaxParallelRequestsHandler_v3,
|
||||
"cache_control_check": _PROXY_CacheControlCheck,
|
||||
}
|
||||
|
||||
## FEATURE FLAG HOOKS ##
|
||||
if os.getenv("EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING", "false").lower() == "true":
|
||||
PROXY_HOOKS["parallel_request_limiter"] = _PROXY_MaxParallelRequestsHandler_v3
|
||||
if os.getenv("LEGACY_MULTI_INSTANCE_RATE_LIMITING", "false").lower() == "true":
|
||||
PROXY_HOOKS["parallel_request_limiter"] = _PROXY_MaxParallelRequestsHandler
|
||||
|
||||
|
||||
### update PROXY_HOOKS with ENTERPRISE_PROXY_HOOKS ###
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ This is currently in development and not yet ready for production.
|
|||
|
||||
import os
|
||||
from datetime import datetime
|
||||
from math import floor
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
|
|
@ -17,7 +18,7 @@ from typing import (
|
|||
Union,
|
||||
cast,
|
||||
)
|
||||
from math import floor
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
from litellm import DualCache
|
||||
|
|
@ -95,6 +96,7 @@ end
|
|||
return results
|
||||
"""
|
||||
|
||||
|
||||
class RateLimitDescriptorRateLimitObject(TypedDict, total=False):
|
||||
requests_per_unit: Optional[int]
|
||||
tokens_per_unit: Optional[int]
|
||||
|
|
@ -266,7 +268,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
if current_limit is None or rate_limit_type is None:
|
||||
continue
|
||||
|
||||
if counter_value is not None and int(counter_value) + 1 > current_limit:
|
||||
if counter_value is not None and int(counter_value) > current_limit:
|
||||
overall_code = "OVER_LIMIT"
|
||||
item_code = "OVER_LIMIT"
|
||||
|
||||
|
|
@ -480,10 +482,15 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
},
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
# Team Member rate limits
|
||||
if user_api_key_dict.user_id and (user_api_key_dict.team_member_rpm_limit is not None or user_api_key_dict.team_member_tpm_limit is not None):
|
||||
team_member_value = f"{user_api_key_dict.team_id}:{user_api_key_dict.user_id}"
|
||||
if user_api_key_dict.user_id and (
|
||||
user_api_key_dict.team_member_rpm_limit is not None
|
||||
or user_api_key_dict.team_member_tpm_limit is not None
|
||||
):
|
||||
team_member_value = (
|
||||
f"{user_api_key_dict.team_id}:{user_api_key_dict.user_id}"
|
||||
)
|
||||
descriptors.append(
|
||||
RateLimitDescriptor(
|
||||
key="team_member",
|
||||
|
|
@ -557,13 +564,13 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
# Find which descriptor hit the limit
|
||||
for i, status in enumerate(response["statuses"]):
|
||||
if status["code"] == "OVER_LIMIT":
|
||||
descriptor = descriptors[floor(i/2)]
|
||||
descriptor = descriptors[floor(i / 2)]
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=f"Rate limit exceeded for {descriptor['key']}: {descriptor['value']}. Remaining: {status['limit_remaining']}",
|
||||
headers={
|
||||
"retry-after": str(self.window_size),
|
||||
"rate_limit_type": str(status["rate_limit_type"])
|
||||
"rate_limit_type": str(status["rate_limit_type"]),
|
||||
}, # Retry after 1 minute
|
||||
)
|
||||
|
||||
|
|
@ -613,7 +620,9 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
|
||||
# Check if script is available
|
||||
if self.token_increment_script is None:
|
||||
verbose_proxy_logger.debug("TTL preservation script not available, using regular pipeline")
|
||||
verbose_proxy_logger.debug(
|
||||
"TTL preservation script not available, using regular pipeline"
|
||||
)
|
||||
await self.internal_usage_cache.dual_cache.async_increment_cache_pipeline(
|
||||
increment_list=pipeline_operations,
|
||||
litellm_parent_otel_span=parent_otel_span,
|
||||
|
|
@ -628,7 +637,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
for op in pipeline_operations:
|
||||
# Convert None TTL to 0 for Lua script
|
||||
ttl_value = op["ttl"] if op["ttl"] is not None else 0
|
||||
|
||||
|
||||
verbose_proxy_logger.debug(
|
||||
f"Executing TTL-preserving increment for key={op['key']}, "
|
||||
f"increment={op['increment_value']}, ttl={ttl_value}"
|
||||
|
|
@ -693,16 +702,15 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
)
|
||||
|
||||
# Get metadata from kwargs
|
||||
user_api_key = kwargs["litellm_params"]["metadata"].get("user_api_key")
|
||||
user_api_key_user_id = kwargs["litellm_params"]["metadata"].get(
|
||||
"user_api_key_user_id"
|
||||
litellm_metadata = kwargs["litellm_params"]["metadata"]
|
||||
if litellm_metadata is None:
|
||||
return
|
||||
user_api_key = litellm_metadata.get("user_api_key")
|
||||
user_api_key_user_id = litellm_metadata.get("user_api_key_user_id")
|
||||
user_api_key_team_id = litellm_metadata.get("user_api_key_team_id")
|
||||
user_api_key_end_user_id = kwargs.get("user") or litellm_metadata.get(
|
||||
"user_api_key_end_user_id"
|
||||
)
|
||||
user_api_key_team_id = kwargs["litellm_params"]["metadata"].get(
|
||||
"user_api_key_team_id"
|
||||
)
|
||||
user_api_key_end_user_id = kwargs.get("user") or kwargs["litellm_params"][
|
||||
"metadata"
|
||||
].get("user_api_key_end_user_id")
|
||||
model_group = get_model_group_from_litellm_kwargs(kwargs)
|
||||
|
||||
# Get total tokens from response
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from litellm.proxy._types import (
|
|||
SpecialHeaders,
|
||||
TeamCallbackMetadata,
|
||||
UserAPIKeyAuth,
|
||||
LitellmUserRoles,
|
||||
)
|
||||
from litellm.proxy.auth.route_checks import RouteChecks
|
||||
from litellm.router import Router
|
||||
|
|
@ -335,6 +336,22 @@ class LiteLLMProxyRequestSetup:
|
|||
return value
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def add_internal_user_from_user_mapping(general_settings: Optional[Dict], user_api_key_dict: UserAPIKeyAuth, headers: dict) -> UserAPIKeyAuth:
|
||||
if general_settings is None:
|
||||
return user_api_key_dict
|
||||
user_header_mapping = general_settings.get("user_header_mappings")
|
||||
if not user_header_mapping:
|
||||
return user_api_key_dict
|
||||
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(user_header_mapping)
|
||||
if not header_name:
|
||||
return user_api_key_dict
|
||||
header_value = LiteLLMProxyRequestSetup._get_case_insensitive_header(headers, header_name)
|
||||
if header_value:
|
||||
user_api_key_dict.user_id = header_value
|
||||
return user_api_key_dict
|
||||
return user_api_key_dict
|
||||
|
||||
@staticmethod
|
||||
def get_user_from_headers(
|
||||
headers: dict, general_settings: Optional[Dict] = None
|
||||
|
|
@ -428,6 +445,26 @@ class LiteLLMProxyRequestSetup:
|
|||
data["headers"] = _headers
|
||||
return data
|
||||
|
||||
@staticmethod
|
||||
def get_internal_user_header_from_mapping(user_header_mapping) -> Optional[str]:
|
||||
if not user_header_mapping:
|
||||
return None
|
||||
items = (
|
||||
user_header_mapping
|
||||
if isinstance(user_header_mapping, list)
|
||||
else [user_header_mapping]
|
||||
)
|
||||
for item in items:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
role = item.get("litellm_user_role")
|
||||
header_name = item.get("header_name")
|
||||
if role is None or not header_name:
|
||||
continue
|
||||
if str(role).lower() == str(LitellmUserRoles.INTERNAL_USER).lower():
|
||||
return header_name
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def add_litellm_data_for_backend_llm_call(
|
||||
*,
|
||||
|
|
@ -726,6 +763,8 @@ async def add_litellm_data_to_request( # noqa: PLR0915
|
|||
data=data, headers=_headers, user_api_key_dict=user_api_key_dict
|
||||
)
|
||||
|
||||
user_api_key_dict = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(general_settings, user_api_key_dict, _headers)
|
||||
|
||||
# Parse user info from headers
|
||||
user = LiteLLMProxyRequestSetup.get_user_from_headers(_headers, general_settings)
|
||||
if user is not None:
|
||||
|
|
|
|||
|
|
@ -346,6 +346,7 @@ def handle_key_type(data: GenerateKeyRequest, data_json: dict) -> dict:
|
|||
data_json["allowed_routes"] = ["info_routes"]
|
||||
return data_json
|
||||
|
||||
|
||||
async def validate_team_id_used_in_service_account_request(
|
||||
team_id: Optional[str],
|
||||
prisma_client: Optional[PrismaClient],
|
||||
|
|
@ -358,13 +359,13 @@ async def validate_team_id_used_in_service_account_request(
|
|||
status_code=400,
|
||||
detail="team_id is required for service account keys. Please specify `team_id` in the request body.",
|
||||
)
|
||||
|
||||
|
||||
if prisma_client is None:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="prisma_client is required for service account keys. Please specify `prisma_client` in the request body.",
|
||||
)
|
||||
|
||||
|
||||
# check if team_id exists in the database
|
||||
team = await prisma_client.db.litellm_teamtable.find_unique(
|
||||
where={"team_id": team_id},
|
||||
|
|
@ -376,6 +377,7 @@ async def validate_team_id_used_in_service_account_request(
|
|||
)
|
||||
return True
|
||||
|
||||
|
||||
async def _common_key_generation_helper( # noqa: PLR0915
|
||||
data: GenerateKeyRequest,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
|
|
@ -557,7 +559,7 @@ async def _common_key_generation_helper( # noqa: PLR0915
|
|||
status_code=400,
|
||||
detail={
|
||||
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {data.key}"
|
||||
}
|
||||
},
|
||||
)
|
||||
|
||||
response = await generate_key_helper_fn(
|
||||
|
|
@ -2885,7 +2887,10 @@ async def unblock_key(
|
|||
param="key",
|
||||
code=status.HTTP_400_BAD_REQUEST,
|
||||
)
|
||||
hashed_token = hash_token(token=data.key)
|
||||
if data.key.startswith("sk-"):
|
||||
hashed_token = hash_token(token=data.key)
|
||||
else:
|
||||
hashed_token = data.key
|
||||
|
||||
if litellm.store_audit_logs is True:
|
||||
# make an audit log for key update
|
||||
|
|
|
|||
|
|
@ -1,18 +1,13 @@
|
|||
model_list:
|
||||
- model_name: db-openai-endpoint
|
||||
- model_name: bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
api_base: https://exampleopenaiendpoint-production-0ee2.up.railway.app/
|
||||
- model_name: bedrock/*
|
||||
litellm_params:
|
||||
model: bedrock/*
|
||||
- model_name: openai/*
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
- model_name: gemini/*
|
||||
litellm_params:
|
||||
model: gemini/*
|
||||
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["cloudzero"]
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
#########################################################
|
||||
########## batch specific params ########################
|
||||
s3_bucket_name: litellm-proxy
|
||||
s3_region_name: us-west-2
|
||||
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
|
||||
model_info:
|
||||
mode: batch
|
||||
|
|
|
|||
|
|
@ -85,6 +85,7 @@ async def route_request(
|
|||
"""
|
||||
team_id = get_team_id_from_data(data)
|
||||
router_model_names = llm_router.model_names if llm_router is not None else []
|
||||
|
||||
if "api_key" in data or "api_base" in data:
|
||||
if llm_router is not None:
|
||||
return getattr(llm_router, f"{route_type}")(**data)
|
||||
|
|
@ -123,24 +124,20 @@ async def route_request(
|
|||
data["model"] in router_model_names
|
||||
or data["model"] in llm_router.get_model_ids()
|
||||
):
|
||||
|
||||
return getattr(llm_router, f"{route_type}")(**data)
|
||||
|
||||
elif (
|
||||
llm_router.model_group_alias is not None
|
||||
and data["model"] in llm_router.model_group_alias
|
||||
):
|
||||
|
||||
return getattr(llm_router, f"{route_type}")(**data)
|
||||
|
||||
elif data["model"] in llm_router.deployment_names:
|
||||
|
||||
return getattr(llm_router, f"{route_type}")(
|
||||
**data, specific_deployment=True
|
||||
)
|
||||
|
||||
elif data["model"] not in router_model_names:
|
||||
|
||||
if llm_router.router_general_settings.pass_through_all_models:
|
||||
return getattr(litellm, f"{route_type}")(**data)
|
||||
elif (
|
||||
|
|
@ -162,7 +159,6 @@ async def route_request(
|
|||
elif user_model is not None:
|
||||
return getattr(litellm, f"{route_type}")(**data)
|
||||
elif route_type == "allm_passthrough_route":
|
||||
|
||||
return getattr(litellm, f"{route_type}")(**data)
|
||||
|
||||
# if no route found then it's a bad request
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ from fastapi import APIRouter, Depends, HTTPException, status
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.router_strategy.budget_limiter import RouterBudgetLimiting
|
||||
from litellm.proxy._types import *
|
||||
from litellm.proxy._types import ProviderBudgetResponse, ProviderBudgetResponseObject
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
|
|
@ -2765,16 +2766,23 @@ async def provider_budgets() -> ProviderBudgetResponse:
|
|||
|
||||
provider_budget_response_dict: Dict[str, ProviderBudgetResponseObject] = {}
|
||||
for _provider, _budget_info in provider_budget_config.items():
|
||||
if llm_router.router_budget_logger is None:
|
||||
router_budget_logger = next(
|
||||
(
|
||||
cb
|
||||
for cb in (llm_router.optional_callbacks or [])
|
||||
if isinstance(cb, RouterBudgetLimiting)
|
||||
),
|
||||
None,
|
||||
)
|
||||
if router_budget_logger is None:
|
||||
raise ValueError("No router budget logger found")
|
||||
_provider_spend = (
|
||||
await llm_router.router_budget_logger._get_current_provider_spend(
|
||||
await router_budget_logger._get_current_provider_spend(_provider) or 0.0
|
||||
)
|
||||
_provider_budget_ttl = (
|
||||
await router_budget_logger._get_current_provider_budget_reset_at(
|
||||
_provider
|
||||
)
|
||||
or 0.0
|
||||
)
|
||||
_provider_budget_ttl = await llm_router.router_budget_logger._get_current_provider_budget_reset_at(
|
||||
_provider
|
||||
)
|
||||
provider_budget_response_object = ProviderBudgetResponseObject(
|
||||
budget_limit=_budget_info.max_budget,
|
||||
|
|
|
|||
|
|
@ -70,7 +70,8 @@ async def create_mcp_list_tools_events(
|
|||
mcp_tools_dict = []
|
||||
for tool in filtered_mcp_tools:
|
||||
if hasattr(tool, 'model_dump') and callable(getattr(tool, 'model_dump')):
|
||||
mcp_tools_dict.append(tool.model_dump())
|
||||
# Type cast to help mypy understand this is safe after hasattr check
|
||||
mcp_tools_dict.append(cast(Any, tool).model_dump())
|
||||
elif hasattr(tool, '__dict__'):
|
||||
mcp_tools_dict.append(tool.__dict__)
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -3046,7 +3046,7 @@ class Router:
|
|||
from litellm.router_utils.common_utils import add_model_file_id_mappings
|
||||
|
||||
verbose_router_logger.debug(
|
||||
f"Inside _atext_completion()- model: {model}; kwargs: {kwargs}"
|
||||
f"Inside _acreate_file()- model: {model}; kwargs: {kwargs}"
|
||||
)
|
||||
parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs)
|
||||
healthy_deployments = await self.async_get_healthy_deployments(
|
||||
|
|
|
|||
|
|
@ -162,6 +162,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
SearchContextCostPerQuery
|
||||
] # Cost for using web search tool
|
||||
citation_cost_per_token: Optional[float] # Cost per citation token for Perplexity
|
||||
tiered_pricing: Optional[List[Dict[str, Any]]] # Tiered pricing structure for models like Dashscope
|
||||
litellm_provider: Required[str]
|
||||
mode: Required[
|
||||
Literal[
|
||||
|
|
@ -1995,7 +1996,7 @@ class StandardLoggingGuardrailInformation(TypedDict, total=False):
|
|||
]
|
||||
guardrail_request: Optional[dict]
|
||||
guardrail_response: Optional[Union[dict, str, List[dict]]]
|
||||
guardrail_status: Literal["success", "failure","blocked"]
|
||||
guardrail_status: Literal["success", "failure", "blocked"]
|
||||
start_time: Optional[float]
|
||||
end_time: Optional[float]
|
||||
duration: Optional[float]
|
||||
|
|
@ -2123,6 +2124,7 @@ all_litellm_params = [
|
|||
"metadata",
|
||||
"litellm_metadata",
|
||||
"litellm_trace_id",
|
||||
"litellm_request_debug",
|
||||
"guardrails",
|
||||
"tags",
|
||||
"acompletion",
|
||||
|
|
|
|||
|
|
@ -2501,6 +2501,23 @@ def get_optional_params_transcription(
|
|||
return optional_params
|
||||
|
||||
|
||||
def _map_openai_size_to_vertex_ai_aspect_ratio(size: Optional[str]) -> str:
|
||||
"""Map OpenAI size parameter to Vertex AI aspectRatio."""
|
||||
if size is None:
|
||||
return "1:1"
|
||||
|
||||
# Map OpenAI size strings to Vertex AI aspect ratio strings
|
||||
# Vertex AI accepts: "1:1", "9:16", "16:9", "4:3", "3:4"
|
||||
size_to_aspect_ratio = {
|
||||
"256x256": "1:1", # Square
|
||||
"512x512": "1:1", # Square
|
||||
"1024x1024": "1:1", # Square (default)
|
||||
"1792x1024": "16:9", # Landscape
|
||||
"1024x1792": "9:16", # Portrait
|
||||
}
|
||||
return size_to_aspect_ratio.get(size, "1:1") # Default to square if size not recognized
|
||||
|
||||
|
||||
def get_optional_params_image_gen(
|
||||
model: Optional[str] = None,
|
||||
n: Optional[int] = None,
|
||||
|
|
@ -2614,19 +2631,7 @@ def get_optional_params_image_gen(
|
|||
|
||||
# Map OpenAI size parameter to Vertex AI aspectRatio
|
||||
if size is not None:
|
||||
# Map OpenAI size strings to Vertex AI aspect ratio strings
|
||||
# Vertex AI accepts: "1:1", "9:16", "16:9", "4:3", "3:4"
|
||||
size_to_aspect_ratio = {
|
||||
"256x256": "1:1", # Square
|
||||
"512x512": "1:1", # Square
|
||||
"1024x1024": "1:1", # Square (default)
|
||||
"1792x1024": "16:9", # Landscape
|
||||
"1024x1792": "9:16", # Portrait
|
||||
}
|
||||
aspect_ratio = size_to_aspect_ratio.get(
|
||||
size, "1:1"
|
||||
) # Default to square if size not recognized
|
||||
optional_params["aspectRatio"] = aspect_ratio
|
||||
optional_params["aspectRatio"] = _map_openai_size_to_vertex_ai_aspect_ratio(size)
|
||||
|
||||
openai_params: list[str] = list(default_params.keys())
|
||||
if provider_config is not None:
|
||||
|
|
@ -2642,6 +2647,12 @@ def get_optional_params_image_gen(
|
|||
openai_params=openai_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
)
|
||||
# remove keys with None or empty dict/list values to avoid sending empty payloads
|
||||
optional_params = {
|
||||
k: v
|
||||
for k, v in optional_params.items()
|
||||
if v is not None and (not isinstance(v, (dict, list)) or len(v) > 0)
|
||||
}
|
||||
return optional_params
|
||||
|
||||
|
||||
|
|
@ -4902,6 +4913,7 @@ def _get_model_info_helper( # noqa: PLR0915
|
|||
citation_cost_per_token=_model_info.get(
|
||||
"citation_cost_per_token", None
|
||||
),
|
||||
tiered_pricing=_model_info.get("tiered_pricing", None),
|
||||
litellm_provider=_model_info.get(
|
||||
"litellm_provider", custom_llm_provider
|
||||
),
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@
|
|||
"input_cost_per_token": 0.0,
|
||||
"output_cost_per_token": 0.0,
|
||||
"output_cost_per_reasoning_token": 0.0,
|
||||
"input_cost_per_audio_token": 0.0,
|
||||
"litellm_provider": "one of https://docs.litellm.ai/docs/providers",
|
||||
"mode": "one of: chat, embedding, completion, image_generation, audio_transcription, audio_speech, image_generation, moderation, rerank",
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -19045,34 +19046,43 @@
|
|||
"max_tokens": 32768,
|
||||
"max_input_tokens": 30720,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 1.6e-06,
|
||||
"output_cost_per_token": 6.4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-latest": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo-latest": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"output_cost_per_reasoning_token": 5e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-30b-a3b": {
|
||||
"max_tokens": 131072,
|
||||
|
|
@ -19083,7 +19093,272 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-max-preview": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 258048,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 6e-06},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 2.4e-06, "output_cost_per_token": 1.2e-05},
|
||||
{"range": [128e3, 252e3], "input_cost_per_token": 3.0e-06, "output_cost_per_token": 1.5e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-flash": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-coder": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.5e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-plus": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06, "cache_read_input_token_cost": 1e-07},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06, "cache_read_input_token_cost": 1.8e-07},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "cache_read_input_token_cost": 3e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05, "cache_read_input_token_cost": 6e-07}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-plus-2025-07-22": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 1e-06, "output_cost_per_token": 5e-06},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 1.8e-06, "output_cost_per_token": 9e-06},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 6e-06, "output_cost_per_token": 6e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-flash": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06, "cache_read_input_token_cost": 8e-08},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06, "cache_read_input_token_cost": 1.2e-07},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06, "cache_read_input_token_cost": 2e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06, "cache_read_input_token_cost": 4e-07}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen3-coder-flash-2025-07-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 65536,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 32e3], "input_cost_per_token": 3e-07, "output_cost_per_token": 1.5e-06},
|
||||
{"range": [32e3, 128e3], "input_cost_per_token": 5e-07, "output_cost_per_token": 2.5e-06},
|
||||
{"range": [128e3, 256e3], "input_cost_per_token": 8e-07, "output_cost_per_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.6e-06, "output_cost_per_token": 9.6e-06}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-09-11": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-07-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 4e-07, "output_cost_per_token": 1.2e-06, "output_cost_per_reasoning_token": 4e-06},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 1.2e-06, "output_cost_per_token": 3.6e-06, "output_cost_per_reasoning_token": 1.2e-05}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-07-14": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_reasoning_token": 4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-04-28": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_reasoning_token": 4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-plus-2025-01-25": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-flash-2025-07-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 997952,
|
||||
"max_output_tokens": 32768,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"tiered_pricing": [
|
||||
{"range": [0, 256e3], "input_cost_per_token": 5e-08, "output_cost_per_token": 4e-07},
|
||||
{"range": [256e3, 1e6], "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2e-06}
|
||||
],
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 129024,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"output_cost_per_reasoning_token": 5e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo-2025-04-28": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"output_cost_per_reasoning_token": 5e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwen-turbo-2024-11-01": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"dashscope/qwq-plus": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 98304,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 8e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"litellm_provider": "dashscope",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"mode": "chat",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models"
|
||||
},
|
||||
"moonshot/moonshot-v1-8k": {
|
||||
"max_tokens": 8192,
|
||||
|
|
|
|||
|
|
@ -1,3 +1,128 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
|
|
|
|||
|
|
@ -18,6 +18,8 @@ sys.path.insert(
|
|||
import pytest
|
||||
from typing import Optional
|
||||
import litellm
|
||||
from unittest.mock import patch, MagicMock
|
||||
import httpx
|
||||
|
||||
|
||||
@pytest.mark.asyncio()
|
||||
|
|
@ -64,7 +66,55 @@ async def test_async_file_and_batch():
|
|||
input_file_id=file_obj.id,
|
||||
metadata={"key1": "value1", "key2": "value2"},
|
||||
custom_llm_provider="bedrock",
|
||||
|
||||
#########################################################
|
||||
# bedrock specific params
|
||||
#########################################################
|
||||
model="us.anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
aws_batch_role_arn="arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV"
|
||||
)
|
||||
print("CREATED BATCH RESPONSE=", create_batch_response)
|
||||
|
||||
|
||||
@pytest.mark.asyncio()
|
||||
async def test_mock_bedrock_file_url_mapping():
|
||||
"""
|
||||
Simple test to capture PUT URL and validate mapping to file ID.
|
||||
"""
|
||||
print("Testing Bedrock file URL mapping")
|
||||
|
||||
captured_put_url = None
|
||||
|
||||
async def mock_async_create_file(transformed_request, **kwargs):
|
||||
nonlocal captured_put_url
|
||||
# Capture PUT URL from transformed request
|
||||
if isinstance(transformed_request, dict) and "url" in transformed_request:
|
||||
captured_put_url = transformed_request["url"]
|
||||
|
||||
# Call the real method to get actual response
|
||||
from litellm.files.main import base_llm_http_handler
|
||||
return await base_llm_http_handler.__class__.async_create_file(
|
||||
base_llm_http_handler, transformed_request, **kwargs
|
||||
)
|
||||
|
||||
with patch('litellm.files.main.base_llm_http_handler.async_create_file', side_effect=mock_async_create_file):
|
||||
file_obj = await litellm.acreate_file(
|
||||
file=open(os.path.join(os.path.dirname(__file__), "bedrock_batch_completions.jsonl"), "rb"),
|
||||
purpose="batch",
|
||||
custom_llm_provider="bedrock",
|
||||
s3_bucket_name="litellm-proxy",
|
||||
)
|
||||
|
||||
print(f"PUT URL: {captured_put_url}")
|
||||
print(f"File ID: {file_obj.id}")
|
||||
|
||||
# Validate URL was captured and response is correct
|
||||
assert captured_put_url is not None
|
||||
assert file_obj.id.startswith("s3://")
|
||||
|
||||
# Verify mapping
|
||||
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
|
||||
bedrock_config = BedrockFilesConfig()
|
||||
expected_s3_uri, _ = bedrock_config._convert_https_url_to_s3_uri(captured_put_url)
|
||||
assert file_obj.id == expected_s3_uri
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,4 @@
|
|||
{
|
||||
"model": "gpt-image-1",
|
||||
"prompt": "test prompt"
|
||||
}
|
||||
|
|
@ -5,6 +5,7 @@ import logging
|
|||
import os
|
||||
import sys
|
||||
import traceback
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
|
||||
sys.path.insert(
|
||||
|
|
@ -329,3 +330,33 @@ async def test_aiml_image_generation_with_dynamic_api_key():
|
|||
assert captured_json_data is not None
|
||||
assert captured_json_data["prompt"] == "A cute baby sea otter"
|
||||
assert captured_json_data["model"] == "flux-pro/v1.1"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_azure_image_generation_request_body():
|
||||
from litellm import aimage_generation
|
||||
test_dir = os.path.dirname(__file__)
|
||||
expected_path = os.path.join(
|
||||
test_dir, "request_payloads", "azure_gpt_image_1.json"
|
||||
)
|
||||
with open(expected_path, "r") as f:
|
||||
expected_body = json.load(f)
|
||||
|
||||
with patch(
|
||||
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
|
||||
new_callable=AsyncMock,
|
||||
) as mock_post:
|
||||
mock_post.side_effect = Exception("test")
|
||||
|
||||
with pytest.raises(Exception):
|
||||
await aimage_generation(
|
||||
model="azure/gpt-image-1",
|
||||
prompt="test prompt",
|
||||
api_base="https://example.azure.com",
|
||||
api_key="test-key",
|
||||
api_version="2025-04-01-preview",
|
||||
)
|
||||
|
||||
mock_post.assert_called_once()
|
||||
call_args = mock_post.call_args
|
||||
request_json = call_args.kwargs.get("json", {})
|
||||
assert request_json == expected_body
|
||||
|
|
|
|||
|
|
@ -259,3 +259,55 @@ def test_get_model_from_request(request_data, expected_model):
|
|||
model = get_model_from_request(request_data, "/v1/files")
|
||||
assert model == ["gpt-3.5-turbo", "gpt-4o-mini-general-deployment"]
|
||||
|
||||
|
||||
def test_get_customer_user_header_from_mapping_returns_customer_header():
|
||||
from litellm.proxy.auth.auth_utils import get_customer_user_header_from_mapping
|
||||
|
||||
mappings = [
|
||||
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
|
||||
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
|
||||
]
|
||||
result = get_customer_user_header_from_mapping(mappings)
|
||||
assert result == "X-OpenWebUI-User-Email"
|
||||
|
||||
|
||||
def test_get_customer_user_header_from_mapping_no_customer_returns_none():
|
||||
from litellm.proxy.auth.auth_utils import get_customer_user_header_from_mapping
|
||||
|
||||
mappings = [
|
||||
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"}
|
||||
]
|
||||
result = get_customer_user_header_from_mapping(mappings)
|
||||
assert result is None
|
||||
|
||||
# Also support a single mapping dict
|
||||
single_mapping = {"header_name": "X-Only-Internal", "litellm_user_role": "internal_user"}
|
||||
result = get_customer_user_header_from_mapping(single_mapping)
|
||||
assert result is None
|
||||
|
||||
|
||||
def test_get_internal_user_header_from_mapping_returns_internal_header():
|
||||
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
|
||||
|
||||
mappings = [
|
||||
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
|
||||
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
|
||||
]
|
||||
|
||||
result = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
|
||||
assert result == "X-OpenWebUI-User-Id"
|
||||
|
||||
|
||||
def test_get_internal_user_header_from_mapping_no_internal_returns_none():
|
||||
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
|
||||
|
||||
mappings = [
|
||||
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"}
|
||||
]
|
||||
result = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
|
||||
assert result is None
|
||||
|
||||
# Also support single mapping dict
|
||||
single_mapping = {"header_name": "X-Only-Customer", "litellm_user_role": "customer"}
|
||||
result = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(single_mapping)
|
||||
assert result is None
|
||||
|
|
|
|||
|
|
@ -1,5 +1,7 @@
|
|||
import os
|
||||
import sys
|
||||
import uuid
|
||||
from functools import partial
|
||||
from typing import Optional
|
||||
|
||||
import pytest
|
||||
|
|
@ -146,24 +148,36 @@ async def test_pass_through_endpoint_rerank(client):
|
|||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"auth, rpm_limit, expected_error_code",
|
||||
[(True, 0, 429), (True, 1, 200), (False, 0, 200)],
|
||||
"auth, rpm_limit, requests_to_make, expected_status_codes, num_users",
|
||||
[
|
||||
# Single user tests
|
||||
(True, 0, 1, [429], 1),
|
||||
(True, 1, 1, [200], 1),
|
||||
(True, 1, 2, [200, 429], 1),
|
||||
(True, 2, 4, [200, 200, 429, 429], 1),
|
||||
(True, 3, 4, [200, 200, 200, 429], 1),
|
||||
(True, 4, 4, [200, 200, 200, 200], 1),
|
||||
(False, 0, 1, [200], 1),
|
||||
(False, 0, 4, [200, 200, 200, 200], 1),
|
||||
# Multiple user tests (same parameters as single user)
|
||||
(True, 0, 1, [429], 2),
|
||||
(True, 1, 1, [200], 2),
|
||||
(True, 1, 2, [200, 429], 2),
|
||||
(True, 2, 4, [200, 200, 429, 429], 2),
|
||||
(True, 3, 4, [200, 200, 200, 429], 2),
|
||||
(True, 4, 4, [200, 200, 200, 200], 2),
|
||||
(False, 0, 1, [200], 2),
|
||||
(False, 0, 4, [200, 200, 200, 200], 2),
|
||||
],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_pass_through_endpoint_rpm_limit(
|
||||
client, auth, expected_error_code, rpm_limit
|
||||
client, auth, rpm_limit, requests_to_make, expected_status_codes, num_users
|
||||
):
|
||||
import litellm
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.proxy_server import ProxyLogging, hash_token, user_api_key_cache
|
||||
|
||||
mock_api_key = "sk-my-test-key"
|
||||
cache_value = UserAPIKeyAuth(token=hash_token(mock_api_key), rpm_limit=rpm_limit)
|
||||
|
||||
_cohere_api_key = os.environ.get("COHERE_API_KEY")
|
||||
|
||||
user_api_key_cache.set_cache(key=hash_token(mock_api_key), value=cache_value)
|
||||
|
||||
proxy_logging_obj = ProxyLogging(user_api_key_cache=user_api_key_cache)
|
||||
proxy_logging_obj._init_litellm_callbacks()
|
||||
|
||||
|
|
@ -173,6 +187,7 @@ async def test_pass_through_endpoint_rpm_limit(
|
|||
setattr(litellm.proxy.proxy_server, "proxy_logging_obj", proxy_logging_obj)
|
||||
|
||||
# Define a pass-through endpoint
|
||||
_cohere_api_key = os.environ.get("COHERE_API_KEY")
|
||||
pass_through_endpoints = [
|
||||
{
|
||||
"path": "/v1/rerank",
|
||||
|
|
@ -190,6 +205,13 @@ async def test_pass_through_endpoint_rpm_limit(
|
|||
general_settings.update({"pass_through_endpoints": pass_through_endpoints})
|
||||
setattr(litellm.proxy.proxy_server, "general_settings", general_settings)
|
||||
|
||||
# Setup API keys and cache
|
||||
mock_api_keys = [f"sk-test-{uuid.uuid4().hex}" for _ in range(num_users)]
|
||||
|
||||
for mock_api_key in mock_api_keys:
|
||||
cache_value = UserAPIKeyAuth(token=hash_token(mock_api_key), rpm_limit=rpm_limit)
|
||||
user_api_key_cache.set_cache(key=hash_token(mock_api_key), value=cache_value)
|
||||
|
||||
_json_data = {
|
||||
"model": "rerank-english-v3.0",
|
||||
"query": "What is the capital of the United States?",
|
||||
|
|
@ -200,16 +222,134 @@ async def test_pass_through_endpoint_rpm_limit(
|
|||
}
|
||||
|
||||
# Make a request to the pass-through endpoint
|
||||
response = client.post(
|
||||
"/v1/rerank",
|
||||
json=_json_data,
|
||||
headers={"Authorization": "Bearer {}".format(mock_api_key)},
|
||||
)
|
||||
tasks = []
|
||||
for mock_api_key in mock_api_keys:
|
||||
for _ in range(requests_to_make):
|
||||
task = asyncio.get_running_loop().run_in_executor(
|
||||
None,
|
||||
partial(
|
||||
client.post,
|
||||
"/v1/rerank",
|
||||
json=_json_data,
|
||||
headers={"Authorization": "Bearer {}".format(mock_api_key)},
|
||||
),
|
||||
)
|
||||
tasks.append(task)
|
||||
|
||||
responses = await asyncio.gather(*tasks)
|
||||
|
||||
if num_users == 1:
|
||||
status_codes = sorted([response.status_code for response in responses])
|
||||
|
||||
assert status_codes == sorted(expected_status_codes)
|
||||
else:
|
||||
first_user_responses = responses[requests_to_make:]
|
||||
second_user_responses = responses[:requests_to_make]
|
||||
|
||||
first_user_status_codes = sorted([response.status_code for response in first_user_responses])
|
||||
second_user_status_codes = sorted([response.status_code for response in second_user_responses])
|
||||
|
||||
expected_status_codes.sort()
|
||||
assert first_user_status_codes == expected_status_codes
|
||||
assert second_user_status_codes == expected_status_codes
|
||||
|
||||
print("JSON response: ", _json_data)
|
||||
|
||||
# Assert the response
|
||||
assert response.status_code == expected_error_code
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"auth, rpm_limit, requests_to_make, expected_status_codes",
|
||||
[
|
||||
# Multiple user tests (same parameters as single user)
|
||||
(True, 0, 1, [429]),
|
||||
(True, 1, 1, [200]),
|
||||
(True, 1, 2, [200, 429]),
|
||||
(True, 2, 4, [200, 200, 429, 429]),
|
||||
(True, 3, 4, [200, 200, 200, 429]),
|
||||
(True, 4, 4, [200, 200, 200, 200]),
|
||||
(False, 0, 1, [200]),
|
||||
(False, 0, 4, [200, 200, 200, 200]),
|
||||
],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_pass_through_endpoint_sequential_rpm_limit(
|
||||
client, auth, rpm_limit, requests_to_make, expected_status_codes
|
||||
):
|
||||
import litellm
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.proxy_server import ProxyLogging, hash_token, user_api_key_cache
|
||||
|
||||
proxy_logging_obj = ProxyLogging(user_api_key_cache=user_api_key_cache)
|
||||
proxy_logging_obj._init_litellm_callbacks()
|
||||
|
||||
setattr(litellm.proxy.proxy_server, "user_api_key_cache", user_api_key_cache)
|
||||
setattr(litellm.proxy.proxy_server, "master_key", "sk-1234")
|
||||
setattr(litellm.proxy.proxy_server, "prisma_client", "FAKE-VAR")
|
||||
setattr(litellm.proxy.proxy_server, "proxy_logging_obj", proxy_logging_obj)
|
||||
|
||||
# Define a pass-through endpoint
|
||||
_cohere_api_key = os.environ.get("COHERE_API_KEY")
|
||||
pass_through_endpoints = [
|
||||
{
|
||||
"path": "/v1/rerank",
|
||||
"target": "https://api.cohere.com/v1/rerank",
|
||||
"auth": auth,
|
||||
"headers": {"Authorization": f"bearer {_cohere_api_key}"},
|
||||
}
|
||||
]
|
||||
|
||||
# Initialize the pass-through endpoint
|
||||
await initialize_pass_through_endpoints(pass_through_endpoints)
|
||||
general_settings: Optional[dict] = (
|
||||
getattr(litellm.proxy.proxy_server, "general_settings", {}) or {}
|
||||
)
|
||||
general_settings.update({"pass_through_endpoints": pass_through_endpoints})
|
||||
setattr(litellm.proxy.proxy_server, "general_settings", general_settings)
|
||||
|
||||
# Setup API keys and cache
|
||||
mock_api_keys = [f"sk-test-{uuid.uuid4().hex}" for _ in range(2)]
|
||||
|
||||
for mock_api_key in mock_api_keys:
|
||||
cache_value = UserAPIKeyAuth(token=hash_token(mock_api_key), rpm_limit=rpm_limit)
|
||||
user_api_key_cache.set_cache(key=hash_token(mock_api_key), value=cache_value)
|
||||
|
||||
_json_data = {
|
||||
"model": "rerank-english-v3.0",
|
||||
"query": "What is the capital of the United States?",
|
||||
"top_n": 3,
|
||||
"documents": [
|
||||
"Carson City is the capital city of the American state of Nevada."
|
||||
],
|
||||
}
|
||||
|
||||
# Make a request to the pass-through endpoint
|
||||
first_user_responses = []
|
||||
second_user_responses = []
|
||||
for _ in range(requests_to_make):
|
||||
requests = []
|
||||
for mock_api_key in mock_api_keys:
|
||||
task = asyncio.get_running_loop().run_in_executor(
|
||||
None,
|
||||
partial(
|
||||
client.post,
|
||||
"/v1/rerank",
|
||||
json=_json_data,
|
||||
headers={"Authorization": "Bearer {}".format(mock_api_key)},
|
||||
),
|
||||
)
|
||||
requests.append(task)
|
||||
|
||||
first_user_response, second_user_response = await asyncio.gather(*requests)
|
||||
first_user_responses.append(first_user_response)
|
||||
second_user_responses.append(second_user_response)
|
||||
|
||||
first_user_status_codes = sorted([response.status_code for response in first_user_responses])
|
||||
second_user_status_codes = sorted([response.status_code for response in second_user_responses])
|
||||
|
||||
expected_status_codes.sort()
|
||||
assert first_user_status_codes == expected_status_codes
|
||||
assert second_user_status_codes == expected_status_codes
|
||||
|
||||
print("JSON response: ", _json_data)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
|
|||
|
|
@ -0,0 +1,2 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
37
tests/openai_endpoints_tests/test_bedrock_batches_api.py
Normal file
37
tests/openai_endpoints_tests/test_bedrock_batches_api.py
Normal file
|
|
@ -0,0 +1,37 @@
|
|||
from openai import OpenAI
|
||||
import pytest
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
|
||||
BEDROCK_BATCH_MODEL = "bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bedrock_batches_api():
|
||||
"""
|
||||
Test bedrock batches api
|
||||
|
||||
E2E Test Creating a File and a Batch on Bedrock
|
||||
"""
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("tests/openai_endpoints_tests/bedrock_batch_completions.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": BEDROCK_BATCH_MODEL}
|
||||
)
|
||||
print(batch_input_file)
|
||||
|
||||
# Create batch
|
||||
batch = client.batches.create(
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
|
||||
assert batch.id is not None
|
||||
350
tests/proxy_unit_tests/test_google_gemini_proxy_request.py
Normal file
350
tests/proxy_unit_tests/test_google_gemini_proxy_request.py
Normal file
|
|
@ -0,0 +1,350 @@
|
|||
"""
|
||||
Test case for Google Gemini API proxy request handling.
|
||||
|
||||
This test verifies that when a request comes to the proxy endpoint:
|
||||
http://localhost:4000/v1beta/models/gemini-2.5-flash:generateContent
|
||||
|
||||
The request payload is correctly processed and forwarded to the httpx client.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import unittest.mock
|
||||
from typing import Optional
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
# Add the parent directory to the system path
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
|
||||
import litellm
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.google_endpoints.endpoints import google_generate_content
|
||||
from litellm.proxy.proxy_server import ProxyConfig
|
||||
from litellm.proxy.utils import ProxyLogging
|
||||
from fastapi import Request, Response
|
||||
from fastapi.datastructures import Headers
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_request_payload():
|
||||
"""Sample request payload as provided in the user query."""
|
||||
return {
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "You are an interactive CLI agent specializing in software engineering tasks. Your primary goal is to help users safely and efficiently, adhering strictly to the following instructions and utilizing your available tools"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"parts": [{"text": "Got it. Thanks for the context!"}],
|
||||
"role": "model"
|
||||
},
|
||||
{
|
||||
"parts": [{"text": "Hello how are you"}],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"parts": [{"text": "I'm doing well, thank you! How can I help you today?\n"}],
|
||||
"role": "model"
|
||||
},
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "Analyze *only* the content and structure of your immediately preceding response (your last turn in the conversation history)."
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"systemInstruction": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "You are an interactive CLI agent specializing in software engineering tasks. Your primary goal is to help users safely and efficiently, adhering strictly to the following instructions and utilizing your available tools"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
"generationConfig": {
|
||||
"temperature": 0,
|
||||
"topP": 1,
|
||||
"responseMimeType": "application/json",
|
||||
"responseJsonSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"reasoning": {
|
||||
"type": "string",
|
||||
"description": "Brief explanation justifying the 'next_speaker' choice based *strictly* on the applicable rule and the content/structure of the preceding turn."
|
||||
},
|
||||
"next_speaker": {
|
||||
"type": "string",
|
||||
"enum": ["user", "model"],
|
||||
"description": "Who should speak next based *only* on the preceding turn and the decision rules"
|
||||
}
|
||||
},
|
||||
"required": ["reasoning", "next_speaker"]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_user_api_key_dict():
|
||||
"""Mock user API key dictionary."""
|
||||
return UserAPIKeyAuth(
|
||||
api_key="test_api_key",
|
||||
user_id="test_user_id",
|
||||
user_email="test@example.com",
|
||||
team_id="test_team_id",
|
||||
max_budget=100.0,
|
||||
spend=0.0,
|
||||
user_role="internal_user",
|
||||
allowed_cache_controls=[],
|
||||
metadata={},
|
||||
tpm_limit=None,
|
||||
rpm_limit=None,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_request(sample_request_payload):
|
||||
"""Create a mock FastAPI request with the sample payload."""
|
||||
mock_request = MagicMock(spec=Request)
|
||||
mock_request.headers = Headers({"content-type": "application/json"})
|
||||
mock_request.method = "POST"
|
||||
mock_request.url.path = "/v1beta/models/gemini-2.5-flash:generateContent"
|
||||
|
||||
# Mock the request body reading
|
||||
async def mock_body():
|
||||
return json.dumps(sample_request_payload).encode('utf-8')
|
||||
|
||||
mock_request.body = mock_body
|
||||
return mock_request
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_response():
|
||||
"""Create a mock FastAPI response."""
|
||||
return MagicMock(spec=Response)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_google_gemini_httpx_request_direct():
|
||||
"""
|
||||
Test that the Google Gemini generate_content_handler correctly processes the request
|
||||
and forwards it to the httpx client with the correct parameters.
|
||||
|
||||
This test directly calls the HTTP handler to verify the httpx integration.
|
||||
"""
|
||||
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
|
||||
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
# Sample request payload
|
||||
sample_payload = {
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "You are an interactive CLI agent specializing in software engineering tasks."
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"parts": [{"text": "Got it. Thanks for the context!"}],
|
||||
"role": "model"
|
||||
},
|
||||
{
|
||||
"parts": [{"text": "Hello how are you"}],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"systemInstruction": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "You are an interactive CLI agent specializing in software engineering tasks."
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
"config": { # Note: already transformed from generationConfig
|
||||
"temperature": 0,
|
||||
"topP": 1,
|
||||
"responseMimeType": "application/json",
|
||||
"responseJsonSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"reasoning": {"type": "string"},
|
||||
"next_speaker": {"type": "string", "enum": ["user", "model"]}
|
||||
},
|
||||
"required": ["reasoning", "next_speaker"]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Mock the HTTP handler to capture the request
|
||||
with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post") as mock_post:
|
||||
# Create mock response
|
||||
mock_http_response = MagicMock()
|
||||
mock_http_response.status_code = 200
|
||||
mock_http_response.json.return_value = {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": '{"reasoning": "The preceding response was a helpful greeting asking how to assist.", "next_speaker": "user"}'
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
mock_post.return_value = mock_http_response
|
||||
|
||||
# Create the HTTP handler and provider config
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
http_handler = BaseLLMHTTPHandler()
|
||||
provider_config = GoogleGenAIConfig()
|
||||
|
||||
# Create proper litellm params
|
||||
litellm_params = GenericLiteLLMParams(
|
||||
api_base="https://generativelanguage.googleapis.com",
|
||||
api_key="test_api_key"
|
||||
)
|
||||
|
||||
logging_obj = LiteLLMLoggingObj(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type="agenerate_content",
|
||||
start_time=None,
|
||||
litellm_call_id="test_call_id",
|
||||
function_id="test_function_id"
|
||||
)
|
||||
|
||||
try:
|
||||
# Call the generate_content_handler directly
|
||||
response = http_handler.generate_content_handler(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
contents=sample_payload["contents"],
|
||||
generate_content_provider_config=provider_config,
|
||||
generate_content_config_dict=sample_payload["config"],
|
||||
tools=None,
|
||||
custom_llm_provider="gemini",
|
||||
litellm_params=litellm_params,
|
||||
logging_obj=logging_obj,
|
||||
extra_headers=None,
|
||||
extra_body=None,
|
||||
timeout=30.0,
|
||||
_is_async=False,
|
||||
client=None,
|
||||
stream=False,
|
||||
litellm_metadata={}
|
||||
)
|
||||
|
||||
# Verify that the HTTP post was called
|
||||
assert mock_post.called, "Expected HTTP POST to be called"
|
||||
|
||||
# Get the call arguments
|
||||
call_args, call_kwargs = mock_post.call_args
|
||||
|
||||
print(f"POST call args: {call_args}")
|
||||
print(f"POST call kwargs: {call_kwargs}")
|
||||
|
||||
# Validate that the request data includes the expected fields
|
||||
request_data = call_kwargs.get('json')
|
||||
if request_data:
|
||||
assert 'contents' in request_data, "Expected 'contents' in request data"
|
||||
|
||||
# The config should be included in the request as generationConfig
|
||||
if 'generationConfig' in request_data:
|
||||
config = request_data['generationConfig']
|
||||
assert config['temperature'] == 0, "Expected temperature to be 0"
|
||||
assert config['topP'] == 1, "Expected topP to be 1"
|
||||
assert config['responseMimeType'] == "application/json", "Expected responseMimeType to be application/json"
|
||||
assert 'responseJsonSchema' in config, "Expected responseJsonSchema in config"
|
||||
|
||||
# Validate the responseJsonSchema structure
|
||||
schema = config['responseJsonSchema']
|
||||
assert schema['type'] == 'object', "Expected schema type to be object"
|
||||
assert 'properties' in schema, "Expected properties in schema"
|
||||
assert 'reasoning' in schema['properties'], "Expected reasoning property in schema"
|
||||
assert 'next_speaker' in schema['properties'], "Expected next_speaker property in schema"
|
||||
|
||||
print("✅ Request data validation passed")
|
||||
print(f"Request data: {json.dumps(request_data, indent=2)}")
|
||||
|
||||
# Validate URL contains the correct endpoint
|
||||
if call_args:
|
||||
url = call_args[0] if len(call_args) > 0 else call_kwargs.get('url')
|
||||
assert url is not None, "Expected URL to be provided"
|
||||
print(f"✅ URL validation passed: {url}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Exception occurred: {e}")
|
||||
|
||||
# Check if the HTTP handler was called despite the exception
|
||||
if mock_post.called:
|
||||
call_args, call_kwargs = mock_post.call_args
|
||||
print(f"HTTP POST was called with args: {call_args}")
|
||||
print(f"HTTP POST was called with kwargs: {call_kwargs}")
|
||||
|
||||
# Even with an exception, we can validate the request structure
|
||||
request_data = call_kwargs.get('json')
|
||||
if request_data:
|
||||
assert 'contents' in request_data, "Expected 'contents' in request data"
|
||||
if 'generationConfig' in request_data:
|
||||
config = request_data['generationConfig']
|
||||
assert config['temperature'] == 0, "Expected temperature to be 0"
|
||||
assert config['responseMimeType'] == "application/json", "Expected responseMimeType to be application/json"
|
||||
print("✅ Request structure validation passed despite exception")
|
||||
else:
|
||||
# If no HTTP call was made, re-raise the exception for debugging
|
||||
raise
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generationconfig_to_config_mapping(sample_request_payload):
|
||||
"""
|
||||
Test that generationConfig is correctly mapped to config parameter
|
||||
for Google GenAI compatibility in the main functions.
|
||||
"""
|
||||
from litellm.google_genai.main import agenerate_content
|
||||
|
||||
# Create a copy of the payload to avoid modifying the fixture
|
||||
test_data = sample_request_payload.copy()
|
||||
|
||||
# Test that agenerate_content can handle generationConfig parameter
|
||||
# This should not raise an error about parameter handling
|
||||
try:
|
||||
# This will fail due to missing API key, but should not fail due to parameter handling
|
||||
await agenerate_content(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
contents=test_data["contents"],
|
||||
generationConfig=test_data["generationConfig"], # Pass as generationConfig
|
||||
custom_llm_provider="gemini"
|
||||
)
|
||||
except Exception as e:
|
||||
# Should not fail due to parameter handling issues
|
||||
error_msg = str(e).lower()
|
||||
if "generationconfig" in error_msg or "config" in error_msg or "parameter" in error_msg:
|
||||
pytest.fail(f"Parameter handling failed: {e}")
|
||||
# Other errors (like API key missing) are expected
|
||||
print(f"✅ Parameter handling worked (API error expected): {type(e).__name__}")
|
||||
|
||||
print("✅ generationConfig to config mapping test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Run the tests
|
||||
pytest.main([__file__, "-v"])
|
||||
|
|
@ -467,3 +467,73 @@ async def test_e2e_generate_cold_storage_object_key_not_configured():
|
|||
assert result is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_logging_opentelemetry_context_propagation():
|
||||
"""
|
||||
Test that OpenTelemtry context propagation works with async completion.
|
||||
"""
|
||||
import asyncio
|
||||
import litellm
|
||||
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from opentelemetry import trace
|
||||
from opentelemetry.sdk.trace import TracerProvider
|
||||
from opentelemetry.sdk.trace.export import SimpleSpanProcessor
|
||||
from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanExporter
|
||||
|
||||
provider = TracerProvider()
|
||||
exporter = InMemorySpanExporter()
|
||||
provider.add_span_processor(SimpleSpanProcessor(exporter))
|
||||
trace.set_tracer_provider(provider)
|
||||
tracer = trace.get_tracer(__name__)
|
||||
|
||||
class MockOpenTelemetryLogger(CustomLogger):
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
span = tracer.start_span(start_time=start_time.timestamp() * 1e9, name="async_log_success_event")
|
||||
span.end(end_time=end_time)
|
||||
|
||||
|
||||
mock_logging_obj = MockOpenTelemetryLogger()
|
||||
|
||||
litellm.callbacks = [mock_logging_obj]
|
||||
|
||||
with tracer.start_as_current_span("span_1") as span:
|
||||
span_1_id = span.get_span_context().span_id
|
||||
await litellm.acompletion(
|
||||
max_tokens=100,
|
||||
messages=[{"role": "user", "content": "Hey"}],
|
||||
model="openai/codex-mini-latest",
|
||||
mock_response="Hello, world!",
|
||||
)
|
||||
|
||||
|
||||
with tracer.start_as_current_span("span_2") as span:
|
||||
span_2_id = span.get_span_context().span_id
|
||||
await litellm.acompletion(
|
||||
max_tokens=100,
|
||||
messages=[{"role": "user", "content": "Hey"}],
|
||||
model="openai/codex-mini-latest",
|
||||
mock_response="Hello, world!",
|
||||
)
|
||||
|
||||
await asyncio.sleep(1)
|
||||
spans = exporter.get_finished_spans()
|
||||
assert len(spans) == 4
|
||||
assert span_1_id != span_2_id
|
||||
sorted_spans = sorted(list(spans), key=lambda x: x.start_time or 0)
|
||||
|
||||
assert sorted_spans[0].name == "span_1"
|
||||
assert sorted_spans[1].name == "async_log_success_event"
|
||||
assert sorted_spans[2].name == "span_2"
|
||||
assert sorted_spans[3].name == "async_log_success_event"
|
||||
|
||||
first_span_context = sorted_spans[0].get_span_context()
|
||||
assert first_span_context is not None and first_span_context.span_id == span_1_id
|
||||
second_span_context = sorted_spans[2].get_span_context()
|
||||
assert second_span_context is not None and second_span_context.span_id == span_2_id
|
||||
first_completion_span_parent = sorted_spans[1].parent
|
||||
assert first_completion_span_parent is not None and first_completion_span_parent.span_id == span_1_id
|
||||
|
||||
# This check would fail without the proper context propagation, and span[3] would end up with span_1_id as the parent
|
||||
second_completion_span_parent = sorted_spans[3].parent
|
||||
assert second_completion_span_parent is not None and second_completion_span_parent.span_id == span_2_id
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@
|
|||
Tests for the LoggingWorker class to ensure graceful shutdown handling.
|
||||
"""
|
||||
import asyncio
|
||||
import contextvars
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
|
|
@ -139,3 +140,81 @@ class TestLoggingWorker:
|
|||
# Should have logged queue full exceptions
|
||||
exception_calls = [call for call in mock_logger.exception.call_args_list if "queue is full" in str(call)]
|
||||
assert len(exception_calls) > 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_context_propagation(self, logging_worker):
|
||||
"""Test that enqueued tasks execute in their original context."""
|
||||
# Create a context variable for testing
|
||||
test_context_var: contextvars.ContextVar[str] = contextvars.ContextVar('test_context_var')
|
||||
|
||||
# Track results from multiple tasks
|
||||
task_results = []
|
||||
|
||||
async def test_task(task_id: str):
|
||||
"""A test coroutine that checks if it can access the context variable."""
|
||||
# Sleep a bit to simulate real work and ensure context persists
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
try:
|
||||
# Try to get the context variable value
|
||||
value = test_context_var.get()
|
||||
task_results.append({
|
||||
'task_id': task_id,
|
||||
'context_value': value,
|
||||
'context_accessible': True
|
||||
})
|
||||
except LookupError:
|
||||
# Context variable not found
|
||||
task_results.append({
|
||||
'task_id': task_id,
|
||||
'context_accessible': False,
|
||||
'context_value': None
|
||||
})
|
||||
|
||||
# Start the logging worker
|
||||
logging_worker.start()
|
||||
|
||||
# Create two separate contexts and enqueue tasks from each
|
||||
|
||||
# Context 1: Set context var to "context_1"
|
||||
ctx1 = contextvars.copy_context()
|
||||
ctx1.run(test_context_var.set, "context_1")
|
||||
ctx1.run(logging_worker.enqueue, test_task("task_1"))
|
||||
|
||||
# Context 2: Set context var to "context_2"
|
||||
ctx2 = contextvars.copy_context()
|
||||
ctx2.run(test_context_var.set, "context_2")
|
||||
ctx2.run(logging_worker.enqueue, test_task("task_2"))
|
||||
|
||||
# Context 3: No context variable set (should get LookupError)
|
||||
ctx3 = contextvars.copy_context()
|
||||
ctx3.run(logging_worker.enqueue, test_task("task_3"))
|
||||
|
||||
# Wait for all tasks to be processed
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
# Stop the worker
|
||||
await logging_worker.stop()
|
||||
|
||||
# Sort results by task_id for consistent testing
|
||||
task_results.sort(key=lambda x: x['task_id'])
|
||||
|
||||
# Verify that each task saw its own context
|
||||
assert len(task_results) == 3, f"Expected 3 results, got {len(task_results)}"
|
||||
|
||||
# Task 1 should see "context_1"
|
||||
task1_result = next((r for r in task_results if r['task_id'] == 'task_1'), None)
|
||||
assert task1_result is not None, "Task 1 result not found"
|
||||
assert task1_result['context_accessible'] is True, "Task 1 should have access to context variable"
|
||||
assert task1_result['context_value'] == "context_1", f"Task 1 should see 'context_1', got: {task1_result['context_value']}"
|
||||
|
||||
# Task 2 should see "context_2"
|
||||
task2_result = next((r for r in task_results if r['task_id'] == 'task_2'), None)
|
||||
assert task2_result is not None, "Task 2 result not found"
|
||||
assert task2_result['context_accessible'] is True, "Task 2 should have access to context variable"
|
||||
assert task2_result['context_value'] == "context_2", f"Task 2 should see 'context_2', got: {task2_result['context_value']}"
|
||||
|
||||
# Task 3 should not have access to the context variable
|
||||
task3_result = next((r for r in task_results if r['task_id'] == 'task_3'), None)
|
||||
assert task3_result is not None, "Task 3 result not found"
|
||||
assert task3_result['context_accessible'] is False, "Task 3 should not have access to context variable"
|
||||
|
|
|
|||
|
|
@ -0,0 +1,343 @@
|
|||
"""
|
||||
Test for AnthropicStreamWrapper handling content blocks that exist after message_delta with stop_reason and usage.
|
||||
|
||||
This tests the scenario where a streaming response includes:
|
||||
1. Initial content blocks
|
||||
2. A message_delta chunk with stop_reason and usage
|
||||
3. Additional content blocks after the stop_reason
|
||||
|
||||
The wrapper should properly handle this by:
|
||||
- Holding the stop_reason chunk until usage is available
|
||||
- Merging usage into the stop_reason chunk
|
||||
- Properly managing content_block_stop/start events for subsequent content
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
from typing import List
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../../../.."))
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
|
||||
AnthropicStreamWrapper,
|
||||
)
|
||||
from litellm.types.utils import Delta, ModelResponse, StreamingChoices, Usage
|
||||
|
||||
|
||||
class MockCompletionStreamWithContentAfterStopReason:
|
||||
"""Mock stream that simulates content blocks existing after message_delta with stop_reason and usage."""
|
||||
|
||||
def __init__(self):
|
||||
self.responses = [
|
||||
# Initial text content
|
||||
ModelResponse(
|
||||
stream=True,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
delta=Delta(content="Hello"), index=0, finish_reason=None
|
||||
)
|
||||
],
|
||||
),
|
||||
ModelResponse(
|
||||
stream=True,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
delta=Delta(content=" world"), index=0, finish_reason=None
|
||||
)
|
||||
],
|
||||
),
|
||||
# Message delta with stop_reason AND usage (this is how it actually comes from the API)
|
||||
ModelResponse(
|
||||
stream=True,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
delta=Delta(content=""), index=0, finish_reason="stop"
|
||||
)
|
||||
],
|
||||
usage=Usage(prompt_tokens=230, completion_tokens=65, total_tokens=295),
|
||||
),
|
||||
# Additional content after the stop_reason - this simulates the scenario
|
||||
# where there might be additional content blocks after the main response
|
||||
ModelResponse(
|
||||
stream=True,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
delta=Delta(content=" Additional content"),
|
||||
index=0,
|
||||
finish_reason=None,
|
||||
)
|
||||
],
|
||||
),
|
||||
]
|
||||
self.index = 0
|
||||
|
||||
def __iter__(self):
|
||||
return self
|
||||
|
||||
def __next__(self):
|
||||
if self.index >= len(self.responses):
|
||||
raise StopIteration
|
||||
response = self.responses[self.index]
|
||||
self.index += 1
|
||||
return response
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
||||
async def __anext__(self):
|
||||
if self.index >= len(self.responses):
|
||||
raise StopAsyncIteration
|
||||
response = self.responses[self.index]
|
||||
self.index += 1
|
||||
return response
|
||||
|
||||
|
||||
def test_anthropic_stream_wrapper_content_after_stop_reason():
|
||||
"""Test that AnthropicStreamWrapper properly handles content blocks after message_delta with stop_reason."""
|
||||
|
||||
wrapper = AnthropicStreamWrapper(
|
||||
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
|
||||
model="claude-3",
|
||||
)
|
||||
|
||||
chunks = []
|
||||
chunk_types = []
|
||||
|
||||
# Collect all chunks
|
||||
for chunk in wrapper:
|
||||
chunks.append(chunk)
|
||||
chunk_types.append(chunk.get("type"))
|
||||
|
||||
# Verify the expected sequence of chunk types
|
||||
expected_types = [
|
||||
"message_start", # Initial message start
|
||||
"content_block_start", # Start of first content block
|
||||
"content_block_delta", # "Hello"
|
||||
"content_block_delta", # " world"
|
||||
"content_block_stop", # End of first content block due to stop_reason
|
||||
"message_delta", # Stop reason with merged usage
|
||||
"message_stop", # Final message stop
|
||||
]
|
||||
|
||||
print(f"Actual chunk types: {chunk_types}")
|
||||
print(f"Expected chunk types: {expected_types}")
|
||||
|
||||
# Verify we have the expected number of chunks
|
||||
assert len(chunk_types) >= len(
|
||||
expected_types
|
||||
), f"Expected at least {len(expected_types)} chunks, got {len(chunk_types)}"
|
||||
|
||||
# Verify key chunk types are present
|
||||
assert "message_start" in chunk_types
|
||||
assert "content_block_start" in chunk_types
|
||||
assert "content_block_delta" in chunk_types
|
||||
assert "content_block_stop" in chunk_types
|
||||
assert "message_delta" in chunk_types
|
||||
assert "message_stop" in chunk_types
|
||||
|
||||
# Find the message_delta chunk with stop_reason
|
||||
message_delta_chunk = None
|
||||
for chunk in chunks:
|
||||
if chunk.get("type") == "message_delta":
|
||||
message_delta_chunk = chunk
|
||||
break
|
||||
|
||||
assert message_delta_chunk is not None, "message_delta chunk not found"
|
||||
|
||||
# Verify that the message_delta chunk has both stop_reason and usage
|
||||
delta = message_delta_chunk.get("delta", {})
|
||||
usage = message_delta_chunk.get("usage", {})
|
||||
|
||||
assert (
|
||||
delta.get("stop_reason") == "end_turn"
|
||||
), f"Expected stop_reason 'end_turn', got {delta.get('stop_reason')}"
|
||||
assert (
|
||||
usage.get("input_tokens") == 230
|
||||
), f"Expected input_tokens 230, got {usage.get('input_tokens')}"
|
||||
assert (
|
||||
usage.get("output_tokens") == 65
|
||||
), f"Expected output_tokens 65, got {usage.get('output_tokens')}"
|
||||
|
||||
# Verify content_block_stop comes before message_delta
|
||||
content_block_stop_index = None
|
||||
message_delta_index = None
|
||||
|
||||
for i, chunk_type in enumerate(chunk_types):
|
||||
if chunk_type == "content_block_stop" and content_block_stop_index is None:
|
||||
content_block_stop_index = i
|
||||
elif chunk_type == "message_delta":
|
||||
message_delta_index = i
|
||||
|
||||
assert content_block_stop_index is not None, "content_block_stop not found"
|
||||
assert message_delta_index is not None, "message_delta not found"
|
||||
assert (
|
||||
content_block_stop_index < message_delta_index
|
||||
), "content_block_stop should come before message_delta"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_anthropic_stream_wrapper_content_after_stop_reason():
|
||||
"""Test async version of AnthropicStreamWrapper handling content blocks after message_delta with stop_reason."""
|
||||
|
||||
wrapper = AnthropicStreamWrapper(
|
||||
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
|
||||
model="claude-3",
|
||||
)
|
||||
|
||||
chunks = []
|
||||
chunk_types = []
|
||||
|
||||
# Collect all chunks asynchronously
|
||||
async for chunk in wrapper:
|
||||
chunks.append(chunk)
|
||||
chunk_types.append(chunk.get("type"))
|
||||
|
||||
print(f"Async - Actual chunk types: {chunk_types}")
|
||||
|
||||
# Verify key chunk types are present
|
||||
assert "message_start" in chunk_types
|
||||
assert "content_block_start" in chunk_types
|
||||
assert "content_block_delta" in chunk_types
|
||||
assert "content_block_stop" in chunk_types
|
||||
assert "message_delta" in chunk_types
|
||||
assert "message_stop" in chunk_types
|
||||
|
||||
# Find the message_delta chunk with stop_reason
|
||||
message_delta_chunk = None
|
||||
for chunk in chunks:
|
||||
if chunk.get("type") == "message_delta":
|
||||
message_delta_chunk = chunk
|
||||
break
|
||||
|
||||
assert message_delta_chunk is not None, "message_delta chunk not found"
|
||||
|
||||
# Verify that the message_delta chunk has both stop_reason and usage
|
||||
delta = message_delta_chunk.get("delta", {})
|
||||
usage = message_delta_chunk.get("usage", {})
|
||||
|
||||
assert (
|
||||
delta.get("stop_reason") == "end_turn"
|
||||
), f"Expected stop_reason 'end_turn', got {delta.get('stop_reason')}"
|
||||
assert (
|
||||
usage.get("input_tokens") == 230
|
||||
), f"Expected input_tokens 230, got {usage.get('input_tokens')}"
|
||||
assert (
|
||||
usage.get("output_tokens") == 65
|
||||
), f"Expected output_tokens 65, got {usage.get('output_tokens')}"
|
||||
|
||||
|
||||
def test_usage_merging_behavior():
|
||||
"""Test that usage information is properly merged with stop_reason chunk."""
|
||||
|
||||
wrapper = AnthropicStreamWrapper(
|
||||
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
|
||||
model="claude-3",
|
||||
)
|
||||
|
||||
# Process chunks and look specifically for the usage merging behavior
|
||||
chunks = []
|
||||
for chunk in wrapper:
|
||||
chunks.append(chunk)
|
||||
# If this is a message_delta with stop_reason, verify it has usage
|
||||
if (
|
||||
chunk.get("type") == "message_delta"
|
||||
and chunk.get("delta", {}).get("stop_reason") is not None
|
||||
):
|
||||
|
||||
usage = chunk.get("usage", {})
|
||||
assert (
|
||||
usage.get("input_tokens") is not None
|
||||
), "Usage should be merged with stop_reason chunk"
|
||||
assert (
|
||||
usage.get("output_tokens") is not None
|
||||
), "Usage should be merged with stop_reason chunk"
|
||||
break
|
||||
|
||||
|
||||
def test_sse_wrapper_with_content_after_stop_reason():
|
||||
"""Test SSE wrapper formatting for the content after stop_reason scenario."""
|
||||
|
||||
wrapper = AnthropicStreamWrapper(
|
||||
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
|
||||
model="claude-3",
|
||||
)
|
||||
|
||||
# Get SSE formatted chunks
|
||||
sse_chunks = []
|
||||
for chunk in wrapper.anthropic_sse_wrapper():
|
||||
sse_chunks.append(chunk)
|
||||
if len(sse_chunks) >= 10: # Limit to avoid infinite loops in tests
|
||||
break
|
||||
|
||||
# Verify all chunks are properly formatted as bytes
|
||||
for chunk in sse_chunks:
|
||||
assert isinstance(chunk, bytes), "SSE chunks should be bytes"
|
||||
|
||||
# Decode and verify SSE format
|
||||
chunk_str = chunk.decode("utf-8")
|
||||
lines = chunk_str.split("\n")
|
||||
|
||||
# Should have event and data lines
|
||||
assert any(
|
||||
line.startswith("event: ") for line in lines
|
||||
), f"Missing event line in: {chunk_str}"
|
||||
assert any(
|
||||
line.startswith("data: ") for line in lines
|
||||
), f"Missing data line in: {chunk_str}"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_sse_wrapper_with_content_after_stop_reason():
|
||||
"""Test async SSE wrapper formatting for the content after stop_reason scenario."""
|
||||
|
||||
wrapper = AnthropicStreamWrapper(
|
||||
completion_stream=MockCompletionStreamWithContentAfterStopReason(),
|
||||
model="claude-3",
|
||||
)
|
||||
|
||||
# Get SSE formatted chunks asynchronously
|
||||
sse_chunks = []
|
||||
async for chunk in wrapper.async_anthropic_sse_wrapper():
|
||||
sse_chunks.append(chunk)
|
||||
if len(sse_chunks) >= 10: # Limit to avoid infinite loops in tests
|
||||
break
|
||||
|
||||
# Verify all chunks are properly formatted as bytes
|
||||
for chunk in sse_chunks:
|
||||
assert isinstance(chunk, bytes), "Async SSE chunks should be bytes"
|
||||
|
||||
# Decode and verify SSE format
|
||||
chunk_str = chunk.decode("utf-8")
|
||||
lines = chunk_str.split("\n")
|
||||
|
||||
# Should have event and data lines
|
||||
assert any(
|
||||
line.startswith("event: ") for line in lines
|
||||
), f"Missing event line in: {chunk_str}"
|
||||
assert any(
|
||||
line.startswith("data: ") for line in lines
|
||||
), f"Missing data line in: {chunk_str}"
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Run a quick test
|
||||
test_anthropic_stream_wrapper_content_after_stop_reason()
|
||||
print("✅ Sync test passed")
|
||||
|
||||
import asyncio
|
||||
|
||||
asyncio.run(test_async_anthropic_stream_wrapper_content_after_stop_reason())
|
||||
print("✅ Async test passed")
|
||||
|
||||
test_usage_merging_behavior()
|
||||
print("✅ Usage merging test passed")
|
||||
|
||||
test_sse_wrapper_with_content_after_stop_reason()
|
||||
print("✅ SSE wrapper test passed")
|
||||
|
||||
asyncio.run(test_async_sse_wrapper_with_content_after_stop_reason())
|
||||
print("✅ Async SSE wrapper test passed")
|
||||
|
||||
print("🎉 All tests passed!")
|
||||
|
|
@ -0,0 +1,2 @@
|
|||
{"recordId": "request-1", "modelInput": {"messages": [{"role": "user", "content": [{"type": "text", "text": "Hello world!"}]}], "max_tokens": 10, "system": [{"type": "text", "text": "You are a helpful assistant."}], "anthropic_version": "bedrock-2023-05-31"}}
|
||||
{"recordId": "request-2", "modelInput": {"messages": [{"role": "user", "content": [{"type": "text", "text": "Hello world!"}]}], "max_tokens": 10, "system": [{"type": "text", "text": "You are an unhelpful assistant."}], "anthropic_version": "bedrock-2023-05-31"}}
|
||||
|
|
@ -0,0 +1,2 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
|
|
@ -0,0 +1,90 @@
|
|||
"""
|
||||
Test bedrock files transformation functionality
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.bedrock.files.transformation import BedrockJsonlFilesTransformation
|
||||
|
||||
|
||||
class TestBedrockFilesTransformation:
|
||||
"""Test bedrock files transformation"""
|
||||
|
||||
def test_transform_openai_jsonl_content_to_bedrock_jsonl_content(self):
|
||||
"""
|
||||
Test transformation of OpenAI JSONL format to Bedrock batch format.
|
||||
|
||||
Validates that the transformation correctly converts OpenAI batch completion
|
||||
format to Bedrock's expected batch format with proper recordId and modelInput structure.
|
||||
"""
|
||||
# Initialize the transformation class
|
||||
transformation = BedrockJsonlFilesTransformation()
|
||||
|
||||
# Load input JSONL file
|
||||
input_file_path = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"input_batch_completions.jsonl"
|
||||
)
|
||||
|
||||
# Read and parse the JSONL content
|
||||
openai_jsonl_content = []
|
||||
with open(input_file_path, 'r') as f:
|
||||
for line in f:
|
||||
if line.strip():
|
||||
openai_jsonl_content.append(json.loads(line))
|
||||
|
||||
# Transform the content
|
||||
bedrock_jsonl_content = transformation._transform_openai_jsonl_content_to_bedrock_jsonl_content(
|
||||
openai_jsonl_content=openai_jsonl_content
|
||||
)
|
||||
|
||||
# Print the transformation results for validation
|
||||
print("\n=== INPUT (OpenAI format) ===")
|
||||
for i, content in enumerate(openai_jsonl_content):
|
||||
print(f"Record {i+1}:")
|
||||
print(json.dumps(content, indent=2))
|
||||
print()
|
||||
|
||||
print("\n=== OUTPUT (Bedrock format) ===")
|
||||
for i, content in enumerate(bedrock_jsonl_content):
|
||||
print(f"Record {i+1}:")
|
||||
print(json.dumps(content, indent=2))
|
||||
print()
|
||||
|
||||
# Basic validation
|
||||
assert len(bedrock_jsonl_content) == len(openai_jsonl_content), "Should have same number of records"
|
||||
|
||||
# Check structure of transformed records
|
||||
for i, record in enumerate(bedrock_jsonl_content):
|
||||
assert "recordId" in record, f"Record {i+1} should have recordId"
|
||||
assert "modelInput" in record, f"Record {i+1} should have modelInput"
|
||||
|
||||
# Check recordId matches custom_id from input
|
||||
expected_custom_id = openai_jsonl_content[i].get("custom_id")
|
||||
assert record["recordId"] == expected_custom_id, f"Record {i+1} recordId should match custom_id"
|
||||
|
||||
# Check modelInput has expected structure
|
||||
model_input = record["modelInput"]
|
||||
assert isinstance(model_input, dict), f"Record {i+1} modelInput should be a dictionary"
|
||||
|
||||
# For Anthropic models, should have anthropic_version and messages
|
||||
if "anthropic.claude" in openai_jsonl_content[i]["body"]["model"]:
|
||||
assert "anthropic_version" in model_input, f"Record {i+1} should have anthropic_version"
|
||||
assert "messages" in model_input, f"Record {i+1} should have messages"
|
||||
assert "max_tokens" in model_input, f"Record {i+1} should have max_tokens"
|
||||
|
||||
# Write expected output to file for reference
|
||||
expected_output_path = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"expected_bedrock_batch_completions.jsonl"
|
||||
)
|
||||
|
||||
with open(expected_output_path, 'w') as f:
|
||||
for record in bedrock_jsonl_content:
|
||||
f.write(json.dumps(record) + '\n')
|
||||
|
||||
print(f"\n=== Expected output written to: {expected_output_path} ===")
|
||||
|
||||
|
|
@ -0,0 +1,183 @@
|
|||
"""
|
||||
Test suite for Dashscope cost calculation functionality.
|
||||
|
||||
Tests the cost calculation for Dashscope models including:
|
||||
- Tiered pricing based on input token ranges
|
||||
- Caching discounts
|
||||
- Reasoning tokens
|
||||
- Standard flat pricing fallback
|
||||
"""
|
||||
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
# Add the project root to Python path
|
||||
sys.path.insert(0, os.path.abspath("../../../.."))
|
||||
|
||||
import litellm
|
||||
from litellm.llms.dashscope.cost_calculator import (
|
||||
cost_per_token as dashscope_cost_per_token,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
CompletionTokensDetailsWrapper,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
class TestDashscopeCostCalculator:
|
||||
"""Test suite for Dashscope cost calculation functionality."""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def setup_model_cost_map(self):
|
||||
"""Set up the model cost map for testing."""
|
||||
# Ensure we use local model cost map for consistent testing
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
|
||||
# Find the project root directory and load model cost map
|
||||
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
project_root = current_dir
|
||||
while not os.path.exists(os.path.join(project_root, "model_prices_and_context_window.json")):
|
||||
parent = os.path.dirname(project_root)
|
||||
if parent == project_root: # Reached filesystem root
|
||||
break
|
||||
project_root = parent
|
||||
|
||||
model_cost_path = os.path.join(project_root, "model_prices_and_context_window.json")
|
||||
with open(model_cost_path, "r") as f:
|
||||
model_cost_map = json.load(f)
|
||||
litellm.model_cost = model_cost_map
|
||||
|
||||
def test_flat_pricing_basic_cost_calculation(self):
|
||||
"""Test basic cost calculation for flat pricing models (qwen-max)."""
|
||||
usage = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=500,
|
||||
total_tokens=1500
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen-max",
|
||||
usage=usage
|
||||
)
|
||||
|
||||
# Expected costs for qwen-max:
|
||||
# Input: 1000 tokens * $1.6e-6 = $0.0016
|
||||
# Output: 500 tokens * $6.4e-6 = $0.0032
|
||||
expected_prompt_cost = 1000 * 1.6e-6
|
||||
expected_completion_cost = 500 * 6.4e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_single_tier(self):
|
||||
"""Test tiered pricing when all tokens fall within first tier."""
|
||||
usage = Usage(
|
||||
prompt_tokens=20000, # Within first tier (0-32K)
|
||||
completion_tokens=1000,
|
||||
total_tokens=21000
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen3-coder-plus",
|
||||
usage=usage
|
||||
)
|
||||
|
||||
# Expected costs for qwen3-coder-plus (tier 1):
|
||||
# Input: 20,000 tokens * $1e-6 = $0.02
|
||||
# Output: 1,000 tokens * $5e-6 = $0.005
|
||||
expected_prompt_cost = 20000 * 1e-6
|
||||
expected_completion_cost = 1000 * 5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_higher_tier(self):
|
||||
"""Test tiered pricing when tokens fall in higher tier (tier 3)."""
|
||||
usage = Usage(
|
||||
prompt_tokens=150000, # Falls in tier 3 (128K-256K)
|
||||
completion_tokens=2000,
|
||||
total_tokens=152000
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen3-coder-plus",
|
||||
usage=usage
|
||||
)
|
||||
|
||||
# Expected input cost calculation:
|
||||
# 150,000 tokens falls in tier 3 (128K-256K), so all tokens are charged at tier 3 rate
|
||||
# Input: 150,000 tokens * $3e-6 = $0.45
|
||||
# Output: 2,000 tokens falls in tier 1 (0-32K), so charged at tier 1 rate
|
||||
# Output: 2,000 tokens * $5e-6 = $0.01
|
||||
|
||||
expected_prompt_cost = 150000 * 3e-6 # All tokens at tier 3 rate
|
||||
expected_completion_cost = 2000 * 5e-6 # All tokens at tier 1 rate
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_with_caching(self):
|
||||
"""Test tiered pricing with cached tokens."""
|
||||
prompt_tokens_details = PromptTokensDetailsWrapper(
|
||||
cached_tokens=10000 # 10K cached tokens
|
||||
)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=50000, # 40K regular + 10K cached = 50K total
|
||||
completion_tokens=1000,
|
||||
total_tokens=51000,
|
||||
prompt_tokens_details=prompt_tokens_details
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen3-coder-plus",
|
||||
usage=usage
|
||||
)
|
||||
|
||||
# Expected cost calculation:
|
||||
# Regular tokens: 40,000 falls in tier 2 (32K-128K), so all charged at tier 2 rate
|
||||
# - Regular: 40,000 * $1.8e-6 = $0.072
|
||||
# Cached tokens: 10,000 falls in tier 1 (0-32K), so charged at tier 1 cached rate
|
||||
# - Cached: 10,000 * $1e-7 = $0.001
|
||||
# Total input cost = $0.072 + $0.001 = $0.073
|
||||
|
||||
regular_tokens = 40000
|
||||
cached_tokens = 10000
|
||||
|
||||
expected_regular_cost = regular_tokens * 1.8e-6 # Tier 2 rate
|
||||
expected_cached_cost = cached_tokens * 1e-7 # Tier 1 cached rate
|
||||
expected_prompt_cost = expected_regular_cost + expected_cached_cost
|
||||
expected_completion_cost = 1000 * 5e-6 # Tier 1 rate
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_highest_tier(self):
|
||||
"""Test tiered pricing when tokens exceed highest tier range."""
|
||||
usage = Usage(
|
||||
prompt_tokens=2000000, # Exceeds tier 4 max (1M), should use tier 4 rate
|
||||
completion_tokens=5000,
|
||||
total_tokens=2005000
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen3-coder-plus",
|
||||
usage=usage
|
||||
)
|
||||
|
||||
# Expected cost calculation:
|
||||
# 2,000,000 tokens exceeds tier 4 (256K-1M), so use tier 4 rate for all tokens
|
||||
# Input: 2,000,000 tokens * $6e-6 = $12.0
|
||||
# Output: 5,000 tokens falls in tier 1 (0-32K), so charged at tier 1 rate
|
||||
# Output: 5,000 tokens * $5e-6 = $0.025
|
||||
|
||||
expected_prompt_cost = 2000000 * 6e-6 # Tier 4 rate (highest tier)
|
||||
expected_completion_cost = 5000 * 5e-6 # Tier 1 rate
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
|
@ -158,3 +158,81 @@ class TestGoogleAIStudioTokenCounter:
|
|||
model=model_to_use,
|
||||
contents=contents
|
||||
)
|
||||
|
||||
def test_clean_contents_for_gemini_api_removes_id_field(self):
|
||||
"""Test that _clean_contents_for_gemini_api removes unsupported 'id' field from function responses"""
|
||||
from litellm.llms.gemini.count_tokens.handler import GoogleAIStudioTokenCounter
|
||||
|
||||
token_counter = GoogleAIStudioTokenCounter()
|
||||
|
||||
# Test contents with function response containing 'id' field (camelCase)
|
||||
contents_with_id = [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "Hello world"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"functionResponse": {
|
||||
"id": "read_many_files-1757526647518-730a691aac11c", # This should be removed
|
||||
"name": "read_many_files",
|
||||
"response": {
|
||||
"output": "No files matching the criteria were found or all were skipped."
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
]
|
||||
|
||||
# Clean the contents
|
||||
cleaned_contents = token_counter._clean_contents_for_gemini_api(contents_with_id)
|
||||
|
||||
# Verify the 'id' field was removed
|
||||
function_response = cleaned_contents[1]["parts"][0]["functionResponse"]
|
||||
assert "id" not in function_response
|
||||
assert "name" in function_response
|
||||
assert "response" in function_response
|
||||
assert function_response["name"] == "read_many_files"
|
||||
assert function_response["response"]["output"] == "No files matching the criteria were found or all were skipped."
|
||||
|
||||
|
||||
def test_clean_contents_for_gemini_api_preserves_other_fields(self):
|
||||
"""Test that _clean_contents_for_gemini_api preserves other fields and structure"""
|
||||
from litellm.llms.gemini.count_tokens.handler import GoogleAIStudioTokenCounter
|
||||
|
||||
token_counter = GoogleAIStudioTokenCounter()
|
||||
|
||||
# Test contents without function responses
|
||||
contents_without_function_response = [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "This is a regular message"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "This is a model response"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
}
|
||||
]
|
||||
|
||||
# Clean the contents
|
||||
cleaned_contents = token_counter._clean_contents_for_gemini_api(contents_without_function_response)
|
||||
|
||||
# Verify the contents are unchanged
|
||||
assert cleaned_contents == contents_without_function_response
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -183,7 +183,9 @@ async def test_budget_reset_and_expires_at_first_of_month(monkeypatch):
|
|||
assert (
|
||||
response_date.month == expected_month
|
||||
), f"Expected month {expected_month}, got {response_date.month} for {key}"
|
||||
assert response_date.day == 1, f"Expected day 1, got {response_date.day} for {key}"
|
||||
assert (
|
||||
response_date.day == 1
|
||||
), f"Expected day 1, got {response_date.day} for {key}"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -507,7 +509,6 @@ def test_get_new_token_with_invalid_key():
|
|||
assert "New key must start with 'sk-'" in str(exc_info.value.detail)
|
||||
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_service_account_requires_team_id():
|
||||
with pytest.raises(HTTPException):
|
||||
|
|
@ -529,11 +530,12 @@ async def test_generate_service_account_works_with_team_id():
|
|||
from unittest.mock import patch
|
||||
|
||||
# Mock the database and router dependencies from proxy_server
|
||||
with patch('litellm.proxy.proxy_server.prisma_client') as mock_prisma, \
|
||||
patch('litellm.proxy.proxy_server.llm_router') as mock_router, \
|
||||
patch('litellm.proxy.proxy_server.premium_user', False), \
|
||||
patch('litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn') as mock_generate_key:
|
||||
|
||||
with patch("litellm.proxy.proxy_server.prisma_client") as mock_prisma, patch(
|
||||
"litellm.proxy.proxy_server.llm_router"
|
||||
) as mock_router, patch("litellm.proxy.proxy_server.premium_user", False), patch(
|
||||
"litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn"
|
||||
) as mock_generate_key:
|
||||
|
||||
# Configure mocks
|
||||
mock_prisma.return_value = AsyncMock()
|
||||
mock_router.return_value = None
|
||||
|
|
@ -542,9 +544,9 @@ async def test_generate_service_account_works_with_team_id():
|
|||
"key": "sk-test-key",
|
||||
"expires": None,
|
||||
"user_id": "test-user",
|
||||
"team_id": "IJ"
|
||||
"team_id": "IJ",
|
||||
}
|
||||
|
||||
|
||||
# This should not raise an exception since team_id is provided
|
||||
await _common_key_generation_helper(
|
||||
data=GenerateKeyRequest(
|
||||
|
|
@ -559,7 +561,6 @@ async def test_generate_service_account_works_with_team_id():
|
|||
)
|
||||
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_service_account_requires_team_id():
|
||||
data = UpdateKeyRequest(key="sk-1", metadata={"service_account_id": "sa"})
|
||||
|
|
@ -571,7 +572,9 @@ async def test_update_service_account_requires_team_id():
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_service_account_works_with_team_id():
|
||||
data = UpdateKeyRequest(key="sk-1", metadata={"service_account_id": "sa"}, team_id="IJ")
|
||||
data = UpdateKeyRequest(
|
||||
key="sk-1", metadata={"service_account_id": "sa"}, team_id="IJ"
|
||||
)
|
||||
existing_key = LiteLLM_VerificationToken(token="hashed")
|
||||
|
||||
await prepare_key_update_data(data=data, existing_key_row=existing_key)
|
||||
|
|
@ -580,22 +583,22 @@ async def test_update_service_account_works_with_team_id():
|
|||
@pytest.mark.asyncio
|
||||
async def test_validate_team_id_used_in_service_account_request_requires_team_id():
|
||||
"""
|
||||
Test that validate_team_id_used_in_service_account_request raises HTTPException
|
||||
Test that validate_team_id_used_in_service_account_request raises HTTPException
|
||||
when team_id is None for service account key generation.
|
||||
"""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
validate_team_id_used_in_service_account_request,
|
||||
)
|
||||
|
||||
|
||||
mock_prisma_client = AsyncMock()
|
||||
|
||||
|
||||
# Test that HTTPException is raised when team_id is None
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await validate_team_id_used_in_service_account_request(
|
||||
team_id=None,
|
||||
prisma_client=mock_prisma_client,
|
||||
)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "team_id is required for service account keys" in str(exc_info.value.detail)
|
||||
|
||||
|
|
@ -603,7 +606,7 @@ async def test_validate_team_id_used_in_service_account_request_requires_team_id
|
|||
@pytest.mark.asyncio
|
||||
async def test_validate_team_id_used_in_service_account_request_requires_prisma_client():
|
||||
"""
|
||||
Test that validate_team_id_used_in_service_account_request raises HTTPException
|
||||
Test that validate_team_id_used_in_service_account_request raises HTTPException
|
||||
when prisma_client is None for service account key generation.
|
||||
"""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
|
|
@ -616,78 +619,76 @@ async def test_validate_team_id_used_in_service_account_request_requires_prisma_
|
|||
team_id="test-team-id",
|
||||
prisma_client=None,
|
||||
)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "prisma_client is required for service account keys" in str(exc_info.value.detail)
|
||||
assert "prisma_client is required for service account keys" in str(
|
||||
exc_info.value.detail
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_validate_team_id_used_in_service_account_request_checks_team_exists():
|
||||
"""
|
||||
Test that validate_team_id_used_in_service_account_request validates that
|
||||
Test that validate_team_id_used_in_service_account_request validates that
|
||||
the team_id exists in the database for service account key generation.
|
||||
"""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
validate_team_id_used_in_service_account_request,
|
||||
)
|
||||
|
||||
|
||||
mock_prisma_client = AsyncMock()
|
||||
|
||||
|
||||
# Mock the database query to return None (team doesn't exist)
|
||||
mock_find_unique = AsyncMock(return_value=None)
|
||||
mock_prisma_client.db.litellm_teamtable.find_unique = mock_find_unique
|
||||
|
||||
|
||||
# Test that HTTPException is raised when team doesn't exist in DB
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await validate_team_id_used_in_service_account_request(
|
||||
team_id="non-existent-team-id",
|
||||
prisma_client=mock_prisma_client,
|
||||
)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "team_id does not exist in the database" in str(exc_info.value.detail)
|
||||
|
||||
|
||||
# Verify the database was queried with the correct parameters
|
||||
mock_find_unique.assert_called_once_with(
|
||||
where={"team_id": "non-existent-team-id"}
|
||||
)
|
||||
mock_find_unique.assert_called_once_with(where={"team_id": "non-existent-team-id"})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_validate_team_id_used_in_service_account_request_success():
|
||||
"""
|
||||
Test that validate_team_id_used_in_service_account_request returns True
|
||||
Test that validate_team_id_used_in_service_account_request returns True
|
||||
when team_id exists in the database for service account key generation.
|
||||
"""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
validate_team_id_used_in_service_account_request,
|
||||
)
|
||||
|
||||
|
||||
mock_prisma_client = AsyncMock()
|
||||
|
||||
|
||||
# Mock the database query to return a team object (team exists)
|
||||
mock_team = {"team_id": "existing-team-id", "team_name": "Test Team"}
|
||||
mock_find_unique = AsyncMock(return_value=mock_team)
|
||||
mock_prisma_client.db.litellm_teamtable.find_unique = mock_find_unique
|
||||
|
||||
|
||||
# Test that function returns True when team exists
|
||||
result = await validate_team_id_used_in_service_account_request(
|
||||
team_id="existing-team-id",
|
||||
prisma_client=mock_prisma_client,
|
||||
)
|
||||
|
||||
|
||||
assert result is True
|
||||
|
||||
|
||||
# Verify the database was queried with the correct parameters
|
||||
mock_find_unique.assert_called_once_with(
|
||||
where={"team_id": "existing-team-id"}
|
||||
)
|
||||
mock_find_unique.assert_called_once_with(where={"team_id": "existing-team-id"})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_service_account_key_endpoint_validation():
|
||||
"""
|
||||
Test that the /key/service-account/generate endpoint properly validates
|
||||
Test that the /key/service-account/generate endpoint properly validates
|
||||
team_id requirement and team existence in database.
|
||||
"""
|
||||
from unittest.mock import patch
|
||||
|
|
@ -705,16 +706,16 @@ async def test_generate_service_account_key_endpoint_validation():
|
|||
),
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "team_id is required for service account keys" in str(exc_info.value.detail)
|
||||
|
||||
# Test case 2: Team doesn't exist in database
|
||||
with patch('litellm.proxy.proxy_server.prisma_client') as mock_prisma:
|
||||
|
||||
# Test case 2: Team doesn't exist in database
|
||||
with patch("litellm.proxy.proxy_server.prisma_client") as mock_prisma:
|
||||
# Mock team not found
|
||||
mock_find_unique = AsyncMock(return_value=None)
|
||||
mock_prisma.db.litellm_teamtable.find_unique = mock_find_unique
|
||||
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await generate_service_account_key_fn(
|
||||
data=GenerateKeyRequest(team_id="non-existent-team"),
|
||||
|
|
@ -723,7 +724,165 @@ async def test_generate_service_account_key_endpoint_validation():
|
|||
),
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "team_id does not exist in the database" in str(exc_info.value.detail)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unblock_key_supports_both_sk_and_hashed_tokens(monkeypatch):
|
||||
"""
|
||||
Test that the unblock_key endpoint correctly handles both sk- prefixed tokens
|
||||
and hashed tokens by properly converting sk- tokens to hashed format before
|
||||
database operations.
|
||||
"""
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from litellm.proxy._types import BlockKeyRequest
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import unblock_key
|
||||
|
||||
# Mock dependencies
|
||||
mock_prisma_client = AsyncMock()
|
||||
mock_user_api_key_cache = MagicMock()
|
||||
mock_proxy_logging_obj = MagicMock()
|
||||
|
||||
# Use a proper 64-character hex hash for testing
|
||||
test_hashed_token = (
|
||||
"a1b2c3d4e5f6789012345678901234567890123456789012345678901234abcd"
|
||||
)
|
||||
|
||||
# Mock the key record that will be returned from database
|
||||
mock_key_record = MagicMock()
|
||||
mock_key_record.token = test_hashed_token
|
||||
mock_key_record.blocked = False
|
||||
mock_key_record.model_dump_json.return_value = (
|
||||
f'{{"token": "{test_hashed_token}", "blocked": false}}'
|
||||
)
|
||||
|
||||
# Mock database operations
|
||||
mock_prisma_client.db.litellm_verificationtoken.find_unique = AsyncMock(
|
||||
return_value=mock_key_record
|
||||
)
|
||||
mock_prisma_client.db.litellm_verificationtoken.update = AsyncMock(
|
||||
return_value=mock_key_record
|
||||
)
|
||||
|
||||
# Mock get_key_object and _cache_key_object functions
|
||||
mock_key_object = MagicMock()
|
||||
mock_key_object.blocked = True # Initially blocked
|
||||
|
||||
# Mock hash_token function
|
||||
def mock_hash_token(token):
|
||||
if token == "sk-test123456789":
|
||||
return test_hashed_token
|
||||
return token
|
||||
|
||||
# Apply monkeypatch
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.user_api_key_cache", mock_user_api_key_cache
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj", mock_proxy_logging_obj
|
||||
)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.hash_token", mock_hash_token)
|
||||
monkeypatch.setattr(
|
||||
"litellm.store_audit_logs", False
|
||||
) # Disable audit logs for simpler test
|
||||
|
||||
# Mock get_key_object and _cache_key_object
|
||||
async def mock_get_key_object(**kwargs):
|
||||
return mock_key_object
|
||||
|
||||
async def mock_cache_key_object(**kwargs):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.management_endpoints.key_management_endpoints.get_key_object",
|
||||
mock_get_key_object,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.management_endpoints.key_management_endpoints._cache_key_object",
|
||||
mock_cache_key_object,
|
||||
)
|
||||
|
||||
# Create mock request and user auth
|
||||
mock_request = MagicMock()
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-admin", user_id="admin_user"
|
||||
)
|
||||
|
||||
# Test Case 1: Using sk- prefixed token
|
||||
sk_token_request = BlockKeyRequest(key="sk-test123456789")
|
||||
|
||||
result = await unblock_key(
|
||||
data=sk_token_request,
|
||||
http_request=mock_request,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
# Verify that the database update was called with hashed token
|
||||
mock_prisma_client.db.litellm_verificationtoken.update.assert_called_with(
|
||||
where={"token": test_hashed_token}, data={"blocked": False}
|
||||
)
|
||||
|
||||
assert result == mock_key_record
|
||||
assert mock_key_object.blocked == False # Should be updated to unblocked
|
||||
|
||||
# Reset mocks for second test
|
||||
mock_prisma_client.db.litellm_verificationtoken.update.reset_mock()
|
||||
mock_key_object.blocked = True # Reset to blocked state
|
||||
|
||||
# Test Case 2: Using already hashed token
|
||||
hashed_token_request = BlockKeyRequest(key=test_hashed_token)
|
||||
|
||||
result = await unblock_key(
|
||||
data=hashed_token_request,
|
||||
http_request=mock_request,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
# Verify that the database update was called with the same hashed token
|
||||
mock_prisma_client.db.litellm_verificationtoken.update.assert_called_with(
|
||||
where={"token": test_hashed_token}, data={"blocked": False}
|
||||
)
|
||||
|
||||
assert result == mock_key_record
|
||||
assert mock_key_object.blocked == False # Should be updated to unblocked
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unblock_key_invalid_key_format(monkeypatch):
|
||||
"""
|
||||
Test that unblock_key properly validates key format and raises appropriate errors
|
||||
for invalid keys.
|
||||
"""
|
||||
from litellm.proxy._types import BlockKeyRequest
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import unblock_key
|
||||
from litellm.proxy.utils import ProxyException
|
||||
|
||||
# Mock prisma_client to avoid DB connection error
|
||||
mock_prisma_client = AsyncMock()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
|
||||
|
||||
# Mock request and user auth
|
||||
mock_request = MagicMock()
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-admin", user_id="admin_user"
|
||||
)
|
||||
|
||||
# Test with invalid key format
|
||||
invalid_key_request = BlockKeyRequest(key="invalid-key-format")
|
||||
|
||||
with pytest.raises(ProxyException) as exc_info:
|
||||
await unblock_key(
|
||||
data=invalid_key_request,
|
||||
http_request=mock_request,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
assert exc_info.value.code == "400"
|
||||
assert "Invalid key format" in str(exc_info.value.message)
|
||||
|
|
|
|||
|
|
@ -18,7 +18,9 @@ import litellm
|
|||
from litellm.proxy._types import SpendLogsPayload
|
||||
from litellm.proxy.hooks.proxy_track_cost_callback import _ProxyDBLogger
|
||||
from litellm.proxy.proxy_server import app, prisma_client
|
||||
from litellm.proxy.spend_tracking import spend_management_endpoints
|
||||
from litellm.router import Router
|
||||
from litellm.types.utils import BudgetConfig
|
||||
|
||||
ignored_keys = [
|
||||
"request_id",
|
||||
|
|
@ -32,6 +34,18 @@ ignored_keys = [
|
|||
"metadata.cold_storage_object_key",
|
||||
]
|
||||
|
||||
MODEL_LIST = [
|
||||
{
|
||||
"model_name": "azure-gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-4o-mini",
|
||||
"mock_response": "Hello, world!",
|
||||
"tags": ["default"],
|
||||
"base_model": "gpt-4o-mini",
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client():
|
||||
|
|
@ -43,6 +57,19 @@ def add_anthropic_api_key_to_env(monkeypatch):
|
|||
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-api03-1234567890")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def disable_budget_sync(monkeypatch):
|
||||
"""Disable periodic sync during tests"""
|
||||
|
||||
async def noop(*a, **k):
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.router_strategy.budget_limiter.RouterBudgetLimiting.periodic_sync_in_memory_spend_with_redis",
|
||||
noop,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ui_view_spend_logs_with_user_id(client, monkeypatch):
|
||||
# Mock data for the test
|
||||
|
|
@ -1318,3 +1345,70 @@ async def test_view_spend_tags_no_database(client, monkeypatch):
|
|||
# Check the actual error message structure
|
||||
assert "error" in data
|
||||
assert "Database not connected" in data["error"]["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_budget_under(disable_budget_sync):
|
||||
"""Test that router allows completion when under budget"""
|
||||
provider_budget_config = {
|
||||
"azure": BudgetConfig(max_budget=0.01, budget_duration="10d")
|
||||
}
|
||||
|
||||
router = Router(
|
||||
enable_pre_call_checks=True,
|
||||
provider_budget_config=provider_budget_config,
|
||||
model_list=MODEL_LIST,
|
||||
)
|
||||
|
||||
response = await router.acompletion(
|
||||
model="azure-gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
)
|
||||
|
||||
assert response is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_budget_over(disable_budget_sync):
|
||||
"""Test that router allows completion when over budget"""
|
||||
provider_budget_config = {
|
||||
"azure": BudgetConfig(max_budget=-0.01, budget_duration="10d")
|
||||
}
|
||||
|
||||
router = Router(
|
||||
num_retries=0,
|
||||
enable_pre_call_checks=True,
|
||||
provider_budget_config=provider_budget_config,
|
||||
model_list=MODEL_LIST,
|
||||
)
|
||||
|
||||
with pytest.raises(Exception) as e:
|
||||
response = await router.acompletion(
|
||||
model="azure-gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
)
|
||||
assert "Exceeded budget for provider" in str(e.value)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_budget_provider_budgets(disable_budget_sync):
|
||||
"""Test that provider_budgets() returns correct values"""
|
||||
provider = "azure"
|
||||
max_budget = -0.01
|
||||
budget_duration = "10d"
|
||||
provider_budget_config = {
|
||||
provider: BudgetConfig(max_budget=max_budget, budget_duration=budget_duration)
|
||||
}
|
||||
|
||||
router = Router(
|
||||
num_retries=0,
|
||||
enable_pre_call_checks=True,
|
||||
provider_budget_config=provider_budget_config,
|
||||
model_list=MODEL_LIST,
|
||||
)
|
||||
|
||||
with patch("litellm.proxy.proxy_server.llm_router", router):
|
||||
response = await spend_management_endpoints.provider_budgets()
|
||||
provider_budget_response = response.providers[provider]
|
||||
assert provider_budget_response.budget_limit == max_budget
|
||||
assert provider_budget_response.time_period == budget_duration
|
||||
|
|
|
|||
|
|
@ -1059,3 +1059,63 @@ async def test_add_litellm_metadata_from_request_headers():
|
|||
assert SPEND_LOGS_METADATA == dict(json.loads(headers["x-litellm-spend-logs-metadata"])), "spend_logs_metadata should be the same as the headers"
|
||||
|
||||
|
||||
|
||||
def test_get_internal_user_header_from_mapping_returns_expected_header():
|
||||
mappings = [
|
||||
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
|
||||
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
|
||||
]
|
||||
|
||||
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
|
||||
assert header_name == "X-OpenWebUI-User-Id"
|
||||
|
||||
|
||||
def test_get_internal_user_header_from_mapping_none_when_absent():
|
||||
mappings = [
|
||||
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"}
|
||||
]
|
||||
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(mappings)
|
||||
assert header_name is None
|
||||
|
||||
single = {"header_name": "X-Only-Customer", "litellm_user_role": "customer"}
|
||||
header_name = LiteLLMProxyRequestSetup.get_internal_user_header_from_mapping(single)
|
||||
assert header_name is None
|
||||
|
||||
|
||||
def test_add_internal_user_from_user_mapping_sets_user_id_when_header_present():
|
||||
user_api_key_dict = UserAPIKeyAuth(api_key="test-key")
|
||||
headers = {"X-OpenWebUI-User-Id": "internal-user-123"}
|
||||
general_settings = {
|
||||
"user_header_mappings": [
|
||||
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"},
|
||||
{"header_name": "X-OpenWebUI-User-Email", "litellm_user_role": "customer"},
|
||||
]
|
||||
}
|
||||
|
||||
result = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(
|
||||
general_settings, user_api_key_dict, headers
|
||||
)
|
||||
|
||||
assert result is user_api_key_dict
|
||||
assert user_api_key_dict.user_id == "internal-user-123"
|
||||
|
||||
|
||||
def test_add_internal_user_from_user_mapping_no_header_or_mapping_returns_unchanged():
|
||||
user_api_key_dict = UserAPIKeyAuth(api_key="test-key")
|
||||
|
||||
result = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(
|
||||
None, user_api_key_dict, {"X-OpenWebUI-User-Id": "abc"}
|
||||
)
|
||||
assert result is user_api_key_dict
|
||||
assert user_api_key_dict.user_id is None
|
||||
|
||||
general_settings = {
|
||||
"user_header_mappings": [
|
||||
{"header_name": "X-OpenWebUI-User-Id", "litellm_user_role": "internal_user"}
|
||||
]
|
||||
}
|
||||
result = LiteLLMProxyRequestSetup.add_internal_user_from_user_mapping(
|
||||
general_settings, user_api_key_dict, {"Other": "value"}
|
||||
)
|
||||
assert result is user_api_key_dict
|
||||
assert user_api_key_dict.user_id is None
|
||||
|
|
|
|||
|
|
@ -84,6 +84,15 @@ def test_get_optional_params_image_gen_vertex_ai_size():
|
|||
assert optional_params["sampleCount"] == 1
|
||||
|
||||
|
||||
def test_get_optional_params_image_gen_filters_empty_values():
|
||||
optional_params = get_optional_params_image_gen(
|
||||
model="gpt-image-1",
|
||||
custom_llm_provider="openai",
|
||||
extra_body={},
|
||||
)
|
||||
assert optional_params == {}
|
||||
|
||||
|
||||
def test_all_model_configs():
|
||||
from litellm.llms.vertex_ai.vertex_ai_partner_models.ai21.transformation import (
|
||||
VertexAIAi21Config,
|
||||
|
|
@ -643,6 +652,26 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
},
|
||||
},
|
||||
"supports_native_streaming": {"type": "boolean"},
|
||||
"tiered_pricing": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"range": {
|
||||
"type": "array",
|
||||
"items": {"type": "number"},
|
||||
"minItems": 2,
|
||||
"maxItems": 2
|
||||
},
|
||||
"input_cost_per_token": {"type": "number"},
|
||||
"output_cost_per_token": {"type": "number"},
|
||||
"cache_read_input_token_cost": {"type": "number"},
|
||||
"output_cost_per_reasoning_token": {"type": "number"}
|
||||
},
|
||||
"required": ["range"],
|
||||
"additionalProperties": False
|
||||
}
|
||||
},
|
||||
},
|
||||
"additionalProperties": False,
|
||||
},
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
BIN
ui/litellm-dashboard/out/assets/logos/qwen.png
Normal file
BIN
ui/litellm-dashboard/out/assets/logos/qwen.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 48 KiB |
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[75832,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","220","static/chunks/220-1c8d82f7ce7658c4.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-8dc8d9524a1f3965.js"],"default",1]
|
||||
3:I[30628,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","220","static/chunks/220-5061c4cea850d728.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","931","static/chunks/app/page-127adcf8da2b5294.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
|
||||
3:I[52829,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","418","static/chunks/app/model_hub/page-d6e5fb7de2cedde9.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,7 +1,7 @@
|
|||
2:I[19107,[],"ClientPageRoot"]
|
||||
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-3523e0e07cf314f6.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-714ca0ed10a07f66.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
|
||||
3:I[22775,["50","static/chunks/50-fe160ecfa8bc4059.js","521","static/chunks/521-d97d355792d44830.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-fff436ed72b19a24.js","162","static/chunks/162-4e7640b4d68e1ae4.js","172","static/chunks/172-0f7049c565983c4d.js","25","static/chunks/app/model_hub_table/page-e06e934de1021ee4.js"],"default",1]
|
||||
4:I[4707,[],""]
|
||||
5:I[36423,[],""]
|
||||
0:["0GF-OyXnYlAPMWfyPAZSs",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/060d5ddee53e45ce.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
0:["FMlZjJYLUentCU02Wj6R_",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/c528590c6415a94c.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
|
||||
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
|
||||
1:null
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue