mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Litellm stable release notes 05 03 2025 (#10536)
* build(release_cycle.md): document bar for minor vs. patch updates * docs(index.md): initial changelog doc * docs(index.md): update llama docs * docs(index.md): add docs for llm api endpoints + spend tracking/budget improvements * docs: more doc cleanup * docs(index.md): more doc cleanup * docs(index.md): final doc cleanup
This commit is contained in:
parent
a9ee95e0cf
commit
7ce687ef39
13 changed files with 1010 additions and 8 deletions
|
|
@ -421,3 +421,9 @@ async with stdio_client(server_params) as (read, write):
|
|||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Permission Management
|
||||
|
||||
Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs.
|
||||
|
||||
Join the discussion [here](https://github.com/BerriAI/litellm/discussions/9891)
|
||||
|
|
@ -1,4 +1,6 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Langsmith - Logging LLM Input/Output
|
||||
|
||||
|
|
@ -22,10 +24,13 @@ pip install litellm
|
|||
## Quick Start
|
||||
Use just 2 lines of code, to instantly log your responses **across all providers** with Langsmith
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="SDK">
|
||||
|
||||
```python
|
||||
litellm.success_callback = ["langsmith"]
|
||||
litellm.callbacks = ["langsmith"]
|
||||
```
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
|
@ -37,7 +42,7 @@ os.environ["LANGSMITH_DEFAULT_RUN_NAME"] = "" # defaults to LLMRun
|
|||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
# set langsmith as a callback, litellm will send the data to langsmith
|
||||
litellm.success_callback = ["langsmith"]
|
||||
litellm.callbacks = ["langsmith"]
|
||||
|
||||
# openai call
|
||||
response = litellm.completion(
|
||||
|
|
@ -47,8 +52,124 @@ response = litellm.completion(
|
|||
]
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["langsmith"]
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hey, how are you?"
|
||||
}
|
||||
],
|
||||
"max_completion_tokens": 250
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## Advanced
|
||||
|
||||
### Local Testing - Control Batch Size
|
||||
|
||||
Set the size of the batch that Langsmith will process at a time, default is 512.
|
||||
|
||||
Set `langsmith_batch_size=1` when testing locally, to see logs land quickly.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["LANGSMITH_API_KEY"] = ""
|
||||
# LLM API Keys
|
||||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
# set langsmith as a callback, litellm will send the data to langsmith
|
||||
litellm.callbacks = ["langsmith"]
|
||||
litellm.langsmith_batch_size = 1 # 👈 KEY CHANGE
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hi 👋 - i'm openai"}
|
||||
]
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
langsmith_batch_size: 1
|
||||
callbacks: ["langsmith"]
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hey, how are you?"
|
||||
}
|
||||
],
|
||||
"max_completion_tokens": 250
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
|
||||
### Set Langsmith fields
|
||||
|
||||
```python
|
||||
|
|
|
|||
|
|
@ -60,9 +60,9 @@ Here's how to call Bedrock with the LiteLLM Proxy Server
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-claude-v1
|
||||
- model_name: bedrock-claude-3-5-sonnet
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-instant-v1
|
||||
model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: os.environ/AWS_REGION_NAME
|
||||
|
|
|
|||
320
docs/my-website/docs/providers/openai/responses_api.md
Normal file
320
docs/my-website/docs/providers/openai/responses_api.md
Normal file
|
|
@ -0,0 +1,320 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# OpenAI - Response API
|
||||
|
||||
## Usage
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers title="OpenAI Non-streaming Response"
|
||||
import litellm
|
||||
|
||||
# Non-streaming response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="OpenAI Streaming Response"
|
||||
import litellm
|
||||
|
||||
# Streaming response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
#### GET a Response
|
||||
```python showLineNumbers title="Get Response by ID"
|
||||
import litellm
|
||||
|
||||
# First, create a response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Retrieve the response by ID
|
||||
retrieved_response = litellm.get_responses(
|
||||
response_id=response_id
|
||||
)
|
||||
|
||||
print(retrieved_response)
|
||||
|
||||
# For async usage
|
||||
# retrieved_response = await litellm.aget_responses(response_id=response_id)
|
||||
```
|
||||
|
||||
#### DELETE a Response
|
||||
```python showLineNumbers title="Delete Response by ID"
|
||||
import litellm
|
||||
|
||||
# First, create a response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Delete the response by ID
|
||||
delete_response = litellm.delete_responses(
|
||||
response_id=response_id
|
||||
)
|
||||
|
||||
print(delete_response)
|
||||
|
||||
# For async usage
|
||||
# delete_response = await litellm.adelete_responses(response_id=response_id)
|
||||
```
|
||||
|
||||
|
||||
### LiteLLM Proxy with OpenAI SDK
|
||||
|
||||
1. Set up config.yaml
|
||||
|
||||
```yaml showLineNumbers title="OpenAI Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: openai/o1-pro
|
||||
litellm_params:
|
||||
model: openai/o1-pro
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Use OpenAI SDK with LiteLLM Proxy
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers title="OpenAI Proxy Non-streaming Response"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="OpenAI Proxy Streaming Response"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
#### GET a Response
|
||||
```python showLineNumbers title="Get Response by ID with OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# First, create a response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Retrieve the response by ID
|
||||
retrieved_response = client.responses.retrieve(response_id)
|
||||
|
||||
print(retrieved_response)
|
||||
```
|
||||
|
||||
#### DELETE a Response
|
||||
```python showLineNumbers title="Delete Response by ID with OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# First, create a response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Delete the response by ID
|
||||
delete_response = client.responses.delete(response_id)
|
||||
|
||||
print(delete_response)
|
||||
```
|
||||
|
||||
|
||||
## Supported Responses API Parameters
|
||||
|
||||
| Provider | Supported Parameters |
|
||||
|----------|---------------------|
|
||||
| `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) |
|
||||
|
||||
## Computer Use
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="LiteLLM Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Non-streaming response
|
||||
response = litellm.responses(
|
||||
model="computer-use-preview",
|
||||
tools=[{
|
||||
"type": "computer_use_preview",
|
||||
"display_width": 1024,
|
||||
"display_height": 768,
|
||||
"environment": "browser" # other possible values: "mac", "windows", "ubuntu"
|
||||
}],
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Check the latest OpenAI news on bing.com."
|
||||
}
|
||||
# Optional: include a screenshot of the initial state of the environment
|
||||
# {
|
||||
# type: "input_image",
|
||||
# image_url: f"data:image/png;base64,{screenshot_base64}"
|
||||
# }
|
||||
]
|
||||
}
|
||||
],
|
||||
reasoning={
|
||||
"summary": "concise",
|
||||
},
|
||||
truncation="auto"
|
||||
)
|
||||
|
||||
print(response.output)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Set up config.yaml
|
||||
|
||||
```yaml showLineNumbers title="OpenAI Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: openai/o1-pro
|
||||
litellm_params:
|
||||
model: openai/o1-pro
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="OpenAI Proxy Non-streaming Response"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="computer-use-preview",
|
||||
tools=[{
|
||||
"type": "computer_use_preview",
|
||||
"display_width": 1024,
|
||||
"display_height": 768,
|
||||
"environment": "browser" # other possible values: "mac", "windows", "ubuntu"
|
||||
}],
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Check the latest OpenAI news on bing.com."
|
||||
}
|
||||
# Optional: include a screenshot of the initial state of the environment
|
||||
# {
|
||||
# type: "input_image",
|
||||
# image_url: f"data:image/png;base64,{screenshot_base64}"
|
||||
# }
|
||||
]
|
||||
}
|
||||
],
|
||||
reasoning={
|
||||
"summary": "concise",
|
||||
},
|
||||
truncation="auto"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
122
docs/my-website/docs/providers/openai/text_to_speech.md
Normal file
122
docs/my-website/docs/providers/openai/text_to_speech.md
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# OpenAI - Text-to-speech
|
||||
|
||||
## **LiteLLM Python SDK Usage**
|
||||
### Quick Start
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
from litellm import speech
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = speech(
|
||||
model="openai/tts-1",
|
||||
voice="alloy",
|
||||
input="the quick brown fox jumped over the lazy dogs",
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
```
|
||||
|
||||
### Async Usage
|
||||
|
||||
```python
|
||||
from litellm import aspeech
|
||||
from pathlib import Path
|
||||
import os, asyncio
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
async def test_async_speech():
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = await litellm.aspeech(
|
||||
model="openai/tts-1",
|
||||
voice="alloy",
|
||||
input="the quick brown fox jumped over the lazy dogs",
|
||||
api_base=None,
|
||||
api_key=None,
|
||||
organization=None,
|
||||
project=None,
|
||||
max_retries=1,
|
||||
timeout=600,
|
||||
client=None,
|
||||
optional_params={},
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
|
||||
asyncio.run(test_async_speech())
|
||||
```
|
||||
|
||||
## **LiteLLM Proxy Usage**
|
||||
|
||||
LiteLLM provides an openai-compatible `/audio/speech` endpoint for Text-to-speech calls.
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "tts-1",
|
||||
"input": "The quick brown fox jumped over the lazy dog.",
|
||||
"voice": "alloy"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
**Setup**
|
||||
|
||||
```bash
|
||||
- model_name: tts
|
||||
litellm_params:
|
||||
model: openai/tts-1
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model | Example |
|
||||
|-------|-------------|
|
||||
| tts-1 | speech(model="tts-1", voice="alloy", input="Hello, world!") |
|
||||
| tts-1-hd | speech(model="tts-1-hd", voice="alloy", input="Hello, world!") |
|
||||
| gpt-4o-mini-tts | speech(model="gpt-4o-mini-tts", voice="alloy", input="Hello, world!") |
|
||||
|
||||
|
||||
## ✨ Enterprise LiteLLM Proxy - Set Max Request File Size
|
||||
|
||||
Use this when you want to limit the file size for requests sent to `audio/transcriptions`
|
||||
|
||||
```yaml
|
||||
- model_name: whisper
|
||||
litellm_params:
|
||||
model: whisper-1
|
||||
api_key: sk-*******
|
||||
max_file_size_mb: 0.00001 # 👈 max file size in MB (Set this intentionally very small for testing)
|
||||
model_info:
|
||||
mode: audio_transcription
|
||||
```
|
||||
|
||||
Make a test Request with a valid file
|
||||
```shell
|
||||
curl --location 'http://localhost:4000/v1/audio/transcriptions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--form 'file=@"/Users/ishaanjaffer/Github/litellm/tests/gettysburg.wav"' \
|
||||
--form 'model="whisper"'
|
||||
```
|
||||
|
||||
|
||||
Expect to see the follow response
|
||||
|
||||
```shell
|
||||
{"error":{"message":"File size is too large. Please check your file size. Passed file size: 0.7392807006835938 MB. Max file size: 0.0001 MB","type":"bad_request","param":"file","code":500}}%
|
||||
```
|
||||
|
|
@ -1284,11 +1284,18 @@ ModelResponse(
|
|||
|
||||
|
||||
|
||||
## Llama 3 API
|
||||
## Meta/Llama API
|
||||
|
||||
| Model Name | Function Call |
|
||||
|------------------|--------------------------------------|
|
||||
| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
|
||||
| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
|
||||
| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
|
||||
| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
|
||||
| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
|
||||
| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
|
||||
| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
|
||||
| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
|
||||
|
||||
### Usage
|
||||
|
||||
|
|
|
|||
265
docs/my-website/docs/proxy/management_client.md
Normal file
265
docs/my-website/docs/proxy/management_client.md
Normal file
|
|
@ -0,0 +1,265 @@
|
|||
# LiteLLM Proxy Client
|
||||
|
||||
A Python client library for interacting with the LiteLLM proxy server. This client provides a clean, typed interface for managing models, keys, credentials, and making chat completions.
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install litellm
|
||||
```
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python
|
||||
from litellm.proxy.client import Client
|
||||
|
||||
# Initialize the client
|
||||
client = Client(
|
||||
base_url="http://localhost:4000", # Your LiteLLM proxy server URL
|
||||
api_key="sk-api-key" # Optional: API key for authentication
|
||||
)
|
||||
|
||||
# Make a chat completion request
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hello, how are you?"}
|
||||
]
|
||||
)
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
## Features
|
||||
|
||||
The client is organized into several resource clients for different functionality:
|
||||
|
||||
- `chat`: Chat completions
|
||||
- `models`: Model management
|
||||
- `model_groups`: Model group management
|
||||
- `keys`: API key management
|
||||
- `credentials`: Credential management
|
||||
- `http`: Low-level HTTP client
|
||||
|
||||
## Chat Completions
|
||||
|
||||
Make chat completion requests to your LiteLLM proxy:
|
||||
|
||||
```python
|
||||
# Basic chat completion
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{"role": "user", "content": "What's the capital of France?"}
|
||||
]
|
||||
)
|
||||
|
||||
# Stream responses
|
||||
for chunk in client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Tell me a story"}],
|
||||
stream=True
|
||||
):
|
||||
print(chunk.choices[0].delta.content or "", end="")
|
||||
```
|
||||
|
||||
## Model Management
|
||||
|
||||
Manage available models on your proxy:
|
||||
|
||||
```python
|
||||
# List available models
|
||||
models = client.models.list()
|
||||
|
||||
# Add a new model
|
||||
client.models.add(
|
||||
model_name="gpt-4",
|
||||
litellm_params={
|
||||
"api_key": "your-openai-key",
|
||||
"api_base": "https://api.openai.com/v1"
|
||||
}
|
||||
)
|
||||
|
||||
# Delete a model
|
||||
client.models.delete(model_name="gpt-4")
|
||||
```
|
||||
|
||||
## API Key Management
|
||||
|
||||
Manage virtual API keys:
|
||||
|
||||
```python
|
||||
# Generate a new API key
|
||||
key = client.keys.generate(
|
||||
models=["gpt-4", "gpt-3.5-turbo"],
|
||||
aliases={"gpt4": "gpt-4"},
|
||||
duration="24h",
|
||||
key_alias="my-key",
|
||||
team_id="team123"
|
||||
)
|
||||
|
||||
# List all keys
|
||||
keys = client.keys.list(
|
||||
page=1,
|
||||
size=10,
|
||||
return_full_object=True
|
||||
)
|
||||
|
||||
# Delete keys
|
||||
client.keys.delete(
|
||||
keys=["sk-key1", "sk-key2"],
|
||||
key_aliases=["alias1", "alias2"]
|
||||
)
|
||||
```
|
||||
|
||||
## Credential Management
|
||||
|
||||
Manage model credentials:
|
||||
|
||||
```python
|
||||
# Create new credentials
|
||||
client.credentials.create(
|
||||
credential_name="azure1",
|
||||
credential_info={"api_type": "azure"},
|
||||
credential_values={
|
||||
"api_key": "your-azure-key",
|
||||
"api_base": "https://example.azure.openai.com"
|
||||
}
|
||||
)
|
||||
|
||||
# List all credentials
|
||||
credentials = client.credentials.list()
|
||||
|
||||
# Get a specific credential
|
||||
credential = client.credentials.get(credential_name="azure1")
|
||||
|
||||
# Delete credentials
|
||||
client.credentials.delete(credential_name="azure1")
|
||||
```
|
||||
|
||||
## Model Groups
|
||||
|
||||
Manage model groups for load balancing and fallbacks:
|
||||
|
||||
```python
|
||||
# Create a model group
|
||||
client.model_groups.create(
|
||||
name="gpt4-group",
|
||||
models=[
|
||||
{"model_name": "gpt-4", "litellm_params": {"api_key": "key1"}},
|
||||
{"model_name": "gpt-4-backup", "litellm_params": {"api_key": "key2"}}
|
||||
]
|
||||
)
|
||||
|
||||
# List model groups
|
||||
groups = client.model_groups.list()
|
||||
|
||||
# Delete a model group
|
||||
client.model_groups.delete(name="gpt4-group")
|
||||
```
|
||||
|
||||
## Low-Level HTTP Client
|
||||
|
||||
The client provides access to a low-level HTTP client for making direct requests
|
||||
to the LiteLLM proxy server. This is useful when you need more control or when
|
||||
working with endpoints that don't yet have a high-level interface.
|
||||
|
||||
```python
|
||||
# Access the HTTP client
|
||||
client = Client(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="sk-api-key"
|
||||
)
|
||||
|
||||
# Make a custom request
|
||||
response = client.http.request(
|
||||
method="POST",
|
||||
uri="/health/test_connection",
|
||||
json={
|
||||
"litellm_params": {
|
||||
"model": "gpt-4",
|
||||
"api_key": "your-api-key",
|
||||
"api_base": "https://api.openai.com/v1"
|
||||
},
|
||||
"mode": "chat"
|
||||
}
|
||||
)
|
||||
|
||||
# The response is automatically parsed from JSON
|
||||
print(response)
|
||||
```
|
||||
|
||||
### HTTP Client Features
|
||||
|
||||
- Automatic URL handling (handles trailing/leading slashes)
|
||||
- Built-in authentication (adds Bearer token if `api_key` is provided)
|
||||
- JSON request/response handling
|
||||
- Configurable timeout (default: 30 seconds)
|
||||
- Comprehensive error handling
|
||||
- Support for custom headers and request parameters
|
||||
|
||||
### HTTP Client `request` method parameters
|
||||
|
||||
- `method`: HTTP method (GET, POST, PUT, DELETE, etc.)
|
||||
- `uri`: URI path (will be appended to base_url)
|
||||
- `data`: (optional) Data to send in the request body
|
||||
- `json`: (optional) JSON data to send in the request body
|
||||
- `headers`: (optional) Custom HTTP headers
|
||||
- Additional keyword arguments are passed to the underlying requests library
|
||||
|
||||
## Error Handling
|
||||
|
||||
The client provides clear error handling with custom exceptions:
|
||||
|
||||
```python
|
||||
from litellm.proxy.client.exceptions import UnauthorizedError
|
||||
|
||||
try:
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
except UnauthorizedError as e:
|
||||
print("Authentication failed:", e)
|
||||
except Exception as e:
|
||||
print("Request failed:", e)
|
||||
```
|
||||
|
||||
## Advanced Usage
|
||||
|
||||
### Request Customization
|
||||
|
||||
All methods support returning the raw request object for inspection or modification:
|
||||
|
||||
```python
|
||||
# Get the prepared request without sending it
|
||||
request = client.models.list(return_request=True)
|
||||
print(request.method) # GET
|
||||
print(request.url) # http://localhost:8000/models
|
||||
print(request.headers) # {'Content-Type': 'application/json', ...}
|
||||
```
|
||||
|
||||
### Pagination
|
||||
|
||||
Methods that return lists support pagination:
|
||||
|
||||
```python
|
||||
# Get the first page of keys
|
||||
page1 = client.keys.list(page=1, size=10)
|
||||
|
||||
# Get the second page
|
||||
page2 = client.keys.list(page=2, size=10)
|
||||
```
|
||||
|
||||
### Filtering
|
||||
|
||||
Many list methods support filtering:
|
||||
|
||||
```python
|
||||
# Filter keys by user and team
|
||||
keys = client.keys.list(
|
||||
user_id="user123",
|
||||
team_id="team456",
|
||||
include_team_keys=True
|
||||
)
|
||||
```
|
||||
|
|
@ -18,3 +18,8 @@ Follow our release notes [here](https://github.com/BerriAI/litellm/releases).
|
|||
|
||||
Stable releases come out every week (typically Sunday)
|
||||
|
||||
### What is considered a 'minor' bump vs. 'patch' bump?
|
||||
|
||||
- 'patch' bumps: extremely minor addition that doesn't affect any existing functionality or add any user-facing features. (e.g. a 'created_at' column in a database table)
|
||||
- 'minor' bumps: add a new feature or a new database table that is backward compatible.
|
||||
- 'major' bumps: break backward compatibility.
|
||||
|
|
@ -786,6 +786,17 @@ Expected Response:
|
|||
}
|
||||
}
|
||||
```
|
||||
|
||||
### [BETA] Multi-instance rate limiting
|
||||
|
||||
Enable multi-instance rate limiting with the env var `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"`
|
||||
|
||||
Changes:
|
||||
- This moves to using async_increment instead of async_set_cache when updating current requests/tokens.
|
||||
- The in-memory cache is synced with redis every 0.01s, to avoid calling redis for every request.
|
||||
- In testing, this was found to be 2x faster than the previous implementation, and reduced drift between expected and actual fails to at most 10 requests at high-traffic (100 RPS across 3 instances).
|
||||
|
||||
|
||||
## Grant Access to new model
|
||||
|
||||
Use model access groups to give users access to select models, and add new ones to it over time (e.g. mistral, llama-2, etc.).
|
||||
|
|
|
|||
136
docs/my-website/release_notes/v1.68.0-stable/index.md
Normal file
136
docs/my-website/release_notes/v1.68.0-stable/index.md
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
---
|
||||
title: v1.68.0-stable
|
||||
slug: v1.68.0-stable
|
||||
date: 2025-05-03T10:00:00
|
||||
authors:
|
||||
- name: Krrish Dholakia
|
||||
title: CEO, LiteLLM
|
||||
url: https://www.linkedin.com/in/krish-d/
|
||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8
|
||||
- name: Ishaan Jaffer
|
||||
title: CTO, LiteLLM
|
||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
|
||||
|
||||
## Deploy this version
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
``` showLineNumbers title="docker run litellm"
|
||||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.68.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="pip" label="Pip">
|
||||
|
||||
``` showLineNumbers title="pip install litellm"
|
||||
pip install litellm==1.68.0.post1
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## New Models / Updated Models
|
||||
- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))**
|
||||
- Handle more json schema - openapi schema conversion edge cases [PR](https://github.com/BerriAI/litellm/pull/10351)
|
||||
- Tool calls - return ‘finish_reason=“tool_calls”’ on gemini tool calling response [PR](https://github.com/BerriAI/litellm/pull/10485)
|
||||
- **[VertexAI](../../docs/providers/vertex#metallama-api)**
|
||||
- Meta/llama-4 model support [PR](https://github.com/BerriAI/litellm/pull/10492)
|
||||
- Meta/llama3 - handle tool call result in content [PR](https://github.com/BerriAI/litellm/pull/10492)
|
||||
- Meta/* - return ‘finish_reason=“tool_calls”’ on tool calling response [PR](https://github.com/BerriAI/litellm/pull/10492)
|
||||
- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)**
|
||||
- [Image Generation](../../docs/providers/bedrock#image-generation) - Support new ‘stable-image-core’ models - [PR](https://github.com/BerriAI/litellm/pull/10351)
|
||||
- [Knowledge Bases](../../docs/completion/knowledgebase) - support using Bedrock knowledge bases with `/chat/completions` [PR](https://github.com/BerriAI/litellm/pull/10413)
|
||||
- [Anthropic](../../docs/providers/bedrock#litellm-proxy-usage) - add ‘supports_pdf_input’ for claude-3.7-bedrock models [PR](https://github.com/BerriAI/litellm/pull/9917), [Get Started](../../docs/completion/document_understanding#checking-if-a-model-supports-pdf-input)
|
||||
- **[OpenAI](../../docs/providers/openai)**
|
||||
- Support OPENAI_BASE_URL in addition to OPENAI_API_BASE [PR](https://github.com/BerriAI/litellm/pull/10423)
|
||||
- Correctly re-raise 504 timeout errors [PR](https://github.com/BerriAI/litellm/pull/10462)
|
||||
- Native Gpt-4o-mini-tts support [PR](https://github.com/BerriAI/litellm/pull/10462)
|
||||
- 🆕 **[LlamaFile](../../docs/providers/llamafile)** provider [PR](https://github.com/BerriAI/litellm/pull/10482)
|
||||
|
||||
|
||||
|
||||
## LLM API Endpoints
|
||||
- **[Response API](../../docs/response_api)**
|
||||
- Fix for handling multi turn sessions [PR](https://github.com/BerriAI/litellm/pull/10415)
|
||||
- **[Embeddings](../../docs/embedding/supported_embedding)**
|
||||
- Caching fixes - [PR](https://github.com/BerriAI/litellm/pull/10424)
|
||||
- handle str -> list cache
|
||||
- Return usage tokens for cache hit
|
||||
- Combine usage tokens on partial cache hits
|
||||
- 🆕 **[Vector Stores](../../docs/completion/knowledgebase)**
|
||||
- Allow defining Vector Store Configs - [PR](https://github.com/BerriAI/litellm/pull/10448)
|
||||
- New StandardLoggingPayload field for requests made when a vector store is used - [PR](https://github.com/BerriAI/litellm/pull/10509)
|
||||
- Show Vector Store / KB Request on LiteLLM Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10514)
|
||||
- Allow using vector store in OpenAI API spec with tools - [PR](https://github.com/BerriAI/litellm/pull/10516)
|
||||
- **[MCP](../../docs/mcp)**
|
||||
- Ensure Non-Admin virtual keys can access /mcp routes - [PR](https://github.com/BerriAI/litellm/pull/10473)
|
||||
|
||||
**Note:** Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. Follow [here](https://github.com/BerriAI/litellm/discussions/9891) for updates.
|
||||
- **Moderations**
|
||||
- Add logging callback support for `/moderations` API - [PR](https://github.com/BerriAI/litellm/pull/10390)
|
||||
|
||||
|
||||
## Spend Tracking / Budget Improvements
|
||||
- **[OpenAI](../../docs/providers/openai)**
|
||||
- [computer-use-preview](../../docs/providers/openai/responses_api#computer-use) cost tracking / pricing [PR](https://github.com/BerriAI/litellm/pull/10422)
|
||||
- [gpt-4o-mini-tts](../../docs/providers/openai/text_to_speech) input cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10462)
|
||||
- **[Fireworks AI](../../docs/providers/fireworks_ai)** - pricing updates - new `0-4b` model pricing tier + llama4 model pricing
|
||||
- **[Budgets](../../docs/proxy/users#set-budgets)**
|
||||
- [Budget resets](../../docs/proxy/users#reset-budgets) now happen as start of day/week/month - [PR](https://github.com/BerriAI/litellm/pull/10333)
|
||||
- Trigger [Soft Budget Alerts](../../docs/proxy/alerting#soft-budget-alerts-for-virtual-keys) When Key Crosses Threshold - [PR](https://github.com/BerriAI/litellm/pull/10491)
|
||||
- **[Token Counting](../../docs/completion/token_usage#3-token_counter)**
|
||||
- Rewrite of token_counter() function to handle to prevent undercounting tokens - [PR](https://github.com/BerriAI/litellm/pull/10409)
|
||||
|
||||
|
||||
## Management Endpoints / UI
|
||||
- **Virtual Keys**
|
||||
- Fix filtering on key alias - [PR](https://github.com/BerriAI/litellm/pull/10455)
|
||||
- Support global filtering on keys - [PR](https://github.com/BerriAI/litellm/pull/10455)
|
||||
- Pagination - fix clicking on next/back buttons on table - [PR](https://github.com/BerriAI/litellm/pull/10528)
|
||||
- **Models**
|
||||
- Triton - Support adding model/provider on UI - [PR](https://github.com/BerriAI/litellm/pull/10456)
|
||||
- VertexAI - Fix adding vertex models with reusable credentials - [PR](https://github.com/BerriAI/litellm/pull/10528)
|
||||
- LLM Credentials - show existing credentials for easy editing - [PR](https://github.com/BerriAI/litellm/pull/10519)
|
||||
- **Teams**
|
||||
- Allow reassigning team to other org - [PR](https://github.com/BerriAI/litellm/pull/10527)
|
||||
- **Organizations**
|
||||
- Fix showing org budget on table - [PR](https://github.com/BerriAI/litellm/pull/10528)
|
||||
|
||||
|
||||
|
||||
## Logging / Guardrail Integrations
|
||||
- **[Langsmith](../../docs/observability/langsmith_integration)**
|
||||
- Respect [langsmith_batch_size](../../docs/observability/langsmith_integration#local-testing---control-batch-size) param - [PR](https://github.com/BerriAI/litellm/pull/10411)
|
||||
|
||||
## Performance / Loadbalancing / Reliability improvements
|
||||
- **[Redis](../../docs/proxy/caching)**
|
||||
- Ensure all redis queues are periodically flushed, this fixes an issue where redis queue size was growing indefinitely when request tags were used - [PR](https://github.com/BerriAI/litellm/pull/10393)
|
||||
- **[Rate Limits](../../docs/proxy/users#set-rate-limit)**
|
||||
- [Multi-instance rate limiting](../../docs/proxy/users#beta-multi-instance-rate-limiting) support across keys/teams/users/customers - [PR](https://github.com/BerriAI/litellm/pull/10458), [PR](https://github.com/BerriAI/litellm/pull/10497), [PR](https://github.com/BerriAI/litellm/pull/10500)
|
||||
- **[Azure OpenAI OIDC](../../docs/providers/azure#entra-id---use-azure_ad_token)**
|
||||
- allow using litellm defined params for [OIDC Auth](../../docs/providers/azure#entra-id---use-azure_ad_token) - [PR](https://github.com/BerriAI/litellm/pull/10394)
|
||||
|
||||
|
||||
## General Proxy Improvements
|
||||
- **Security**
|
||||
- Allow [blocking web crawlers](../../docs/proxy/enterprise#blocking-web-crawlers) - [PR](https://github.com/BerriAI/litellm/pull/10420)
|
||||
- **Auth**
|
||||
- Support [`x-litellm-api-key` header param by default](../../docs/pass_through/vertex_ai#use-with-virtual-keys), this fixes an issue from the prior release where `x-litellm-api-key` was not being used on vertex ai passthrough requests - [PR](https://github.com/BerriAI/litellm/pull/10392)
|
||||
- Allow key at max budget to call non-llm api endpoints - [PR](https://github.com/BerriAI/litellm/pull/10392)
|
||||
- 🆕 **[Python Client Library](../../docs/proxy/management_client) for LiteLLM Proxy management endpoints**
|
||||
- Initial PR - [PR](https://github.com/BerriAI/litellm/pull/10445)
|
||||
- Support for doing HTTP requests - [PR](https://github.com/BerriAI/litellm/pull/10452)
|
||||
- **Dependencies**
|
||||
- Don’t require uvloop for windows - [PR](https://github.com/BerriAI/litellm/pull/10483)
|
||||
|
|
@ -61,6 +61,7 @@ const sidebars = {
|
|||
href: "https://litellm-api.up.railway.app/",
|
||||
},
|
||||
"proxy/enterprise",
|
||||
"proxy/management_client",
|
||||
{
|
||||
type: "category",
|
||||
label: "Making LLM Requests",
|
||||
|
|
@ -190,7 +191,15 @@ const sidebars = {
|
|||
slug: "/providers",
|
||||
},
|
||||
items: [
|
||||
"providers/openai",
|
||||
{
|
||||
type: "category",
|
||||
label: "OpenAI",
|
||||
items: [
|
||||
"providers/openai",
|
||||
"providers/openai/responses_api",
|
||||
"providers/openai/text_to_speech",
|
||||
]
|
||||
},
|
||||
"providers/text_completion_openai",
|
||||
"providers/openai_compatible",
|
||||
"providers/azure",
|
||||
|
|
|
|||
|
|
@ -6323,7 +6323,7 @@
|
|||
"supported_modalities": ["text", "image"],
|
||||
"supported_output_modalities": ["text", "code"]
|
||||
},
|
||||
"vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": {
|
||||
"vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": {
|
||||
"max_tokens": 10e6,
|
||||
"max_input_tokens": 10e6,
|
||||
"max_output_tokens": 10e6,
|
||||
|
|
|
|||
|
|
@ -6323,7 +6323,7 @@
|
|||
"supported_modalities": ["text", "image"],
|
||||
"supported_output_modalities": ["text", "code"]
|
||||
},
|
||||
"vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": {
|
||||
"vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": {
|
||||
"max_tokens": 10e6,
|
||||
"max_input_tokens": 10e6,
|
||||
"max_output_tokens": 10e6,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue