diff --git a/docs/my-website/docs/mcp.md b/docs/my-website/docs/mcp.md
index 96d035c9387..f04324f965f 100644
--- a/docs/my-website/docs/mcp.md
+++ b/docs/my-website/docs/mcp.md
@@ -421,3 +421,9 @@ async with stdio_client(server_params) as (read, write):
+
+### Permission Management
+
+Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs.
+
+Join the discussion [here](https://github.com/BerriAI/litellm/discussions/9891)
\ No newline at end of file
diff --git a/docs/my-website/docs/observability/langsmith_integration.md b/docs/my-website/docs/observability/langsmith_integration.md
index 8f55c854db8..cada4122b20 100644
--- a/docs/my-website/docs/observability/langsmith_integration.md
+++ b/docs/my-website/docs/observability/langsmith_integration.md
@@ -1,4 +1,6 @@
import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
# Langsmith - Logging LLM Input/Output
@@ -22,10 +24,13 @@ pip install litellm
## Quick Start
Use just 2 lines of code, to instantly log your responses **across all providers** with Langsmith
+
+
```python
-litellm.success_callback = ["langsmith"]
+litellm.callbacks = ["langsmith"]
```
+
```python
import litellm
import os
@@ -37,7 +42,7 @@ os.environ["LANGSMITH_DEFAULT_RUN_NAME"] = "" # defaults to LLMRun
os.environ['OPENAI_API_KEY']=""
# set langsmith as a callback, litellm will send the data to langsmith
-litellm.success_callback = ["langsmith"]
+litellm.callbacks = ["langsmith"]
# openai call
response = litellm.completion(
@@ -47,8 +52,124 @@ response = litellm.completion(
]
)
```
+
+
+
+1. Setup config.yaml
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: openai/gpt-3.5-turbo
+ api_key: os.environ/OPENAI_API_KEY
+
+litellm_settings:
+ callbacks: ["langsmith"]
+```
+
+2. Start LiteLLM Proxy
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+```bash
+curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
+-H 'Content-Type: application/json' \
+-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \
+-d '{
+ "model": "gpt-3.5-turbo",
+ "messages": [
+ {
+ "role": "user",
+ "content": "Hey, how are you?"
+ }
+ ],
+ "max_completion_tokens": 250
+}'
+```
+
+
+
+
## Advanced
+
+### Local Testing - Control Batch Size
+
+Set the size of the batch that Langsmith will process at a time, default is 512.
+
+Set `langsmith_batch_size=1` when testing locally, to see logs land quickly.
+
+
+
+
+```python
+import litellm
+import os
+
+os.environ["LANGSMITH_API_KEY"] = ""
+# LLM API Keys
+os.environ['OPENAI_API_KEY']=""
+
+# set langsmith as a callback, litellm will send the data to langsmith
+litellm.callbacks = ["langsmith"]
+litellm.langsmith_batch_size = 1 # 👈 KEY CHANGE
+
+response = litellm.completion(
+ model="gpt-3.5-turbo",
+ messages=[
+ {"role": "user", "content": "Hi 👋 - i'm openai"}
+ ]
+)
+print(response)
+```
+
+
+
+1. Setup config.yaml
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: openai/gpt-3.5-turbo
+ api_key: os.environ/OPENAI_API_KEY
+
+litellm_settings:
+ langsmith_batch_size: 1
+ callbacks: ["langsmith"]
+```
+
+2. Start LiteLLM Proxy
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+```bash
+curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
+-H 'Content-Type: application/json' \
+-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \
+-d '{
+ "model": "gpt-3.5-turbo",
+ "messages": [
+ {
+ "role": "user",
+ "content": "Hey, how are you?"
+ }
+ ],
+ "max_completion_tokens": 250
+}'
+```
+
+
+
+
+
+
+
+
+
### Set Langsmith fields
```python
diff --git a/docs/my-website/docs/providers/bedrock.md b/docs/my-website/docs/providers/bedrock.md
index 2a9c528a655..8217f429ff3 100644
--- a/docs/my-website/docs/providers/bedrock.md
+++ b/docs/my-website/docs/providers/bedrock.md
@@ -60,9 +60,9 @@ Here's how to call Bedrock with the LiteLLM Proxy Server
```yaml
model_list:
- - model_name: bedrock-claude-v1
+ - model_name: bedrock-claude-3-5-sonnet
litellm_params:
- model: bedrock/anthropic.claude-instant-v1
+ model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
aws_region_name: os.environ/AWS_REGION_NAME
diff --git a/docs/my-website/docs/providers/openai/responses_api.md b/docs/my-website/docs/providers/openai/responses_api.md
new file mode 100644
index 00000000000..578ce038f37
--- /dev/null
+++ b/docs/my-website/docs/providers/openai/responses_api.md
@@ -0,0 +1,320 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# OpenAI - Response API
+
+## Usage
+
+### LiteLLM Python SDK
+
+
+#### Non-streaming
+```python showLineNumbers title="OpenAI Non-streaming Response"
+import litellm
+
+# Non-streaming response
+response = litellm.responses(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ max_output_tokens=100
+)
+
+print(response)
+```
+
+#### Streaming
+```python showLineNumbers title="OpenAI Streaming Response"
+import litellm
+
+# Streaming response
+response = litellm.responses(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ stream=True
+)
+
+for event in response:
+ print(event)
+```
+
+#### GET a Response
+```python showLineNumbers title="Get Response by ID"
+import litellm
+
+# First, create a response
+response = litellm.responses(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ max_output_tokens=100
+)
+
+# Get the response ID
+response_id = response.id
+
+# Retrieve the response by ID
+retrieved_response = litellm.get_responses(
+ response_id=response_id
+)
+
+print(retrieved_response)
+
+# For async usage
+# retrieved_response = await litellm.aget_responses(response_id=response_id)
+```
+
+#### DELETE a Response
+```python showLineNumbers title="Delete Response by ID"
+import litellm
+
+# First, create a response
+response = litellm.responses(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ max_output_tokens=100
+)
+
+# Get the response ID
+response_id = response.id
+
+# Delete the response by ID
+delete_response = litellm.delete_responses(
+ response_id=response_id
+)
+
+print(delete_response)
+
+# For async usage
+# delete_response = await litellm.adelete_responses(response_id=response_id)
+```
+
+
+### LiteLLM Proxy with OpenAI SDK
+
+1. Set up config.yaml
+
+```yaml showLineNumbers title="OpenAI Proxy Configuration"
+model_list:
+ - model_name: openai/o1-pro
+ litellm_params:
+ model: openai/o1-pro
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+2. Start LiteLLM Proxy Server
+
+```bash title="Start LiteLLM Proxy Server"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+3. Use OpenAI SDK with LiteLLM Proxy
+
+#### Non-streaming
+```python showLineNumbers title="OpenAI Proxy Non-streaming Response"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.responses.create(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn."
+)
+
+print(response)
+```
+
+#### Streaming
+```python showLineNumbers title="OpenAI Proxy Streaming Response"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Streaming response
+response = client.responses.create(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ stream=True
+)
+
+for event in response:
+ print(event)
+```
+
+#### GET a Response
+```python showLineNumbers title="Get Response by ID with OpenAI SDK"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# First, create a response
+response = client.responses.create(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn."
+)
+
+# Get the response ID
+response_id = response.id
+
+# Retrieve the response by ID
+retrieved_response = client.responses.retrieve(response_id)
+
+print(retrieved_response)
+```
+
+#### DELETE a Response
+```python showLineNumbers title="Delete Response by ID with OpenAI SDK"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# First, create a response
+response = client.responses.create(
+ model="openai/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn."
+)
+
+# Get the response ID
+response_id = response.id
+
+# Delete the response by ID
+delete_response = client.responses.delete(response_id)
+
+print(delete_response)
+```
+
+
+## Supported Responses API Parameters
+
+| Provider | Supported Parameters |
+|----------|---------------------|
+| `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) |
+
+## Computer Use
+
+
+
+
+```python
+import litellm
+
+# Non-streaming response
+response = litellm.responses(
+ model="computer-use-preview",
+ tools=[{
+ "type": "computer_use_preview",
+ "display_width": 1024,
+ "display_height": 768,
+ "environment": "browser" # other possible values: "mac", "windows", "ubuntu"
+ }],
+ input=[
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "Check the latest OpenAI news on bing.com."
+ }
+ # Optional: include a screenshot of the initial state of the environment
+ # {
+ # type: "input_image",
+ # image_url: f"data:image/png;base64,{screenshot_base64}"
+ # }
+ ]
+ }
+ ],
+ reasoning={
+ "summary": "concise",
+ },
+ truncation="auto"
+)
+
+print(response.output)
+```
+
+
+
+
+1. Set up config.yaml
+
+```yaml showLineNumbers title="OpenAI Proxy Configuration"
+model_list:
+ - model_name: openai/o1-pro
+ litellm_params:
+ model: openai/o1-pro
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+2. Start LiteLLM Proxy Server
+
+```bash title="Start LiteLLM Proxy Server"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+3. Test it!
+
+```python showLineNumbers title="OpenAI Proxy Non-streaming Response"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.responses.create(
+ model="computer-use-preview",
+ tools=[{
+ "type": "computer_use_preview",
+ "display_width": 1024,
+ "display_height": 768,
+ "environment": "browser" # other possible values: "mac", "windows", "ubuntu"
+ }],
+ input=[
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "Check the latest OpenAI news on bing.com."
+ }
+ # Optional: include a screenshot of the initial state of the environment
+ # {
+ # type: "input_image",
+ # image_url: f"data:image/png;base64,{screenshot_base64}"
+ # }
+ ]
+ }
+ ],
+ reasoning={
+ "summary": "concise",
+ },
+ truncation="auto"
+)
+
+print(response)
+```
+
+
+
+
diff --git a/docs/my-website/docs/providers/openai/text_to_speech.md b/docs/my-website/docs/providers/openai/text_to_speech.md
new file mode 100644
index 00000000000..34cd0f069e6
--- /dev/null
+++ b/docs/my-website/docs/providers/openai/text_to_speech.md
@@ -0,0 +1,122 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# OpenAI - Text-to-speech
+
+## **LiteLLM Python SDK Usage**
+### Quick Start
+
+```python
+from pathlib import Path
+from litellm import speech
+import os
+
+os.environ["OPENAI_API_KEY"] = "sk-.."
+
+speech_file_path = Path(__file__).parent / "speech.mp3"
+response = speech(
+ model="openai/tts-1",
+ voice="alloy",
+ input="the quick brown fox jumped over the lazy dogs",
+ )
+response.stream_to_file(speech_file_path)
+```
+
+### Async Usage
+
+```python
+from litellm import aspeech
+from pathlib import Path
+import os, asyncio
+
+os.environ["OPENAI_API_KEY"] = "sk-.."
+
+async def test_async_speech():
+ speech_file_path = Path(__file__).parent / "speech.mp3"
+ response = await litellm.aspeech(
+ model="openai/tts-1",
+ voice="alloy",
+ input="the quick brown fox jumped over the lazy dogs",
+ api_base=None,
+ api_key=None,
+ organization=None,
+ project=None,
+ max_retries=1,
+ timeout=600,
+ client=None,
+ optional_params={},
+ )
+ response.stream_to_file(speech_file_path)
+
+asyncio.run(test_async_speech())
+```
+
+## **LiteLLM Proxy Usage**
+
+LiteLLM provides an openai-compatible `/audio/speech` endpoint for Text-to-speech calls.
+
+```bash
+curl http://0.0.0.0:4000/v1/audio/speech \
+ -H "Authorization: Bearer sk-1234" \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "tts-1",
+ "input": "The quick brown fox jumped over the lazy dog.",
+ "voice": "alloy"
+ }' \
+ --output speech.mp3
+```
+
+**Setup**
+
+```bash
+- model_name: tts
+ litellm_params:
+ model: openai/tts-1
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+## Supported Models
+
+| Model | Example |
+|-------|-------------|
+| tts-1 | speech(model="tts-1", voice="alloy", input="Hello, world!") |
+| tts-1-hd | speech(model="tts-1-hd", voice="alloy", input="Hello, world!") |
+| gpt-4o-mini-tts | speech(model="gpt-4o-mini-tts", voice="alloy", input="Hello, world!") |
+
+
+## ✨ Enterprise LiteLLM Proxy - Set Max Request File Size
+
+Use this when you want to limit the file size for requests sent to `audio/transcriptions`
+
+```yaml
+- model_name: whisper
+ litellm_params:
+ model: whisper-1
+ api_key: sk-*******
+ max_file_size_mb: 0.00001 # 👈 max file size in MB (Set this intentionally very small for testing)
+ model_info:
+ mode: audio_transcription
+```
+
+Make a test Request with a valid file
+```shell
+curl --location 'http://localhost:4000/v1/audio/transcriptions' \
+--header 'Authorization: Bearer sk-1234' \
+--form 'file=@"/Users/ishaanjaffer/Github/litellm/tests/gettysburg.wav"' \
+--form 'model="whisper"'
+```
+
+
+Expect to see the follow response
+
+```shell
+{"error":{"message":"File size is too large. Please check your file size. Passed file size: 0.7392807006835938 MB. Max file size: 0.0001 MB","type":"bad_request","param":"file","code":500}}%
+```
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md
index b328c805770..30887e9f60d 100644
--- a/docs/my-website/docs/providers/vertex.md
+++ b/docs/my-website/docs/providers/vertex.md
@@ -1284,11 +1284,18 @@ ModelResponse(
-## Llama 3 API
+## Meta/Llama API
| Model Name | Function Call |
|------------------|--------------------------------------|
+| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
+| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
+| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
+| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
+| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
+| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
+| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
### Usage
diff --git a/docs/my-website/docs/proxy/management_client.md b/docs/my-website/docs/proxy/management_client.md
new file mode 100644
index 00000000000..7bf09a07a71
--- /dev/null
+++ b/docs/my-website/docs/proxy/management_client.md
@@ -0,0 +1,265 @@
+# LiteLLM Proxy Client
+
+A Python client library for interacting with the LiteLLM proxy server. This client provides a clean, typed interface for managing models, keys, credentials, and making chat completions.
+
+## Installation
+
+```bash
+pip install litellm
+```
+
+## Quick Start
+
+```python
+from litellm.proxy.client import Client
+
+# Initialize the client
+client = Client(
+ base_url="http://localhost:4000", # Your LiteLLM proxy server URL
+ api_key="sk-api-key" # Optional: API key for authentication
+)
+
+# Make a chat completion request
+response = client.chat.completions.create(
+ model="gpt-3.5-turbo",
+ messages=[
+ {"role": "user", "content": "Hello, how are you?"}
+ ]
+)
+print(response.choices[0].message.content)
+```
+
+## Features
+
+The client is organized into several resource clients for different functionality:
+
+- `chat`: Chat completions
+- `models`: Model management
+- `model_groups`: Model group management
+- `keys`: API key management
+- `credentials`: Credential management
+- `http`: Low-level HTTP client
+
+## Chat Completions
+
+Make chat completion requests to your LiteLLM proxy:
+
+```python
+# Basic chat completion
+response = client.chat.completions.create(
+ model="gpt-4",
+ messages=[
+ {"role": "system", "content": "You are a helpful assistant."},
+ {"role": "user", "content": "What's the capital of France?"}
+ ]
+)
+
+# Stream responses
+for chunk in client.chat.completions.create(
+ model="gpt-4",
+ messages=[{"role": "user", "content": "Tell me a story"}],
+ stream=True
+):
+ print(chunk.choices[0].delta.content or "", end="")
+```
+
+## Model Management
+
+Manage available models on your proxy:
+
+```python
+# List available models
+models = client.models.list()
+
+# Add a new model
+client.models.add(
+ model_name="gpt-4",
+ litellm_params={
+ "api_key": "your-openai-key",
+ "api_base": "https://api.openai.com/v1"
+ }
+)
+
+# Delete a model
+client.models.delete(model_name="gpt-4")
+```
+
+## API Key Management
+
+Manage virtual API keys:
+
+```python
+# Generate a new API key
+key = client.keys.generate(
+ models=["gpt-4", "gpt-3.5-turbo"],
+ aliases={"gpt4": "gpt-4"},
+ duration="24h",
+ key_alias="my-key",
+ team_id="team123"
+)
+
+# List all keys
+keys = client.keys.list(
+ page=1,
+ size=10,
+ return_full_object=True
+)
+
+# Delete keys
+client.keys.delete(
+ keys=["sk-key1", "sk-key2"],
+ key_aliases=["alias1", "alias2"]
+)
+```
+
+## Credential Management
+
+Manage model credentials:
+
+```python
+# Create new credentials
+client.credentials.create(
+ credential_name="azure1",
+ credential_info={"api_type": "azure"},
+ credential_values={
+ "api_key": "your-azure-key",
+ "api_base": "https://example.azure.openai.com"
+ }
+)
+
+# List all credentials
+credentials = client.credentials.list()
+
+# Get a specific credential
+credential = client.credentials.get(credential_name="azure1")
+
+# Delete credentials
+client.credentials.delete(credential_name="azure1")
+```
+
+## Model Groups
+
+Manage model groups for load balancing and fallbacks:
+
+```python
+# Create a model group
+client.model_groups.create(
+ name="gpt4-group",
+ models=[
+ {"model_name": "gpt-4", "litellm_params": {"api_key": "key1"}},
+ {"model_name": "gpt-4-backup", "litellm_params": {"api_key": "key2"}}
+ ]
+)
+
+# List model groups
+groups = client.model_groups.list()
+
+# Delete a model group
+client.model_groups.delete(name="gpt4-group")
+```
+
+## Low-Level HTTP Client
+
+The client provides access to a low-level HTTP client for making direct requests
+to the LiteLLM proxy server. This is useful when you need more control or when
+working with endpoints that don't yet have a high-level interface.
+
+```python
+# Access the HTTP client
+client = Client(
+ base_url="http://localhost:4000",
+ api_key="sk-api-key"
+)
+
+# Make a custom request
+response = client.http.request(
+ method="POST",
+ uri="/health/test_connection",
+ json={
+ "litellm_params": {
+ "model": "gpt-4",
+ "api_key": "your-api-key",
+ "api_base": "https://api.openai.com/v1"
+ },
+ "mode": "chat"
+ }
+)
+
+# The response is automatically parsed from JSON
+print(response)
+```
+
+### HTTP Client Features
+
+- Automatic URL handling (handles trailing/leading slashes)
+- Built-in authentication (adds Bearer token if `api_key` is provided)
+- JSON request/response handling
+- Configurable timeout (default: 30 seconds)
+- Comprehensive error handling
+- Support for custom headers and request parameters
+
+### HTTP Client `request` method parameters
+
+- `method`: HTTP method (GET, POST, PUT, DELETE, etc.)
+- `uri`: URI path (will be appended to base_url)
+- `data`: (optional) Data to send in the request body
+- `json`: (optional) JSON data to send in the request body
+- `headers`: (optional) Custom HTTP headers
+- Additional keyword arguments are passed to the underlying requests library
+
+## Error Handling
+
+The client provides clear error handling with custom exceptions:
+
+```python
+from litellm.proxy.client.exceptions import UnauthorizedError
+
+try:
+ response = client.chat.completions.create(
+ model="gpt-4",
+ messages=[{"role": "user", "content": "Hello"}]
+ )
+except UnauthorizedError as e:
+ print("Authentication failed:", e)
+except Exception as e:
+ print("Request failed:", e)
+```
+
+## Advanced Usage
+
+### Request Customization
+
+All methods support returning the raw request object for inspection or modification:
+
+```python
+# Get the prepared request without sending it
+request = client.models.list(return_request=True)
+print(request.method) # GET
+print(request.url) # http://localhost:8000/models
+print(request.headers) # {'Content-Type': 'application/json', ...}
+```
+
+### Pagination
+
+Methods that return lists support pagination:
+
+```python
+# Get the first page of keys
+page1 = client.keys.list(page=1, size=10)
+
+# Get the second page
+page2 = client.keys.list(page=2, size=10)
+```
+
+### Filtering
+
+Many list methods support filtering:
+
+```python
+# Filter keys by user and team
+keys = client.keys.list(
+ user_id="user123",
+ team_id="team456",
+ include_team_keys=True
+)
+```
\ No newline at end of file
diff --git a/docs/my-website/docs/proxy/release_cycle.md b/docs/my-website/docs/proxy/release_cycle.md
index c5782087f21..10dd6d8b3c5 100644
--- a/docs/my-website/docs/proxy/release_cycle.md
+++ b/docs/my-website/docs/proxy/release_cycle.md
@@ -18,3 +18,8 @@ Follow our release notes [here](https://github.com/BerriAI/litellm/releases).
Stable releases come out every week (typically Sunday)
+### What is considered a 'minor' bump vs. 'patch' bump?
+
+- 'patch' bumps: extremely minor addition that doesn't affect any existing functionality or add any user-facing features. (e.g. a 'created_at' column in a database table)
+- 'minor' bumps: add a new feature or a new database table that is backward compatible.
+- 'major' bumps: break backward compatibility.
\ No newline at end of file
diff --git a/docs/my-website/docs/proxy/users.md b/docs/my-website/docs/proxy/users.md
index 92ea73b9d2e..b4457b8d553 100644
--- a/docs/my-website/docs/proxy/users.md
+++ b/docs/my-website/docs/proxy/users.md
@@ -786,6 +786,17 @@ Expected Response:
}
}
```
+
+### [BETA] Multi-instance rate limiting
+
+Enable multi-instance rate limiting with the env var `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"`
+
+Changes:
+- This moves to using async_increment instead of async_set_cache when updating current requests/tokens.
+- The in-memory cache is synced with redis every 0.01s, to avoid calling redis for every request.
+- In testing, this was found to be 2x faster than the previous implementation, and reduced drift between expected and actual fails to at most 10 requests at high-traffic (100 RPS across 3 instances).
+
+
## Grant Access to new model
Use model access groups to give users access to select models, and add new ones to it over time (e.g. mistral, llama-2, etc.).
diff --git a/docs/my-website/release_notes/v1.68.0-stable/index.md b/docs/my-website/release_notes/v1.68.0-stable/index.md
new file mode 100644
index 00000000000..c47cfda2348
--- /dev/null
+++ b/docs/my-website/release_notes/v1.68.0-stable/index.md
@@ -0,0 +1,136 @@
+---
+title: v1.68.0-stable
+slug: v1.68.0-stable
+date: 2025-05-03T10:00:00
+authors:
+ - name: Krrish Dholakia
+ title: CEO, LiteLLM
+ url: https://www.linkedin.com/in/krish-d/
+ image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8
+ - name: Ishaan Jaffer
+ title: CTO, LiteLLM
+ url: https://www.linkedin.com/in/reffajnaahsi/
+ image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
+
+hide_table_of_contents: false
+---
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+
+
+## Deploy this version
+
+
+
+
+``` showLineNumbers title="docker run litellm"
+docker run
+-e STORE_MODEL_IN_DB=True
+-p 4000:4000
+ghcr.io/berriai/litellm:main-v1.68.0-stable
+```
+
+
+
+
+``` showLineNumbers title="pip install litellm"
+pip install litellm==1.68.0.post1
+```
+
+
+
+## New Models / Updated Models
+- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))**
+ - Handle more json schema - openapi schema conversion edge cases [PR](https://github.com/BerriAI/litellm/pull/10351)
+ - Tool calls - return ‘finish_reason=“tool_calls”’ on gemini tool calling response [PR](https://github.com/BerriAI/litellm/pull/10485)
+- **[VertexAI](../../docs/providers/vertex#metallama-api)**
+ - Meta/llama-4 model support [PR](https://github.com/BerriAI/litellm/pull/10492)
+ - Meta/llama3 - handle tool call result in content [PR](https://github.com/BerriAI/litellm/pull/10492)
+ - Meta/* - return ‘finish_reason=“tool_calls”’ on tool calling response [PR](https://github.com/BerriAI/litellm/pull/10492)
+- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)**
+ - [Image Generation](../../docs/providers/bedrock#image-generation) - Support new ‘stable-image-core’ models - [PR](https://github.com/BerriAI/litellm/pull/10351)
+ - [Knowledge Bases](../../docs/completion/knowledgebase) - support using Bedrock knowledge bases with `/chat/completions` [PR](https://github.com/BerriAI/litellm/pull/10413)
+ - [Anthropic](../../docs/providers/bedrock#litellm-proxy-usage) - add ‘supports_pdf_input’ for claude-3.7-bedrock models [PR](https://github.com/BerriAI/litellm/pull/9917), [Get Started](../../docs/completion/document_understanding#checking-if-a-model-supports-pdf-input)
+- **[OpenAI](../../docs/providers/openai)**
+ - Support OPENAI_BASE_URL in addition to OPENAI_API_BASE [PR](https://github.com/BerriAI/litellm/pull/10423)
+ - Correctly re-raise 504 timeout errors [PR](https://github.com/BerriAI/litellm/pull/10462)
+ - Native Gpt-4o-mini-tts support [PR](https://github.com/BerriAI/litellm/pull/10462)
+- 🆕 **[LlamaFile](../../docs/providers/llamafile)** provider [PR](https://github.com/BerriAI/litellm/pull/10482)
+
+
+
+## LLM API Endpoints
+- **[Response API](../../docs/response_api)**
+ - Fix for handling multi turn sessions [PR](https://github.com/BerriAI/litellm/pull/10415)
+- **[Embeddings](../../docs/embedding/supported_embedding)**
+ - Caching fixes - [PR](https://github.com/BerriAI/litellm/pull/10424)
+ - handle str -> list cache
+ - Return usage tokens for cache hit
+ - Combine usage tokens on partial cache hits
+- 🆕 **[Vector Stores](../../docs/completion/knowledgebase)**
+ - Allow defining Vector Store Configs - [PR](https://github.com/BerriAI/litellm/pull/10448)
+ - New StandardLoggingPayload field for requests made when a vector store is used - [PR](https://github.com/BerriAI/litellm/pull/10509)
+ - Show Vector Store / KB Request on LiteLLM Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10514)
+ - Allow using vector store in OpenAI API spec with tools - [PR](https://github.com/BerriAI/litellm/pull/10516)
+- **[MCP](../../docs/mcp)**
+ - Ensure Non-Admin virtual keys can access /mcp routes - [PR](https://github.com/BerriAI/litellm/pull/10473)
+
+ **Note:** Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. Follow [here](https://github.com/BerriAI/litellm/discussions/9891) for updates.
+- **Moderations**
+ - Add logging callback support for `/moderations` API - [PR](https://github.com/BerriAI/litellm/pull/10390)
+
+
+## Spend Tracking / Budget Improvements
+- **[OpenAI](../../docs/providers/openai)**
+ - [computer-use-preview](../../docs/providers/openai/responses_api#computer-use) cost tracking / pricing [PR](https://github.com/BerriAI/litellm/pull/10422)
+ - [gpt-4o-mini-tts](../../docs/providers/openai/text_to_speech) input cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10462)
+- **[Fireworks AI](../../docs/providers/fireworks_ai)** - pricing updates - new `0-4b` model pricing tier + llama4 model pricing
+- **[Budgets](../../docs/proxy/users#set-budgets)**
+ - [Budget resets](../../docs/proxy/users#reset-budgets) now happen as start of day/week/month - [PR](https://github.com/BerriAI/litellm/pull/10333)
+ - Trigger [Soft Budget Alerts](../../docs/proxy/alerting#soft-budget-alerts-for-virtual-keys) When Key Crosses Threshold - [PR](https://github.com/BerriAI/litellm/pull/10491)
+- **[Token Counting](../../docs/completion/token_usage#3-token_counter)**
+ - Rewrite of token_counter() function to handle to prevent undercounting tokens - [PR](https://github.com/BerriAI/litellm/pull/10409)
+
+
+## Management Endpoints / UI
+- **Virtual Keys**
+ - Fix filtering on key alias - [PR](https://github.com/BerriAI/litellm/pull/10455)
+ - Support global filtering on keys - [PR](https://github.com/BerriAI/litellm/pull/10455)
+ - Pagination - fix clicking on next/back buttons on table - [PR](https://github.com/BerriAI/litellm/pull/10528)
+- **Models**
+ - Triton - Support adding model/provider on UI - [PR](https://github.com/BerriAI/litellm/pull/10456)
+ - VertexAI - Fix adding vertex models with reusable credentials - [PR](https://github.com/BerriAI/litellm/pull/10528)
+ - LLM Credentials - show existing credentials for easy editing - [PR](https://github.com/BerriAI/litellm/pull/10519)
+- **Teams**
+ - Allow reassigning team to other org - [PR](https://github.com/BerriAI/litellm/pull/10527)
+- **Organizations**
+ - Fix showing org budget on table - [PR](https://github.com/BerriAI/litellm/pull/10528)
+
+
+
+## Logging / Guardrail Integrations
+- **[Langsmith](../../docs/observability/langsmith_integration)**
+ - Respect [langsmith_batch_size](../../docs/observability/langsmith_integration#local-testing---control-batch-size) param - [PR](https://github.com/BerriAI/litellm/pull/10411)
+
+## Performance / Loadbalancing / Reliability improvements
+- **[Redis](../../docs/proxy/caching)**
+ - Ensure all redis queues are periodically flushed, this fixes an issue where redis queue size was growing indefinitely when request tags were used - [PR](https://github.com/BerriAI/litellm/pull/10393)
+- **[Rate Limits](../../docs/proxy/users#set-rate-limit)**
+ - [Multi-instance rate limiting](../../docs/proxy/users#beta-multi-instance-rate-limiting) support across keys/teams/users/customers - [PR](https://github.com/BerriAI/litellm/pull/10458), [PR](https://github.com/BerriAI/litellm/pull/10497), [PR](https://github.com/BerriAI/litellm/pull/10500)
+- **[Azure OpenAI OIDC](../../docs/providers/azure#entra-id---use-azure_ad_token)**
+ - allow using litellm defined params for [OIDC Auth](../../docs/providers/azure#entra-id---use-azure_ad_token) - [PR](https://github.com/BerriAI/litellm/pull/10394)
+
+
+## General Proxy Improvements
+- **Security**
+ - Allow [blocking web crawlers](../../docs/proxy/enterprise#blocking-web-crawlers) - [PR](https://github.com/BerriAI/litellm/pull/10420)
+- **Auth**
+ - Support [`x-litellm-api-key` header param by default](../../docs/pass_through/vertex_ai#use-with-virtual-keys), this fixes an issue from the prior release where `x-litellm-api-key` was not being used on vertex ai passthrough requests - [PR](https://github.com/BerriAI/litellm/pull/10392)
+ - Allow key at max budget to call non-llm api endpoints - [PR](https://github.com/BerriAI/litellm/pull/10392)
+- 🆕 **[Python Client Library](../../docs/proxy/management_client) for LiteLLM Proxy management endpoints**
+ - Initial PR - [PR](https://github.com/BerriAI/litellm/pull/10445)
+ - Support for doing HTTP requests - [PR](https://github.com/BerriAI/litellm/pull/10452)
+- **Dependencies**
+ - Don’t require uvloop for windows - [PR](https://github.com/BerriAI/litellm/pull/10483)
diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js
index 1e97e76b150..9f4c392d41f 100644
--- a/docs/my-website/sidebars.js
+++ b/docs/my-website/sidebars.js
@@ -61,6 +61,7 @@ const sidebars = {
href: "https://litellm-api.up.railway.app/",
},
"proxy/enterprise",
+ "proxy/management_client",
{
type: "category",
label: "Making LLM Requests",
@@ -190,7 +191,15 @@ const sidebars = {
slug: "/providers",
},
items: [
- "providers/openai",
+ {
+ type: "category",
+ label: "OpenAI",
+ items: [
+ "providers/openai",
+ "providers/openai/responses_api",
+ "providers/openai/text_to_speech",
+ ]
+ },
"providers/text_completion_openai",
"providers/openai_compatible",
"providers/azure",
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index c148f04e336..efef685d696 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -6323,7 +6323,7 @@
"supported_modalities": ["text", "image"],
"supported_output_modalities": ["text", "code"]
},
- "vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": {
+ "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": {
"max_tokens": 10e6,
"max_input_tokens": 10e6,
"max_output_tokens": 10e6,
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index c148f04e336..efef685d696 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -6323,7 +6323,7 @@
"supported_modalities": ["text", "image"],
"supported_output_modalities": ["text", "code"]
},
- "vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": {
+ "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": {
"max_tokens": 10e6,
"max_input_tokens": 10e6,
"max_output_tokens": 10e6,