diff --git a/docs/my-website/docs/mcp.md b/docs/my-website/docs/mcp.md index 96d035c9387..f04324f965f 100644 --- a/docs/my-website/docs/mcp.md +++ b/docs/my-website/docs/mcp.md @@ -421,3 +421,9 @@ async with stdio_client(server_params) as (read, write): + +### Permission Management + +Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. + +Join the discussion [here](https://github.com/BerriAI/litellm/discussions/9891) \ No newline at end of file diff --git a/docs/my-website/docs/observability/langsmith_integration.md b/docs/my-website/docs/observability/langsmith_integration.md index 8f55c854db8..cada4122b20 100644 --- a/docs/my-website/docs/observability/langsmith_integration.md +++ b/docs/my-website/docs/observability/langsmith_integration.md @@ -1,4 +1,6 @@ import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; # Langsmith - Logging LLM Input/Output @@ -22,10 +24,13 @@ pip install litellm ## Quick Start Use just 2 lines of code, to instantly log your responses **across all providers** with Langsmith + + ```python -litellm.success_callback = ["langsmith"] +litellm.callbacks = ["langsmith"] ``` + ```python import litellm import os @@ -37,7 +42,7 @@ os.environ["LANGSMITH_DEFAULT_RUN_NAME"] = "" # defaults to LLMRun os.environ['OPENAI_API_KEY']="" # set langsmith as a callback, litellm will send the data to langsmith -litellm.success_callback = ["langsmith"] +litellm.callbacks = ["langsmith"] # openai call response = litellm.completion( @@ -47,8 +52,124 @@ response = litellm.completion( ] ) ``` + + + +1. Setup config.yaml +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +litellm_settings: + callbacks: ["langsmith"] +``` + +2. Start LiteLLM Proxy +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \ +-d '{ + "model": "gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": "Hey, how are you?" + } + ], + "max_completion_tokens": 250 +}' +``` + + + + ## Advanced + +### Local Testing - Control Batch Size + +Set the size of the batch that Langsmith will process at a time, default is 512. + +Set `langsmith_batch_size=1` when testing locally, to see logs land quickly. + + + + +```python +import litellm +import os + +os.environ["LANGSMITH_API_KEY"] = "" +# LLM API Keys +os.environ['OPENAI_API_KEY']="" + +# set langsmith as a callback, litellm will send the data to langsmith +litellm.callbacks = ["langsmith"] +litellm.langsmith_batch_size = 1 # 👈 KEY CHANGE + +response = litellm.completion( + model="gpt-3.5-turbo", + messages=[ + {"role": "user", "content": "Hi 👋 - i'm openai"} + ] +) +print(response) +``` + + + +1. Setup config.yaml +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +litellm_settings: + langsmith_batch_size: 1 + callbacks: ["langsmith"] +``` + +2. Start LiteLLM Proxy +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \ +-d '{ + "model": "gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": "Hey, how are you?" + } + ], + "max_completion_tokens": 250 +}' +``` + + + + + + + + + ### Set Langsmith fields ```python diff --git a/docs/my-website/docs/providers/bedrock.md b/docs/my-website/docs/providers/bedrock.md index 2a9c528a655..8217f429ff3 100644 --- a/docs/my-website/docs/providers/bedrock.md +++ b/docs/my-website/docs/providers/bedrock.md @@ -60,9 +60,9 @@ Here's how to call Bedrock with the LiteLLM Proxy Server ```yaml model_list: - - model_name: bedrock-claude-v1 + - model_name: bedrock-claude-3-5-sonnet litellm_params: - model: bedrock/anthropic.claude-instant-v1 + model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0 aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY aws_region_name: os.environ/AWS_REGION_NAME diff --git a/docs/my-website/docs/providers/openai/responses_api.md b/docs/my-website/docs/providers/openai/responses_api.md new file mode 100644 index 00000000000..578ce038f37 --- /dev/null +++ b/docs/my-website/docs/providers/openai/responses_api.md @@ -0,0 +1,320 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# OpenAI - Response API + +## Usage + +### LiteLLM Python SDK + + +#### Non-streaming +```python showLineNumbers title="OpenAI Non-streaming Response" +import litellm + +# Non-streaming response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100 +) + +print(response) +``` + +#### Streaming +```python showLineNumbers title="OpenAI Streaming Response" +import litellm + +# Streaming response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True +) + +for event in response: + print(event) +``` + +#### GET a Response +```python showLineNumbers title="Get Response by ID" +import litellm + +# First, create a response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100 +) + +# Get the response ID +response_id = response.id + +# Retrieve the response by ID +retrieved_response = litellm.get_responses( + response_id=response_id +) + +print(retrieved_response) + +# For async usage +# retrieved_response = await litellm.aget_responses(response_id=response_id) +``` + +#### DELETE a Response +```python showLineNumbers title="Delete Response by ID" +import litellm + +# First, create a response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100 +) + +# Get the response ID +response_id = response.id + +# Delete the response by ID +delete_response = litellm.delete_responses( + response_id=response_id +) + +print(delete_response) + +# For async usage +# delete_response = await litellm.adelete_responses(response_id=response_id) +``` + + +### LiteLLM Proxy with OpenAI SDK + +1. Set up config.yaml + +```yaml showLineNumbers title="OpenAI Proxy Configuration" +model_list: + - model_name: openai/o1-pro + litellm_params: + model: openai/o1-pro + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start LiteLLM Proxy Server + +```bash title="Start LiteLLM Proxy Server" +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +3. Use OpenAI SDK with LiteLLM Proxy + +#### Non-streaming +```python showLineNumbers title="OpenAI Proxy Non-streaming Response" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +print(response) +``` + +#### Streaming +```python showLineNumbers title="OpenAI Proxy Streaming Response" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Streaming response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True +) + +for event in response: + print(event) +``` + +#### GET a Response +```python showLineNumbers title="Get Response by ID with OpenAI SDK" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# First, create a response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +# Get the response ID +response_id = response.id + +# Retrieve the response by ID +retrieved_response = client.responses.retrieve(response_id) + +print(retrieved_response) +``` + +#### DELETE a Response +```python showLineNumbers title="Delete Response by ID with OpenAI SDK" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# First, create a response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +# Get the response ID +response_id = response.id + +# Delete the response by ID +delete_response = client.responses.delete(response_id) + +print(delete_response) +``` + + +## Supported Responses API Parameters + +| Provider | Supported Parameters | +|----------|---------------------| +| `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) | + +## Computer Use + + + + +```python +import litellm + +# Non-streaming response +response = litellm.responses( + model="computer-use-preview", + tools=[{ + "type": "computer_use_preview", + "display_width": 1024, + "display_height": 768, + "environment": "browser" # other possible values: "mac", "windows", "ubuntu" + }], + input=[ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "Check the latest OpenAI news on bing.com." + } + # Optional: include a screenshot of the initial state of the environment + # { + # type: "input_image", + # image_url: f"data:image/png;base64,{screenshot_base64}" + # } + ] + } + ], + reasoning={ + "summary": "concise", + }, + truncation="auto" +) + +print(response.output) +``` + + + + +1. Set up config.yaml + +```yaml showLineNumbers title="OpenAI Proxy Configuration" +model_list: + - model_name: openai/o1-pro + litellm_params: + model: openai/o1-pro + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start LiteLLM Proxy Server + +```bash title="Start LiteLLM Proxy Server" +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +3. Test it! + +```python showLineNumbers title="OpenAI Proxy Non-streaming Response" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.responses.create( + model="computer-use-preview", + tools=[{ + "type": "computer_use_preview", + "display_width": 1024, + "display_height": 768, + "environment": "browser" # other possible values: "mac", "windows", "ubuntu" + }], + input=[ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "Check the latest OpenAI news on bing.com." + } + # Optional: include a screenshot of the initial state of the environment + # { + # type: "input_image", + # image_url: f"data:image/png;base64,{screenshot_base64}" + # } + ] + } + ], + reasoning={ + "summary": "concise", + }, + truncation="auto" +) + +print(response) +``` + + + + diff --git a/docs/my-website/docs/providers/openai/text_to_speech.md b/docs/my-website/docs/providers/openai/text_to_speech.md new file mode 100644 index 00000000000..34cd0f069e6 --- /dev/null +++ b/docs/my-website/docs/providers/openai/text_to_speech.md @@ -0,0 +1,122 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# OpenAI - Text-to-speech + +## **LiteLLM Python SDK Usage** +### Quick Start + +```python +from pathlib import Path +from litellm import speech +import os + +os.environ["OPENAI_API_KEY"] = "sk-.." + +speech_file_path = Path(__file__).parent / "speech.mp3" +response = speech( + model="openai/tts-1", + voice="alloy", + input="the quick brown fox jumped over the lazy dogs", + ) +response.stream_to_file(speech_file_path) +``` + +### Async Usage + +```python +from litellm import aspeech +from pathlib import Path +import os, asyncio + +os.environ["OPENAI_API_KEY"] = "sk-.." + +async def test_async_speech(): + speech_file_path = Path(__file__).parent / "speech.mp3" + response = await litellm.aspeech( + model="openai/tts-1", + voice="alloy", + input="the quick brown fox jumped over the lazy dogs", + api_base=None, + api_key=None, + organization=None, + project=None, + max_retries=1, + timeout=600, + client=None, + optional_params={}, + ) + response.stream_to_file(speech_file_path) + +asyncio.run(test_async_speech()) +``` + +## **LiteLLM Proxy Usage** + +LiteLLM provides an openai-compatible `/audio/speech` endpoint for Text-to-speech calls. + +```bash +curl http://0.0.0.0:4000/v1/audio/speech \ + -H "Authorization: Bearer sk-1234" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "tts-1", + "input": "The quick brown fox jumped over the lazy dog.", + "voice": "alloy" + }' \ + --output speech.mp3 +``` + +**Setup** + +```bash +- model_name: tts + litellm_params: + model: openai/tts-1 + api_key: os.environ/OPENAI_API_KEY +``` + +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +## Supported Models + +| Model | Example | +|-------|-------------| +| tts-1 | speech(model="tts-1", voice="alloy", input="Hello, world!") | +| tts-1-hd | speech(model="tts-1-hd", voice="alloy", input="Hello, world!") | +| gpt-4o-mini-tts | speech(model="gpt-4o-mini-tts", voice="alloy", input="Hello, world!") | + + +## ✨ Enterprise LiteLLM Proxy - Set Max Request File Size + +Use this when you want to limit the file size for requests sent to `audio/transcriptions` + +```yaml +- model_name: whisper + litellm_params: + model: whisper-1 + api_key: sk-******* + max_file_size_mb: 0.00001 # 👈 max file size in MB (Set this intentionally very small for testing) + model_info: + mode: audio_transcription +``` + +Make a test Request with a valid file +```shell +curl --location 'http://localhost:4000/v1/audio/transcriptions' \ +--header 'Authorization: Bearer sk-1234' \ +--form 'file=@"/Users/ishaanjaffer/Github/litellm/tests/gettysburg.wav"' \ +--form 'model="whisper"' +``` + + +Expect to see the follow response + +```shell +{"error":{"message":"File size is too large. Please check your file size. Passed file size: 0.7392807006835938 MB. Max file size: 0.0001 MB","type":"bad_request","param":"file","code":500}}% +``` \ No newline at end of file diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index b328c805770..30887e9f60d 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -1284,11 +1284,18 @@ ModelResponse( -## Llama 3 API +## Meta/Llama API | Model Name | Function Call | |------------------|--------------------------------------| +| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` | +| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` | +| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` | | meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` | +| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` | +| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` | +| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` | +| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` | ### Usage diff --git a/docs/my-website/docs/proxy/management_client.md b/docs/my-website/docs/proxy/management_client.md new file mode 100644 index 00000000000..7bf09a07a71 --- /dev/null +++ b/docs/my-website/docs/proxy/management_client.md @@ -0,0 +1,265 @@ +# LiteLLM Proxy Client + +A Python client library for interacting with the LiteLLM proxy server. This client provides a clean, typed interface for managing models, keys, credentials, and making chat completions. + +## Installation + +```bash +pip install litellm +``` + +## Quick Start + +```python +from litellm.proxy.client import Client + +# Initialize the client +client = Client( + base_url="http://localhost:4000", # Your LiteLLM proxy server URL + api_key="sk-api-key" # Optional: API key for authentication +) + +# Make a chat completion request +response = client.chat.completions.create( + model="gpt-3.5-turbo", + messages=[ + {"role": "user", "content": "Hello, how are you?"} + ] +) +print(response.choices[0].message.content) +``` + +## Features + +The client is organized into several resource clients for different functionality: + +- `chat`: Chat completions +- `models`: Model management +- `model_groups`: Model group management +- `keys`: API key management +- `credentials`: Credential management +- `http`: Low-level HTTP client + +## Chat Completions + +Make chat completion requests to your LiteLLM proxy: + +```python +# Basic chat completion +response = client.chat.completions.create( + model="gpt-4", + messages=[ + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "What's the capital of France?"} + ] +) + +# Stream responses +for chunk in client.chat.completions.create( + model="gpt-4", + messages=[{"role": "user", "content": "Tell me a story"}], + stream=True +): + print(chunk.choices[0].delta.content or "", end="") +``` + +## Model Management + +Manage available models on your proxy: + +```python +# List available models +models = client.models.list() + +# Add a new model +client.models.add( + model_name="gpt-4", + litellm_params={ + "api_key": "your-openai-key", + "api_base": "https://api.openai.com/v1" + } +) + +# Delete a model +client.models.delete(model_name="gpt-4") +``` + +## API Key Management + +Manage virtual API keys: + +```python +# Generate a new API key +key = client.keys.generate( + models=["gpt-4", "gpt-3.5-turbo"], + aliases={"gpt4": "gpt-4"}, + duration="24h", + key_alias="my-key", + team_id="team123" +) + +# List all keys +keys = client.keys.list( + page=1, + size=10, + return_full_object=True +) + +# Delete keys +client.keys.delete( + keys=["sk-key1", "sk-key2"], + key_aliases=["alias1", "alias2"] +) +``` + +## Credential Management + +Manage model credentials: + +```python +# Create new credentials +client.credentials.create( + credential_name="azure1", + credential_info={"api_type": "azure"}, + credential_values={ + "api_key": "your-azure-key", + "api_base": "https://example.azure.openai.com" + } +) + +# List all credentials +credentials = client.credentials.list() + +# Get a specific credential +credential = client.credentials.get(credential_name="azure1") + +# Delete credentials +client.credentials.delete(credential_name="azure1") +``` + +## Model Groups + +Manage model groups for load balancing and fallbacks: + +```python +# Create a model group +client.model_groups.create( + name="gpt4-group", + models=[ + {"model_name": "gpt-4", "litellm_params": {"api_key": "key1"}}, + {"model_name": "gpt-4-backup", "litellm_params": {"api_key": "key2"}} + ] +) + +# List model groups +groups = client.model_groups.list() + +# Delete a model group +client.model_groups.delete(name="gpt4-group") +``` + +## Low-Level HTTP Client + +The client provides access to a low-level HTTP client for making direct requests +to the LiteLLM proxy server. This is useful when you need more control or when +working with endpoints that don't yet have a high-level interface. + +```python +# Access the HTTP client +client = Client( + base_url="http://localhost:4000", + api_key="sk-api-key" +) + +# Make a custom request +response = client.http.request( + method="POST", + uri="/health/test_connection", + json={ + "litellm_params": { + "model": "gpt-4", + "api_key": "your-api-key", + "api_base": "https://api.openai.com/v1" + }, + "mode": "chat" + } +) + +# The response is automatically parsed from JSON +print(response) +``` + +### HTTP Client Features + +- Automatic URL handling (handles trailing/leading slashes) +- Built-in authentication (adds Bearer token if `api_key` is provided) +- JSON request/response handling +- Configurable timeout (default: 30 seconds) +- Comprehensive error handling +- Support for custom headers and request parameters + +### HTTP Client `request` method parameters + +- `method`: HTTP method (GET, POST, PUT, DELETE, etc.) +- `uri`: URI path (will be appended to base_url) +- `data`: (optional) Data to send in the request body +- `json`: (optional) JSON data to send in the request body +- `headers`: (optional) Custom HTTP headers +- Additional keyword arguments are passed to the underlying requests library + +## Error Handling + +The client provides clear error handling with custom exceptions: + +```python +from litellm.proxy.client.exceptions import UnauthorizedError + +try: + response = client.chat.completions.create( + model="gpt-4", + messages=[{"role": "user", "content": "Hello"}] + ) +except UnauthorizedError as e: + print("Authentication failed:", e) +except Exception as e: + print("Request failed:", e) +``` + +## Advanced Usage + +### Request Customization + +All methods support returning the raw request object for inspection or modification: + +```python +# Get the prepared request without sending it +request = client.models.list(return_request=True) +print(request.method) # GET +print(request.url) # http://localhost:8000/models +print(request.headers) # {'Content-Type': 'application/json', ...} +``` + +### Pagination + +Methods that return lists support pagination: + +```python +# Get the first page of keys +page1 = client.keys.list(page=1, size=10) + +# Get the second page +page2 = client.keys.list(page=2, size=10) +``` + +### Filtering + +Many list methods support filtering: + +```python +# Filter keys by user and team +keys = client.keys.list( + user_id="user123", + team_id="team456", + include_team_keys=True +) +``` \ No newline at end of file diff --git a/docs/my-website/docs/proxy/release_cycle.md b/docs/my-website/docs/proxy/release_cycle.md index c5782087f21..10dd6d8b3c5 100644 --- a/docs/my-website/docs/proxy/release_cycle.md +++ b/docs/my-website/docs/proxy/release_cycle.md @@ -18,3 +18,8 @@ Follow our release notes [here](https://github.com/BerriAI/litellm/releases). Stable releases come out every week (typically Sunday) +### What is considered a 'minor' bump vs. 'patch' bump? + +- 'patch' bumps: extremely minor addition that doesn't affect any existing functionality or add any user-facing features. (e.g. a 'created_at' column in a database table) +- 'minor' bumps: add a new feature or a new database table that is backward compatible. +- 'major' bumps: break backward compatibility. \ No newline at end of file diff --git a/docs/my-website/docs/proxy/users.md b/docs/my-website/docs/proxy/users.md index 92ea73b9d2e..b4457b8d553 100644 --- a/docs/my-website/docs/proxy/users.md +++ b/docs/my-website/docs/proxy/users.md @@ -786,6 +786,17 @@ Expected Response: } } ``` + +### [BETA] Multi-instance rate limiting + +Enable multi-instance rate limiting with the env var `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"` + +Changes: +- This moves to using async_increment instead of async_set_cache when updating current requests/tokens. +- The in-memory cache is synced with redis every 0.01s, to avoid calling redis for every request. +- In testing, this was found to be 2x faster than the previous implementation, and reduced drift between expected and actual fails to at most 10 requests at high-traffic (100 RPS across 3 instances). + + ## Grant Access to new model Use model access groups to give users access to select models, and add new ones to it over time (e.g. mistral, llama-2, etc.). diff --git a/docs/my-website/release_notes/v1.68.0-stable/index.md b/docs/my-website/release_notes/v1.68.0-stable/index.md new file mode 100644 index 00000000000..c47cfda2348 --- /dev/null +++ b/docs/my-website/release_notes/v1.68.0-stable/index.md @@ -0,0 +1,136 @@ +--- +title: v1.68.0-stable +slug: v1.68.0-stable +date: 2025-05-03T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8 + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg + +hide_table_of_contents: false +--- +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + + + +## Deploy this version + + + + +``` showLineNumbers title="docker run litellm" +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.68.0-stable +``` + + + + +``` showLineNumbers title="pip install litellm" +pip install litellm==1.68.0.post1 +``` + + + +## New Models / Updated Models +- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - Handle more json schema - openapi schema conversion edge cases [PR](https://github.com/BerriAI/litellm/pull/10351) + - Tool calls - return ‘finish_reason=“tool_calls”’ on gemini tool calling response [PR](https://github.com/BerriAI/litellm/pull/10485) +- **[VertexAI](../../docs/providers/vertex#metallama-api)** + - Meta/llama-4 model support [PR](https://github.com/BerriAI/litellm/pull/10492) + - Meta/llama3 - handle tool call result in content [PR](https://github.com/BerriAI/litellm/pull/10492) + - Meta/* - return ‘finish_reason=“tool_calls”’ on tool calling response [PR](https://github.com/BerriAI/litellm/pull/10492) +- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)** + - [Image Generation](../../docs/providers/bedrock#image-generation) - Support new ‘stable-image-core’ models - [PR](https://github.com/BerriAI/litellm/pull/10351) + - [Knowledge Bases](../../docs/completion/knowledgebase) - support using Bedrock knowledge bases with `/chat/completions` [PR](https://github.com/BerriAI/litellm/pull/10413) + - [Anthropic](../../docs/providers/bedrock#litellm-proxy-usage) - add ‘supports_pdf_input’ for claude-3.7-bedrock models [PR](https://github.com/BerriAI/litellm/pull/9917), [Get Started](../../docs/completion/document_understanding#checking-if-a-model-supports-pdf-input) +- **[OpenAI](../../docs/providers/openai)** + - Support OPENAI_BASE_URL in addition to OPENAI_API_BASE [PR](https://github.com/BerriAI/litellm/pull/10423) + - Correctly re-raise 504 timeout errors [PR](https://github.com/BerriAI/litellm/pull/10462) + - Native Gpt-4o-mini-tts support [PR](https://github.com/BerriAI/litellm/pull/10462) +- 🆕 **[LlamaFile](../../docs/providers/llamafile)** provider [PR](https://github.com/BerriAI/litellm/pull/10482) + + + +## LLM API Endpoints +- **[Response API](../../docs/response_api)** + - Fix for handling multi turn sessions [PR](https://github.com/BerriAI/litellm/pull/10415) +- **[Embeddings](../../docs/embedding/supported_embedding)** + - Caching fixes - [PR](https://github.com/BerriAI/litellm/pull/10424) + - handle str -> list cache + - Return usage tokens for cache hit + - Combine usage tokens on partial cache hits +- 🆕 **[Vector Stores](../../docs/completion/knowledgebase)** + - Allow defining Vector Store Configs - [PR](https://github.com/BerriAI/litellm/pull/10448) + - New StandardLoggingPayload field for requests made when a vector store is used - [PR](https://github.com/BerriAI/litellm/pull/10509) + - Show Vector Store / KB Request on LiteLLM Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10514) + - Allow using vector store in OpenAI API spec with tools - [PR](https://github.com/BerriAI/litellm/pull/10516) +- **[MCP](../../docs/mcp)** + - Ensure Non-Admin virtual keys can access /mcp routes - [PR](https://github.com/BerriAI/litellm/pull/10473) + + **Note:** Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. Follow [here](https://github.com/BerriAI/litellm/discussions/9891) for updates. +- **Moderations** + - Add logging callback support for `/moderations` API - [PR](https://github.com/BerriAI/litellm/pull/10390) + + +## Spend Tracking / Budget Improvements +- **[OpenAI](../../docs/providers/openai)** + - [computer-use-preview](../../docs/providers/openai/responses_api#computer-use) cost tracking / pricing [PR](https://github.com/BerriAI/litellm/pull/10422) + - [gpt-4o-mini-tts](../../docs/providers/openai/text_to_speech) input cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10462) +- **[Fireworks AI](../../docs/providers/fireworks_ai)** - pricing updates - new `0-4b` model pricing tier + llama4 model pricing +- **[Budgets](../../docs/proxy/users#set-budgets)** + - [Budget resets](../../docs/proxy/users#reset-budgets) now happen as start of day/week/month - [PR](https://github.com/BerriAI/litellm/pull/10333) + - Trigger [Soft Budget Alerts](../../docs/proxy/alerting#soft-budget-alerts-for-virtual-keys) When Key Crosses Threshold - [PR](https://github.com/BerriAI/litellm/pull/10491) +- **[Token Counting](../../docs/completion/token_usage#3-token_counter)** + - Rewrite of token_counter() function to handle to prevent undercounting tokens - [PR](https://github.com/BerriAI/litellm/pull/10409) + + +## Management Endpoints / UI +- **Virtual Keys** + - Fix filtering on key alias - [PR](https://github.com/BerriAI/litellm/pull/10455) + - Support global filtering on keys - [PR](https://github.com/BerriAI/litellm/pull/10455) + - Pagination - fix clicking on next/back buttons on table - [PR](https://github.com/BerriAI/litellm/pull/10528) +- **Models** + - Triton - Support adding model/provider on UI - [PR](https://github.com/BerriAI/litellm/pull/10456) + - VertexAI - Fix adding vertex models with reusable credentials - [PR](https://github.com/BerriAI/litellm/pull/10528) + - LLM Credentials - show existing credentials for easy editing - [PR](https://github.com/BerriAI/litellm/pull/10519) +- **Teams** + - Allow reassigning team to other org - [PR](https://github.com/BerriAI/litellm/pull/10527) +- **Organizations** + - Fix showing org budget on table - [PR](https://github.com/BerriAI/litellm/pull/10528) + + + +## Logging / Guardrail Integrations +- **[Langsmith](../../docs/observability/langsmith_integration)** + - Respect [langsmith_batch_size](../../docs/observability/langsmith_integration#local-testing---control-batch-size) param - [PR](https://github.com/BerriAI/litellm/pull/10411) + +## Performance / Loadbalancing / Reliability improvements +- **[Redis](../../docs/proxy/caching)** + - Ensure all redis queues are periodically flushed, this fixes an issue where redis queue size was growing indefinitely when request tags were used - [PR](https://github.com/BerriAI/litellm/pull/10393) +- **[Rate Limits](../../docs/proxy/users#set-rate-limit)** + - [Multi-instance rate limiting](../../docs/proxy/users#beta-multi-instance-rate-limiting) support across keys/teams/users/customers - [PR](https://github.com/BerriAI/litellm/pull/10458), [PR](https://github.com/BerriAI/litellm/pull/10497), [PR](https://github.com/BerriAI/litellm/pull/10500) +- **[Azure OpenAI OIDC](../../docs/providers/azure#entra-id---use-azure_ad_token)** + - allow using litellm defined params for [OIDC Auth](../../docs/providers/azure#entra-id---use-azure_ad_token) - [PR](https://github.com/BerriAI/litellm/pull/10394) + + +## General Proxy Improvements +- **Security** + - Allow [blocking web crawlers](../../docs/proxy/enterprise#blocking-web-crawlers) - [PR](https://github.com/BerriAI/litellm/pull/10420) +- **Auth** + - Support [`x-litellm-api-key` header param by default](../../docs/pass_through/vertex_ai#use-with-virtual-keys), this fixes an issue from the prior release where `x-litellm-api-key` was not being used on vertex ai passthrough requests - [PR](https://github.com/BerriAI/litellm/pull/10392) + - Allow key at max budget to call non-llm api endpoints - [PR](https://github.com/BerriAI/litellm/pull/10392) +- 🆕 **[Python Client Library](../../docs/proxy/management_client) for LiteLLM Proxy management endpoints** + - Initial PR - [PR](https://github.com/BerriAI/litellm/pull/10445) + - Support for doing HTTP requests - [PR](https://github.com/BerriAI/litellm/pull/10452) +- **Dependencies** + - Don’t require uvloop for windows - [PR](https://github.com/BerriAI/litellm/pull/10483) diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 1e97e76b150..9f4c392d41f 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -61,6 +61,7 @@ const sidebars = { href: "https://litellm-api.up.railway.app/", }, "proxy/enterprise", + "proxy/management_client", { type: "category", label: "Making LLM Requests", @@ -190,7 +191,15 @@ const sidebars = { slug: "/providers", }, items: [ - "providers/openai", + { + type: "category", + label: "OpenAI", + items: [ + "providers/openai", + "providers/openai/responses_api", + "providers/openai/text_to_speech", + ] + }, "providers/text_completion_openai", "providers/openai_compatible", "providers/azure", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c148f04e336..efef685d696 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6323,7 +6323,7 @@ "supported_modalities": ["text", "image"], "supported_output_modalities": ["text", "code"] }, - "vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": { + "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": { "max_tokens": 10e6, "max_input_tokens": 10e6, "max_output_tokens": 10e6, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c148f04e336..efef685d696 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -6323,7 +6323,7 @@ "supported_modalities": ["text", "image"], "supported_output_modalities": ["text", "code"] }, - "vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": { + "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": { "max_tokens": 10e6, "max_input_tokens": 10e6, "max_output_tokens": 10e6,