From af3c49def3500c39d00f90c3c847c7dd8ab7daaf Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 21 Jun 2025 11:38:37 -0700 Subject: [PATCH] [Docs] [Pre-Release] v1.73.0-stable (#11950) * draft 1.73.0 * fixes * docs pass through * clean up * docs fix * docs fixes * docs fix * docs - Logging / Guardrails Integrations * v1.73.0 * docs fixes * fixes v2 heath check * docs fix * docs vertex img gen * docs fix * docs link * docs azure responses * azure codex * docs fixes * fixes release notes * fix docs * docs pre release --- docs/my-website/docs/providers/azure/azure.md | 125 +------ .../docs/providers/azure/azure_responses.md | 235 ++++++++++++++ .../docs/providers/openai/responses_api.md | 2 +- docs/my-website/docs/providers/vertex.md | 39 --- .../my-website/docs/providers/vertex_image.md | 83 +++++ .../release_notes/v1.72.6-stable/index.md | 8 - .../release_notes/v1.73.0-stable/index.md | 306 ++++++++++++++++++ docs/my-website/sidebars.js | 10 +- 8 files changed, 635 insertions(+), 173 deletions(-) create mode 100644 docs/my-website/docs/providers/azure/azure_responses.md create mode 100644 docs/my-website/docs/providers/vertex_image.md create mode 100644 docs/my-website/release_notes/v1.73.0-stable/index.md diff --git a/docs/my-website/docs/providers/azure/azure.md b/docs/my-website/docs/providers/azure/azure.md index 065654df6f6..5317b744abe 100644 --- a/docs/my-website/docs/providers/azure/azure.md +++ b/docs/my-website/docs/providers/azure/azure.md @@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem'; |-------|-------| | Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series | | Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#azure-o-series-models) | -| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) | +| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/responses`](./azure_responses), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) | | Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview) ## API Keys, Params @@ -1003,129 +1003,6 @@ Expected Response: {"data":[{"id":"batch_R3V...} ``` - -## **Azure Responses API** - -| Property | Details | -|-------|-------| -| Description | Azure OpenAI Responses API | -| `custom_llm_provider` on LiteLLM | `azure/` | -| Supported Operations | `/v1/responses`| -| Azure OpenAI Responses API | [Azure OpenAI Responses API ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/responses?tabs=python-secure) | -| Cost Tracking, Logging Support | ✅ LiteLLM will log, track cost for Responses API Requests | -| Supported OpenAI Params | ✅ All OpenAI params are supported, [See here](https://github.com/BerriAI/litellm/blob/0717369ae6969882d149933da48eeb8ab0e691bd/litellm/llms/openai/responses/transformation.py#L23) | - -## Usage - -## Create a model response - - - - -#### Non-streaming - -```python showLineNumbers title="Azure Responses API" -import litellm - -# Non-streaming response -response = litellm.responses( - model="azure/o1-pro", - input="Tell me a three sentence bedtime story about a unicorn.", - max_output_tokens=100, - api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"), - api_base="https://litellm8397336933.openai.azure.com/", - api_version="2023-03-15-preview", -) - -print(response) -``` - -#### Streaming -```python showLineNumbers title="Azure Responses API" -import litellm - -# Streaming response -response = litellm.responses( - model="azure/o1-pro", - input="Tell me a three sentence bedtime story about a unicorn.", - stream=True, - api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"), - api_base="https://litellm8397336933.openai.azure.com/", - api_version="2023-03-15-preview", -) - -for event in response: - print(event) -``` - - - - -First, add this to your litellm proxy config.yaml: -```yaml showLineNumbers title="Azure Responses API" -model_list: - - model_name: o1-pro - litellm_params: - model: azure/o1-pro - api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY - api_base: https://litellm8397336933.openai.azure.com/ - api_version: 2023-03-15-preview -``` - -Start your LiteLLM proxy: -```bash -litellm --config /path/to/config.yaml - -# RUNNING on http://0.0.0.0:4000 -``` - -Then use the OpenAI SDK pointed to your proxy: - -#### Non-streaming -```python showLineNumbers -from openai import OpenAI - -# Initialize client with your proxy URL -client = OpenAI( - base_url="http://localhost:4000", # Your proxy URL - api_key="your-api-key" # Your proxy API key -) - -# Non-streaming response -response = client.responses.create( - model="o1-pro", - input="Tell me a three sentence bedtime story about a unicorn." -) - -print(response) -``` - -#### Streaming -```python showLineNumbers -from openai import OpenAI - -# Initialize client with your proxy URL -client = OpenAI( - base_url="http://localhost:4000", # Your proxy URL - api_key="your-api-key" # Your proxy API key -) - -# Streaming response -response = client.responses.create( - model="o1-pro", - input="Tell me a three sentence bedtime story about a unicorn.", - stream=True -) - -for event in response: - print(event) -``` - - - - - - ## Advanced ### Azure API Load-Balancing diff --git a/docs/my-website/docs/providers/azure/azure_responses.md b/docs/my-website/docs/providers/azure/azure_responses.md new file mode 100644 index 00000000000..b17ef8a2853 --- /dev/null +++ b/docs/my-website/docs/providers/azure/azure_responses.md @@ -0,0 +1,235 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Azure Responses API + +| Property | Details | +|-------|-------| +| Description | Azure OpenAI Responses API | +| `custom_llm_provider` on LiteLLM | `azure/` | +| Supported Operations | `/v1/responses`| +| Azure OpenAI Responses API | [Azure OpenAI Responses API ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/responses?tabs=python-secure) | +| Cost Tracking, Logging Support | ✅ LiteLLM will log, track cost for Responses API Requests | +| Supported OpenAI Params | ✅ All OpenAI params are supported, [See here](https://github.com/BerriAI/litellm/blob/0717369ae6969882d149933da48eeb8ab0e691bd/litellm/llms/openai/responses/transformation.py#L23) | + +## Usage + +## Create a model response + + + + +#### Non-streaming + +```python showLineNumbers title="Azure Responses API" +import litellm + +# Non-streaming response +response = litellm.responses( + model="azure/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100, + api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"), + api_base="https://litellm8397336933.openai.azure.com/", + api_version="2023-03-15-preview", +) + +print(response) +``` + +#### Streaming +```python showLineNumbers title="Azure Responses API" +import litellm + +# Streaming response +response = litellm.responses( + model="azure/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True, + api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"), + api_base="https://litellm8397336933.openai.azure.com/", + api_version="2023-03-15-preview", +) + +for event in response: + print(event) +``` + + + + +First, add this to your litellm proxy config.yaml: +```yaml showLineNumbers title="Azure Responses API" +model_list: + - model_name: o1-pro + litellm_params: + model: azure/o1-pro + api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY + api_base: https://litellm8397336933.openai.azure.com/ + api_version: 2023-03-15-preview +``` + +Start your LiteLLM proxy: +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +Then use the OpenAI SDK pointed to your proxy: + +#### Non-streaming +```python showLineNumbers +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.responses.create( + model="o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +print(response) +``` + +#### Streaming +```python showLineNumbers +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Streaming response +response = client.responses.create( + model="o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True +) + +for event in response: + print(event) +``` + + + + +## Azure Codex Models + +Codex models use Azure's new [/v1/preview API](https://learn.microsoft.com/en-us/azure/ai-services/openai/api-version-lifecycle?tabs=key#next-generation-api) which provides ongoing access to the latest features with no need to update `api-version` each month. + +**LiteLLM will send your requests to the `/v1/preview` endpoint when you set `api_version="preview"`.** + + + + +#### Non-streaming + +```python showLineNumbers title="Azure Codex Models" +import litellm + +# Non-streaming response with Codex models +response = litellm.responses( + model="azure/codex-mini", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100, + api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"), + api_base="https://litellm8397336933.openai.azure.com", + api_version="preview", # 👈 key difference +) + +print(response) +``` + +#### Streaming +```python showLineNumbers title="Azure Codex Models" +import litellm + +# Streaming response with Codex models +response = litellm.responses( + model="azure/codex-mini", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True, + api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"), + api_base="https://litellm8397336933.openai.azure.com", + api_version="preview", # 👈 key difference +) + +for event in response: + print(event) +``` + + + + +First, add this to your litellm proxy config.yaml: +```yaml showLineNumbers title="Azure Codex Models" +model_list: + - model_name: codex-mini + litellm_params: + model: azure/codex-mini + api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY + api_base: https://litellm8397336933.openai.azure.com + api_version: preview # 👈 key difference +``` + +Start your LiteLLM proxy: +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +Then use the OpenAI SDK pointed to your proxy: + +#### Non-streaming +```python showLineNumbers +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.responses.create( + model="codex-mini", + input="Tell me a three sentence bedtime story about a unicorn." +) + +print(response) +``` + +#### Streaming +```python showLineNumbers +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Streaming response +response = client.responses.create( + model="codex-mini", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True +) + +for event in response: + print(event) +``` + + + + diff --git a/docs/my-website/docs/providers/openai/responses_api.md b/docs/my-website/docs/providers/openai/responses_api.md index 3dcf3096159..db2d781ca15 100644 --- a/docs/my-website/docs/providers/openai/responses_api.md +++ b/docs/my-website/docs/providers/openai/responses_api.md @@ -207,7 +207,7 @@ print(delete_response) |----------|---------------------| | `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) | -### Reusable Prompts +## Reusable Prompts Use the `prompt` parameter to reference a stored prompt template and optionally supply variables. diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index 233e3c6480c..21c17933b1b 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -2802,45 +2802,6 @@ print(response) -## **Image Generation Models** - -Usage - -```python -response = await litellm.aimage_generation( - prompt="An olympic size swimming pool", - model="vertex_ai/imagegeneration@006", - vertex_ai_project="adroit-crow-413218", - vertex_ai_location="us-central1", -) -``` - -**Generating multiple images** - -Use the `n` parameter to pass how many images you want generated -```python -response = await litellm.aimage_generation( - prompt="An olympic size swimming pool", - model="vertex_ai/imagegeneration@006", - vertex_ai_project="adroit-crow-413218", - vertex_ai_location="us-central1", - n=1, -) -``` - -### Supported Image Generation Models - -| Model Name | FUsage | -|------------------------------|--------------------------------------------------------------| -| `imagen-3.0-generate-001` | `litellm.image_generation('vertex_ai/imagen-3.0-generate-001', prompt)` | -| `imagen-3.0-fast-generate-001` | `litellm.image_generation('vertex_ai/imagen-3.0-fast-generate-001', prompt)` | -| `imagegeneration@006` | `litellm.image_generation('vertex_ai/imagegeneration@006', prompt)` | -| `imagegeneration@005` | `litellm.image_generation('vertex_ai/imagegeneration@005', prompt)` | -| `imagegeneration@002` | `litellm.image_generation('vertex_ai/imagegeneration@002', prompt)` | - - - - ## **Gemini TTS (Text-to-Speech) Audio Output** :::info diff --git a/docs/my-website/docs/providers/vertex_image.md b/docs/my-website/docs/providers/vertex_image.md new file mode 100644 index 00000000000..2434c3a9a57 --- /dev/null +++ b/docs/my-website/docs/providers/vertex_image.md @@ -0,0 +1,83 @@ +# Vertex AI Image Generation + +Vertex AI Image Generation uses Google's Imagen models to generate high-quality images from text descriptions. + +| Property | Details | +|----------|---------| +| Description | Vertex AI Image Generation uses Google's Imagen models to generate high-quality images from text descriptions. | +| Provider Route on LiteLLM | `vertex_ai/` | +| Provider Doc | [Google Cloud Vertex AI Image Generation ↗](https://cloud.google.com/vertex-ai/docs/generative-ai/image/generate-images) | + +## Quick Start + +### LiteLLM Python SDK + +```python showLineNumbers title="Basic Image Generation" +import litellm + +# Generate a single image +response = await litellm.aimage_generation( + prompt="An olympic size swimming pool with crystal clear water and modern architecture", + model="vertex_ai/imagen-4.0-generate-preview-06-06", + vertex_ai_project="your-project-id", + vertex_ai_location="us-central1", +) + +print(response.data[0].url) +``` + +### LiteLLM Proxy + +#### 1. Configure your config.yaml + +```yaml showLineNumbers title="Vertex AI Image Generation Configuration" +model_list: + - model_name: vertex-imagen + litellm_params: + model: vertex_ai/imagen-4.0-generate-preview-06-06 + vertex_ai_project: "your-project-id" + vertex_ai_location: "us-central1" + vertex_ai_credentials: "path/to/service-account.json" # Optional if using environment auth +``` + +#### 2. Start LiteLLM Proxy Server + +```bash title="Start LiteLLM Proxy Server" +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +#### 3. Make requests with OpenAI Python SDK + +```python showLineNumbers title="Basic Image Generation via Proxy" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-proxy-api-key" # Your proxy API key +) + +# Generate image +response = client.images.generate( + model="vertex-imagen", + prompt="An olympic size swimming pool with crystal clear water and modern architecture", +) + +print(response.data[0].url) +``` + +## Supported Models + + +:::tip + +**We support ALL Vertex AI Image Generation models, just set `model=vertex_ai/` as a prefix when sending litellm requests** + +::: + +LiteLLM supports all Vertex AI Imagen models available through Google Cloud. + +For the complete and up-to-date list of supported models, visit: [https://models.litellm.ai/](https://models.litellm.ai/) + diff --git a/docs/my-website/release_notes/v1.72.6-stable/index.md b/docs/my-website/release_notes/v1.72.6-stable/index.md index f5c8f06c7bf..5603548364f 100644 --- a/docs/my-website/release_notes/v1.72.6-stable/index.md +++ b/docs/my-website/release_notes/v1.72.6-stable/index.md @@ -19,14 +19,6 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; - -:::info - -This is a pre-release version. - -The production version will be released on Wednesday. - -::: ## Deploy this version diff --git a/docs/my-website/release_notes/v1.73.0-stable/index.md b/docs/my-website/release_notes/v1.73.0-stable/index.md new file mode 100644 index 00000000000..a92371f55b0 --- /dev/null +++ b/docs/my-website/release_notes/v1.73.0-stable/index.md @@ -0,0 +1,306 @@ +--- +title: "[Pre-Release] v1.73.0-stable" +slug: "v1-73-0-stable" +date: 2025-06-21T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg + +hide_table_of_contents: false +--- + +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + + +:::info + +This is a pre-release version. + +The production version will be released on Wednesday. + +::: +## Deploy this version + + + + +``` showLineNumbers title="docker run litellm" +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +ghcr.io/berriai/litellm:main-v1.73.0.rc +``` + + + + +``` showLineNumbers title="pip install litellm" +pip install litellm==1.73.0.rc +``` + + + + + +## TLDR + + +* **Why Upgrade** + - Passthrough Endpoints v2: Enhanced support for subroutes and custom cost tracking for passthrough endpoints. + - Health Check Dashboard: New frontend UI for monitoring model health and status. +* **Who Should Read** + - Teams using **Passthrough Endpoints** + - Teams using **Health Check Dashboard** for models + - Teams using **Claude Code** with LiteLLM +* **Risk of Upgrade** + - **Low** + - No major breaking changes to existing functionality. + + +--- + +## Key Highlights + + +### Passthrough Endpoints v2 + +This release brings support for subroutes and custom cost tracking in passthrough endpoints. When teams use external APIs through LiteLLM, they can now add one passthrough route (e.g. `/bria`) and access multiple endpoints through subroutes like `/bria/text-to-image/base`, `/bria/enhance_image` - all with custom costs per request. + +This is great for API providers like [Bria AI](https://platform.bria.ai/) with multiple endpoints where you want one unified route instead of managing separate passthrough endpoints. + +For Proxy Admins, this means adding one passthrough route and having developers access all subroutes (image generation, editing, etc.). For developers, this means simplified endpoint access with transparent cost visibility. + +[Learn more about Passthrough Endpoints](../../docs/pass_through) + + +### v2 Health Checks + +This release introduces v2 of the health check page with asynchronous result loading + storing the results of the last health check. Previously, we waited for all endpoints to respond before showing results. + +Now Proxy Admins see incremental health check results in real-time, making it easier to identify problematic models while confirming that the overall system is functioning properly. + + + +--- + + +## New / Updated Models + +### Pricing / Context Window Updates + +| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Type | +| ----------- | -------------------------------------- | -------------- | ------------------- | -------------------- | ---- | +| Google VertexAI | `vertex_ai/imagen-4` | N/A | Image Generation | Image Generation | New | +| Google VertexAI | `vertex_ai/imagen-4-preview` | N/A | Image Generation | Image Generation | New | +| Gemini | `gemini-2.5-pro` | 2M | $1.25 | $5.00 | New | +| Gemini | `gemini-2.5-flash-lite` | 1M | $0.075 | $0.30 | New | +| OpenRouter | Various models | Updated | Updated | Updated | Updated | +| Azure | `azure/o3` | 200k | $2.00 | $8.00 | Updated | +| Azure | `azure/o3-pro` | 200k | $2.00 | $8.00 | Updated | +| Azure OpenAI | Azure Codex Models | Various | Various | Various | New | + +## Updated Models + +#### Features +- **[Azure](../../docs/providers/azure)** + - Support for new /v1 preview Azure OpenAI API - [PR](https://github.com/BerriAI/litellm/pull/11934), [Get Started](../../docs/providers/azure/azure_responses#azure-codex-models) + - Add Azure Codex Models support - [PR](https://github.com/BerriAI/litellm/pull/11934), [Get Started](../../docs/providers/azure/azure_responses#azure-codex-models) + - Make Azure AD scope configurable - [PR](https://github.com/BerriAI/litellm/pull/11621) + - Handle more GPT custom naming patterns - [PR](https://github.com/BerriAI/litellm/pull/11914) + - Update o3 pricing to match OpenAI pricing - [PR](https://github.com/BerriAI/litellm/pull/11937) +- **[VertexAI](../../docs/providers/vertex)** + - Add Vertex Imagen-4 models - [PR](https://github.com/BerriAI/litellm/pull/11767), [Get Started](../../docs/providers/vertex_image) + - Anthropic streaming passthrough cost tracking - [PR](https://github.com/BerriAI/litellm/pull/11734) +- **[Gemini](../../docs/providers/gemini)** + - Working Gemini TTS support via `/v1/speech` endpoint - [PR](https://github.com/BerriAI/litellm/pull/11832) + - Fix gemini 2.5 flash config - [PR](https://github.com/BerriAI/litellm/pull/11830) + - Add missing `flash-2.5-flash-lite` model and fix pricing - [PR](https://github.com/BerriAI/litellm/pull/11901) + - Mark all gemini-2.5 models as supporting PDF input - [PR](https://github.com/BerriAI/litellm/pull/11907) + - Add `gemini-2.5-pro` with reasoning support - [PR](https://github.com/BerriAI/litellm/pull/11927) +- **[AWS Bedrock](../../docs/providers/bedrock)** + - AWS credentials no longer mandatory - [PR](https://github.com/BerriAI/litellm/pull/11765) + - Add AWS Bedrock profiles for APAC region - [PR](https://github.com/BerriAI/litellm/pull/11883) + - Fix AWS Bedrock Claude tool call index - [PR](https://github.com/BerriAI/litellm/pull/11842) + - Handle base64 file data with `qs:..` prefix - [PR](https://github.com/BerriAI/litellm/pull/11908) + - Add Mistral Small to BEDROCK_CONVERSE_MODELS - [PR](https://github.com/BerriAI/litellm/pull/11760) +- **[Mistral](../../docs/providers/mistral)** + - Enhance Mistral API with parallel tool calls support - [PR](https://github.com/BerriAI/litellm/pull/11770) +- **[Meta Llama API](../../docs/providers/meta_llama)** + - Enable tool calling for meta_llama models - [PR](https://github.com/BerriAI/litellm/pull/11895) +- **[Volcengine](../../docs/providers/volcengine)** + - Add thinking parameter support - [PR](https://github.com/BerriAI/litellm/pull/11914) + + +### Bugs + +- **[VertexAI](../../docs/providers/vertex)** + - Handle missing tokenCount in promptTokensDetails - [PR](https://github.com/BerriAI/litellm/pull/11896) + - Fix vertex AI claude thinking params - [PR](https://github.com/BerriAI/litellm/pull/11796) +- **[Gemini](../../docs/providers/gemini)** + - Fix web search error with responses API - [PR](https://github.com/BerriAI/litellm/pull/11894), [Get Started](../../docs/completion/web_search#responses-litellmresponses) +- **[Custom LLM](../../docs/providers/custom_llm_server)** + - Set anthropic custom LLM provider property - [PR](https://github.com/BerriAI/litellm/pull/11907) +- **[Anthropic](../../docs/providers/anthropic)** + - Bump anthropic package version - [PR](https://github.com/BerriAI/litellm/pull/11851) +- **[Ollama](../../docs/providers/ollama)** + - Update ollama_embeddings to work on sync API - [PR](https://github.com/BerriAI/litellm/pull/11746) + - Fix response_format not working - [PR](https://github.com/BerriAI/litellm/pull/11880) + +--- + +## LLM API Endpoints + +#### Features +- **[Responses API](../../docs/response_api)** + - Day-0 support for OpenAI re-usable prompts Responses API - [PR](https://github.com/BerriAI/litellm/pull/11782), [Get Started](../../docs/providers/openai/responses_api#reusable-prompts) + - Support passing image URLs in Completion-to-Responses bridge - [PR](https://github.com/BerriAI/litellm/pull/11833) +- **[MCP Gateway](../../docs/mcp)** + - Add Allowed MCPs to Creating/Editing Organizations - [PR](https://github.com/BerriAI/litellm/pull/11893), [Get Started](../../docs/mcp#-mcp-permission-management) + - Allow connecting to MCP with authentication headers - [PR](https://github.com/BerriAI/litellm/pull/11891), [Get Started](../../docs/mcp#using-your-mcp-with-client-side-credentials) +- **[Speech API](../../docs/speech)** + - Working Gemini TTS support via OpenAI's `/v1/speech` endpoint - [PR](https://github.com/BerriAI/litellm/pull/11832) +- **[Passthrough Endpoints](../../docs/pass_through/custom_routes)** + - Add support for subroutes for passthrough endpoints - [PR](https://github.com/BerriAI/litellm/pull/11827) + - Support for setting custom cost per passthrough request - [PR](https://github.com/BerriAI/litellm/pull/11870) + - Ensure "Request" is tracked for passthrough requests on LiteLLM Proxy - [PR](https://github.com/BerriAI/litellm/pull/11873) + - Add V2 Passthrough endpoints on UI - [PR](https://github.com/BerriAI/litellm/pull/11905) + - Move passthrough endpoints under Models + Endpoints in UI - [PR](https://github.com/BerriAI/litellm/pull/11871) + - QA improvements for adding passthrough endpoints - [PR](https://github.com/BerriAI/litellm/pull/11909), [PR](https://github.com/BerriAI/litellm/pull/11939) +- **[Models API](../../docs/completion/model_alias)** + - Allow `/models` to return correct models for custom wildcard prefixes - [PR](https://github.com/BerriAI/litellm/pull/11784) + +#### Bugs + +- **[Messages API](../../docs/anthropic_unified)** + - Fix `/v1/messages` endpoint always using us-central1 with vertex_ai-anthropic models - [PR](https://github.com/BerriAI/litellm/pull/11831) + - Fix model_group tracking for `/v1/messages` and `/moderations` - [PR](https://github.com/BerriAI/litellm/pull/11933) + - Fix cost tracking and logging via `/v1/messages` API when using Claude Code - [PR](https://github.com/BerriAI/litellm/pull/11928) +- **[MCP Gateway](../../docs/mcp)** + - Fix using MCPs defined on config.yaml - [PR](https://github.com/BerriAI/litellm/pull/11824) +- **[Chat Completion API](../../docs/completion/input)** + - Allow dict for tool_choice argument in acompletion - [PR](https://github.com/BerriAI/litellm/pull/11860) +- **[Passthrough Endpoints](../../docs/pass_through/langfuse)** + - Don't log request to Langfuse passthrough on Langfuse - [PR](https://github.com/BerriAI/litellm/pull/11768) + +--- + +## Spend Tracking + +#### Features +- **[User Agent Tracking](../../docs/proxy/cost_tracking)** + - Automatically track spend by user agent (allows cost tracking for Claude Code) - [PR](https://github.com/BerriAI/litellm/pull/11781) + - Add user agent tags in spend logs payload - [PR](https://github.com/BerriAI/litellm/pull/11872) +- **[Tag Management](../../docs/proxy/cost_tracking)** + - Support adding public model names in tag management - [PR](https://github.com/BerriAI/litellm/pull/11908) + +--- + +## Management Endpoints / UI + +#### Features +- **Test Key Page** + - Allow testing `/v1/messages` on the Test Key Page - [PR](https://github.com/BerriAI/litellm/pull/11930) +- **[SSO](../../docs/proxy/sso)** + - Allow passing additional headers - [PR](https://github.com/BerriAI/litellm/pull/11781) +- **[JWT Auth](../../docs/proxy/jwt_auth)** + - Correctly return user email - [PR](https://github.com/BerriAI/litellm/pull/11783) +- **[Model Management](../../docs/proxy/model_management)** + - Allow editing model access group for existing model - [PR](https://github.com/BerriAI/litellm/pull/11783) +- **[Team Management](../../docs/proxy/team_management)** + - Allow setting default team for new users - [PR](https://github.com/BerriAI/litellm/pull/11874), [PR](https://github.com/BerriAI/litellm/pull/11877) + - Fix default team settings - [PR](https://github.com/BerriAI/litellm/pull/11887) +- **[SCIM](../../docs/proxy/scim)** + - Add error handling for existing user on SCIM - [PR](https://github.com/BerriAI/litellm/pull/11862) + - Add SCIM PATCH and PUT operations for users - [PR](https://github.com/BerriAI/litellm/pull/11863) +- **Health Check Dashboard** + - Implement health check backend API and storage functionality - [PR](https://github.com/BerriAI/litellm/pull/11852) + - Add LiteLLM_HealthCheckTable to database schema - [PR](https://github.com/BerriAI/litellm/pull/11677) + - Implement health check frontend UI components and dashboard integration - [PR](https://github.com/BerriAI/litellm/pull/11679) + - Add success modal for health check responses - [PR](https://github.com/BerriAI/litellm/pull/11899) + - Fix clickable model ID in health check table - [PR](https://github.com/BerriAI/litellm/pull/11898) + - Fix health check UI table design - [PR](https://github.com/BerriAI/litellm/pull/11897) + +--- + +### Logging / Guardrails Integrations + +#### Bugs +- **[Prometheus](../../docs/observability/prometheus)** + - Fix bug for using prometheus metrics config - [PR](https://github.com/BerriAI/litellm/pull/11779) + +--- + +## Security & Reliability + +#### Security Fixes +- **[Documentation Security](../../docs)** + - Security fixes for docs - [PR](https://github.com/BerriAI/litellm/pull/11776) + - Add Trivy Security Scan for UI + Docs folder - remove all vulnerabilities - [PR](https://github.com/BerriAI/litellm/pull/11778) + +#### Reliability Improvements +- **[Dependencies](../../docs)** + - Fix aiohttp version requirement - [PR](https://github.com/BerriAI/litellm/pull/11777) + - Bump next from 14.2.26 to 14.2.30 in UI dashboard - [PR](https://github.com/BerriAI/litellm/pull/11720) +- **[Networking](../../docs)** + - Allow using CA Bundles - [PR](https://github.com/BerriAI/litellm/pull/11906) + - Add workload identity federation between GCP and AWS - [PR](https://github.com/BerriAI/litellm/pull/10210) + +--- + +## General Proxy Improvements + +#### Features +- **[Deployment](../../docs/proxy/deploy)** + - Add deployment annotations for Kubernetes - [PR](https://github.com/BerriAI/litellm/pull/11849) + - Add ciphers in command and pass to hypercorn for proxy - [PR](https://github.com/BerriAI/litellm/pull/11916) +- **[Custom Root Path](../../docs/proxy/deploy)** + - Fix loading UI on custom root path - [PR](https://github.com/BerriAI/litellm/pull/11912) +- **[SDK Improvements](../../docs/proxy/reliability)** + - LiteLLM SDK / Proxy improvement (don't transform message client-side) - [PR](https://github.com/BerriAI/litellm/pull/11908) + +#### Bugs +- **[Observability](../../docs/observability)** + - Fix boto3 tracer wrapping for observability - [PR](https://github.com/BerriAI/litellm/pull/11869) + + +--- + +## New Contributors +* @kjoth made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11621) +* @shagunb-acn made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11760) +* @MadsRC made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11765) +* @Abiji-2020 made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11746) +* @salzubi401 made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11803) +* @orolega made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11826) +* @X4tar made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11796) +* @karen-veigas made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11858) +* @Shankyg made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11859) +* @pascallim made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/10210) +* @lgruen-vcgs made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11883) +* @rinormaloku made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11851) +* @InvisibleMan1306 made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11849) +* @ervwalter made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11937) +* @ThakeeNathees made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11880) +* @jnhyperion made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11842) +* @Jannchie made their first contribution in [PR](https://github.com/BerriAI/litellm/pull/11860) + +--- + +## Demo Instance + +Here's a Demo Instance to test changes: + +- Instance: https://demo.litellm.ai/ +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## [Git Diff](https://github.com/BerriAI/litellm/compare/v1.72.6-stable...v1.73.0.rc) diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index ace00f64e8b..cca9d2a7fa1 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -347,12 +347,20 @@ const sidebars = { label: "Azure OpenAI", items: [ "providers/azure/azure", + "providers/azure/azure_responses", "providers/azure/azure_embedding", ] }, "providers/azure_ai", "providers/aiml", - "providers/vertex", + { + type: "category", + label: "Vertex AI", + items: [ + "providers/vertex", + "providers/vertex_image", + ] + }, { type: "category", label: "Google AI Studio",