diff --git a/docs/my-website/docs/providers/manus.md b/docs/my-website/docs/providers/manus.md
index 2981ec6d247..92bf2b9b966 100644
--- a/docs/my-website/docs/providers/manus.md
+++ b/docs/my-website/docs/providers/manus.md
@@ -9,7 +9,7 @@ Use Manus AI agents through LiteLLM's OpenAI-compatible Responses API.
|----------|---------|
| Description | Manus is an AI agent platform for complex reasoning tasks, document analysis, and multi-step workflows with asynchronous task execution. |
| Provider Route on LiteLLM | `manus/{agent_profile}` |
-| Supported Operations | `/responses` (Responses API) |
+| Supported Operations | `/responses` (Responses API), `/files` (Files API) |
| Provider Doc | [Manus API ā](https://open.manus.im/docs/openai-compatibility) |
## Model Format
@@ -188,7 +188,182 @@ For production applications, use [webhooks](https://open.manus.im/docs/webhooks)
| `max_output_tokens` | ā
| Limits response length |
| `previous_response_id` | ā
| For multi-turn conversations |
+## Files API
+
+Manus supports file uploads for document analysis and processing. Files can be uploaded and then referenced in Responses API calls.
+
+### LiteLLM Python SDK
+
+```python showLineNumbers title="Upload, Use, Retrieve, and Delete Files"
+import litellm
+import os
+
+# Set API key
+os.environ["MANUS_API_KEY"] = "your-manus-api-key"
+
+# Upload file
+file_content = b"This is a document for analysis."
+created_file = await litellm.acreate_file(
+ file=("document.txt", file_content),
+ purpose="assistants",
+ custom_llm_provider="manus",
+)
+print(f"Uploaded file: {created_file.id}")
+
+# Use file with Responses API
+response = await litellm.aresponses(
+ model="manus/manus-1.6",
+ input=[
+ {
+ "role": "user",
+ "content": [
+ {"type": "input_text", "text": "Summarize this document."},
+ {"type": "input_file", "file_id": created_file.id},
+ ],
+ },
+ ],
+ extra_body={"task_mode": "agent", "agent_profile": "manus-1.6-agent"},
+)
+print(f"Response: {response.id}")
+
+# Retrieve file
+retrieved_file = await litellm.afile_retrieve(
+ file_id=created_file.id,
+ custom_llm_provider="manus",
+)
+print(f"File details: {retrieved_file.filename}, {retrieved_file.bytes} bytes")
+
+# Delete file
+deleted_file = await litellm.afile_delete(
+ file_id=created_file.id,
+ custom_llm_provider="manus",
+)
+print(f"Deleted: {deleted_file.deleted}")
+```
+
+### LiteLLM AI Gateway
+
+
+
+
+```bash showLineNumbers title="Upload File"
+# Upload file
+curl -X POST http://localhost:4000/v1/files \
+ -H "Authorization: Bearer your-proxy-key" \
+ -F "file=@document.txt" \
+ -F "purpose=assistants" \
+ -F "custom_llm_provider=manus"
+
+# Response
+{
+ "id": "file_abc123",
+ "object": "file",
+ "bytes": 1024,
+ "created_at": 1234567890,
+ "filename": "document.txt",
+ "purpose": "assistants",
+ "status": "uploaded"
+}
+```
+
+```bash showLineNumbers title="Use File with Responses API"
+# Create response with file
+curl -X POST http://localhost:4000/responses \
+ -H "Authorization: Bearer your-proxy-key" \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "manus-agent",
+ "input": [
+ {
+ "role": "user",
+ "content": [
+ {"type": "input_text", "text": "Summarize this document."},
+ {"type": "input_file", "file_id": "file_abc123"}
+ ]
+ }
+ ]
+ }'
+```
+
+```bash showLineNumbers title="Retrieve File"
+# Get file details
+curl http://localhost:4000/v1/files/file_abc123 \
+ -H "Authorization: Bearer your-proxy-key"
+
+# Response
+{
+ "id": "file_abc123",
+ "object": "file",
+ "bytes": 1024,
+ "created_at": 1234567890,
+ "filename": "document.txt",
+ "purpose": "assistants",
+ "status": "uploaded"
+}
+```
+
+```bash showLineNumbers title="Delete File"
+# Delete file
+curl -X DELETE http://localhost:4000/v1/files/file_abc123 \
+ -H "Authorization: Bearer your-proxy-key"
+
+# Response
+{
+ "id": "file_abc123",
+ "object": "file",
+ "deleted": true
+}
+```
+
+
+
+
+```python showLineNumbers title="Upload, Use, Retrieve, and Delete Files"
+import openai
+
+client = openai.OpenAI(
+ base_url="http://localhost:4000",
+ api_key="your-proxy-key"
+)
+
+# Upload file
+with open("document.txt", "rb") as f:
+ created_file = client.files.create(
+ file=f,
+ purpose="assistants",
+ extra_body={"custom_llm_provider": "manus"}
+ )
+print(f"Uploaded file: {created_file.id}")
+
+# Use file with Responses API
+response = client.responses.create(
+ model="manus-agent",
+ input=[
+ {
+ "role": "user",
+ "content": [
+ {"type": "input_text", "text": "Summarize this document."},
+ {"type": "input_file", "file_id": created_file.id}
+ ]
+ }
+ ]
+)
+print(f"Response: {response.id}")
+
+# Retrieve file
+retrieved_file = client.files.retrieve(created_file.id)
+print(f"File: {retrieved_file.filename}, {retrieved_file.bytes} bytes")
+
+# Delete file
+deleted_file = client.files.delete(created_file.id)
+print(f"Deleted: {deleted_file.deleted}")
+```
+
+
+
+
## Related Documentation
- [LiteLLM Responses API](/docs/response_api)
+- [LiteLLM Files API](/docs/proxy/litellm_managed_files)
- [Manus OpenAI Compatibility](https://open.manus.im/docs/openai-compatibility)
diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md
index f46608aa57c..33ebf535d29 100644
--- a/docs/my-website/docs/providers/vertex.md
+++ b/docs/my-website/docs/providers/vertex.md
@@ -35,8 +35,6 @@ import json
# !gcloud auth application-default login - run this to add vertex credentials to your env
## OR ##
file_path = 'path/to/vertex_ai_service_account.json'
-## OR ##
-export VERTEXAI_API_KEY="your-api-key"
# Load the JSON file
with open(file_path, 'r') as file:
@@ -49,7 +47,7 @@ vertex_credentials_json = json.dumps(vertex_credentials)
response = completion(
model="vertex_ai/gemini-2.5-pro",
messages=[{ "content": "Hello, how are you?","role": "user"}],
- vertex_credentials=vertex_credentials_json # Can remove this is added VERTEXAI_API_KEY in env
+ vertex_credentials=vertex_credentials_json
)
```
@@ -1331,41 +1329,15 @@ Here's how to use Vertex AI with the LiteLLM Proxy Server
## Authentication - vertex_project, vertex_location, etc.
-LiteLLM supports two authentication methods for Vertex AI:
-
-1. **API Key Authentication** (Recommended for getting started)
-2. **Service Account Credentials** (Recommended for production)
-
Set your vertex credentials via:
- dynamic params
OR
- env vars
-### **Authentication Method 1:
-The simplest way to authenticate with Vertex AI. You can set:
-- `api_key` (str) - Your Vertex AI API key
+### **Dynamic Params**
-**Environment Variables:**
-```bash
-export VERTEXAI_API_KEY="your-api-key"
-```
-
-**Or pass as parameters:**
-```python
-from litellm import completion
-
-response = completion(
- model="vertex_ai/gemini-2.0-flash-exp",
- messages=[{"role": "user", "content": "Hello!"}],
- api_key="your-vertex-api-key",
-
-)
-```
-
-### **Authentication Method 2: Service Account Credentials**
-
-For production environments with fine-grained access control. You can set:
+You can set:
- `vertex_credentials` (str) - can be a json string or filepath to your vertex ai service account.json
- `vertex_location` (str) - place where vertex model is deployed (us-central1, asia-southeast1, etc.). Some models support the global location, please see [Vertex AI documentation](https://cloud.google.com/vertex-ai/generative-ai/docs/learn/locations#supported_models)
- `vertex_project` Optional[str] - use if vertex project different from the one in vertex_credentials
@@ -1420,16 +1392,7 @@ model_list:
### **Environment Variables**
-#### For API Key Authentication:
-
-- `VERTEXAI_API_KEY` or `VERTEX_API_KEY` - Your Vertex AI API key
-
-```bash
-export VERTEXAI_API_KEY="your-vertex-api-key"
-```
-
-#### For Service Account Authentication:
-
+You can set:
- `GOOGLE_APPLICATION_CREDENTIALS` - store the filepath for your service_account.json in here (used by vertex sdk directly).
- VERTEXAI_LOCATION - place where vertex model is deployed (us-central1, asia-southeast1, etc.)
- VERTEXAI_PROJECT - Optional[str] - use if vertex project different from the one in vertex_credentials
diff --git a/docs/my-website/img/ui_endpoint_activity.png b/docs/my-website/img/ui_endpoint_activity.png
new file mode 100644
index 00000000000..e2550066a07
Binary files /dev/null and b/docs/my-website/img/ui_endpoint_activity.png differ
diff --git a/docs/my-website/release_notes/v1.80.11-stable/index.md b/docs/my-website/release_notes/v1.80.11-stable/index.md
index b671b795602..ea0e6083dfc 100644
--- a/docs/my-website/release_notes/v1.80.11-stable/index.md
+++ b/docs/my-website/release_notes/v1.80.11-stable/index.md
@@ -1,5 +1,5 @@
---
-title: "[Preview] v1.80.11 - Google Interactions API"
+title: "v1.80.11 - Google Interactions API"
slug: "v1-80-11"
date: 2025-12-20T10:00:00
authors:
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-docker.litellm.ai/berriai/litellm:v1.80.11.rc.1
+docker.litellm.ai/berriai/litellm:v1.80.11-stable
```
diff --git a/docs/my-website/release_notes/v1.80.14/index.md b/docs/my-website/release_notes/v1.80.14/index.md
new file mode 100644
index 00000000000..38751f80aeb
--- /dev/null
+++ b/docs/my-website/release_notes/v1.80.14/index.md
@@ -0,0 +1,589 @@
+---
+title: "v1.80.14 - Manus API Support"
+slug: "v1-80-14"
+date: 2026-01-10T10:00:00
+authors:
+ - name: Krrish Dholakia
+ title: CEO, LiteLLM
+ url: https://www.linkedin.com/in/krish-d/
+ image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
+ - name: Ishaan Jaff
+ title: CTO, LiteLLM
+ url: https://www.linkedin.com/in/reffajnaahsi/
+ image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
+hide_table_of_contents: false
+---
+
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+## Deploy this version
+
+
+
+
+``` showLineNumbers title="docker run litellm"
+docker run \
+-e STORE_MODEL_IN_DB=True \
+-p 4000:4000 \
+docker.litellm.ai/berriai/litellm:v1.80.14-stable
+```
+
+
+
+
+
+``` showLineNumbers title="pip install litellm"
+pip install litellm==1.80.14
+```
+
+
+
+
+---
+
+## Key Highlights
+
+- **Manus API Support** - [New provider support for Manus API on /responses and GET /responses endpoints](../../docs/providers/manus)
+- **MiniMax Provider** - [Full support for MiniMax chat completions, TTS, and Anthropic native endpoint](../../docs/providers/minimax)
+- **AWS Polly TTS** - [New TTS provider using AWS Polly API](../../docs/providers/aws_polly)
+- **SSO Role Mapping** - Configure role mappings for SSO providers directly in the UI
+- **Cost Estimator** - New UI tool for estimating costs across multiple models and requests
+- **MCP Global Mode** - [Configure MCP servers globally with visibility controls](../../docs/mcp)
+- **Interactions API Bridge** - [Use all LiteLLM providers with the Interactions API](../../docs/interactions)
+- **RAG Query Endpoint** - [New RAG Search/Query endpoint for retrieval-augmented generation](../../docs/search/index)
+- **92.7% Faster Provider Config Lookup** - Major performance improvement for provider configuration
+- **UI Usage - Endpoint Activity** - Users can now see Endpoint Activity Metrics in the UI.
+
+
+---
+
+### UI Usage - Endpoint Activity
+
+
+
+Users can now see Endpoint Activity Metrics in the UI.
+
+---
+
+## New Providers and Endpoints
+
+### New Providers (11 new providers)
+
+| Provider | Supported LiteLLM Endpoints | Description |
+| -------- | ------------------- | ----------- |
+| [Manus](../../docs/providers/manus) | `/responses` | Manus API for agentic workflows |
+| [Manus](../../docs/providers/manus) | `GET /responses` | Manus API for retrieving responses |
+| [Manus](../../docs/providers/manus) | `/files` | Manus API for file management |
+| [MiniMax](../../docs/providers/minimax) | `/chat/completions` | MiniMax chat completions |
+| [MiniMax](../../docs/providers/minimax) | `/audio/speech` | MiniMax text-to-speech |
+| [AWS Polly](../../docs/providers/aws_polly) | `/audio/speech` | AWS Polly text-to-speech API |
+| [GigaChat](../../docs/providers/gigachat) | `/chat/completions` | GigaChat provider for Russian language AI |
+| [LlamaGate](../../docs/providers/llamagate) | `/chat/completions` | LlamaGate chat completions |
+| [LlamaGate](../../docs/providers/llamagate) | `/embeddings` | LlamaGate embeddings |
+| [Abliteration AI](../../docs/providers/abliteration) | `/chat/completions` | Abliteration.ai provider support |
+| [Bedrock](../../docs/providers/bedrock) | `/v1/messages/count_tokens` | Bedrock as new provider for token counting |
+
+### New LLM API Endpoints (3 new endpoints)
+
+| Endpoint | Method | Description | Documentation |
+| -------- | ------ | ----------- | ------------- |
+| `/responses/compact` | POST | Compact responses API endpoint | [Docs](../../docs/response_api) |
+| `/rag/query` | POST | RAG Search/Query endpoint | [Docs](../../docs/search/index) |
+| `/containers/{id}/files` | POST | Upload files to containers | [Docs](../../docs/container_files) |
+
+---
+
+## New Models / Updated Models
+
+#### New Model Support (100+ new models)
+
+| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Features |
+| -------- | ----- | -------------- | ------------------- | -------------------- | -------- |
+| Azure | `azure/gpt-5.2` | 400K | $1.75 | $14.00 | Reasoning, vision, caching |
+| Azure | `azure/gpt-5.2-chat` | 128K | $1.75 | $14.00 | Reasoning, vision |
+| Azure | `azure/gpt-5.2-pro` | 400K | $21.00 | $168.00 | Reasoning, vision, web search |
+| Azure | `azure/gpt-image-1.5` | - | Token-based | Token-based | Image generation/editing |
+| Azure AI | `azure_ai/gpt-oss-120b` | 131K | $0.15 | $0.60 | Function calling |
+| Azure AI | `azure_ai/flux.2-pro` | - | - | $0.04/image | Image generation |
+| Azure AI | `azure_ai/deepseek-v3.2` | 164K | $0.58 | $1.68 | Reasoning, function calling |
+| Bedrock | `amazon.nova-2-multimodal-embeddings-v1:0` | 8K | $0.135 | - | Multimodal embeddings |
+| Bedrock | `writer.palmyra-x4-v1:0` | 128K | $2.50 | $10.00 | Function calling, PDF |
+| Bedrock | `writer.palmyra-x5-v1:0` | 1M | $0.60 | $6.00 | Function calling, PDF |
+| Bedrock | `moonshot.kimi-k2-v1:0` | - | - | - | Kimi K2 model |
+| Cerebras | `cerebras/zai-glm-4.6` | 128K | $2.25 | $2.75 | Reasoning, function calling |
+| GigaChat | `gigachat/GigaChat-2-Lite` | - | - | - | Chat completions |
+| GigaChat | `gigachat/GigaChat-2-Max` | - | - | - | Chat completions |
+| GigaChat | `gigachat/GigaChat-2-Pro` | - | - | - | Chat completions |
+| Gemini | `gemini/veo-3.1-generate-001` | - | - | - | Video generation |
+| Gemini | `gemini/veo-3.1-fast-generate-001` | - | - | - | Video generation |
+| GitHub Copilot | 25+ models | Various | - | - | Chat completions |
+| LlamaGate | 15+ models | Various | - | - | Chat, vision, embeddings |
+| MiniMax | `minimax/abab7-chat-preview` | - | - | - | Chat completions |
+| Novita | 80+ models | Various | Various | Various | Chat, vision, embeddings |
+| OpenRouter | `openrouter/google/gemini-3-flash-preview` | - | - | - | Chat completions |
+| Together AI | Multiple models | Various | Various | Various | Response schema support |
+| Vertex AI | `vertex_ai/zai-glm-4.7` | - | - | - | GLM 4.7 support |
+
+#### Features
+
+- **[Gemini](../../docs/providers/gemini)**
+ - Add image tokens in chat completion - [PR #18327](https://github.com/BerriAI/litellm/pull/18327)
+ - Add usage object in image generation - [PR #18328](https://github.com/BerriAI/litellm/pull/18328)
+ - Add thought signature support via tool call id - [PR #18374](https://github.com/BerriAI/litellm/pull/18374)
+ - Add thought signature for non tool call requests - [PR #18581](https://github.com/BerriAI/litellm/pull/18581)
+ - Preserve system instructions - [PR #18585](https://github.com/BerriAI/litellm/pull/18585)
+ - Fix Gemini 3 images in tool response - [PR #18190](https://github.com/BerriAI/litellm/pull/18190)
+ - Support snake_case for google_search tool parameters - [PR #18451](https://github.com/BerriAI/litellm/pull/18451)
+ - Google GenAI adapter inline data support - [PR #18477](https://github.com/BerriAI/litellm/pull/18477)
+ - Add deprecation_date for discontinued Google models - [PR #18550](https://github.com/BerriAI/litellm/pull/18550)
+- **[Vertex AI](../../docs/providers/vertex)**
+ - Add centralized get_vertex_base_url() helper for global location support - [PR #18410](https://github.com/BerriAI/litellm/pull/18410)
+ - Convert image URLs to base64 for Vertex AI Anthropic - [PR #18497](https://github.com/BerriAI/litellm/pull/18497)
+ - Separate Tool objects for each tool type per API spec - [PR #18514](https://github.com/BerriAI/litellm/pull/18514)
+ - Add thought_signatures to VertexGeminiConfig - [PR #18853](https://github.com/BerriAI/litellm/pull/18853)
+ - Add support for Vertex AI API keys - [PR #18806](https://github.com/BerriAI/litellm/pull/18806)
+ - Add zai glm-4.7 model support - [PR #18782](https://github.com/BerriAI/litellm/pull/18782)
+- **[Azure](../../docs/providers/azure/azure)**
+ - Add Azure gpt-image-1.5 pricing to cost map - [PR #18347](https://github.com/BerriAI/litellm/pull/18347)
+ - Add azure/gpt-5.2-chat model - [PR #18361](https://github.com/BerriAI/litellm/pull/18361)
+ - Add support for image generation via Azure AD token - [PR #18413](https://github.com/BerriAI/litellm/pull/18413)
+ - Add logprobs support for Azure OpenAI GPT-5.2 model - [PR #18856](https://github.com/BerriAI/litellm/pull/18856)
+ - Add Azure BFL Flux 2 models for image generation and editing - [PR #18764](https://github.com/BerriAI/litellm/pull/18764), [PR #18766](https://github.com/BerriAI/litellm/pull/18766)
+- **[Bedrock](../../docs/providers/bedrock)**
+ - Add Bedrock Kimi K2 model support - [PR #18797](https://github.com/BerriAI/litellm/pull/18797)
+ - Add support for model id in bedrock passthrough - [PR #18800](https://github.com/BerriAI/litellm/pull/18800)
+ - Fix Nova model detection for Bedrock provider - [PR #18250](https://github.com/BerriAI/litellm/pull/18250)
+ - Ensure toolUse.input is always a dict when converting from OpenAI format - [PR #18414](https://github.com/BerriAI/litellm/pull/18414)
+- **[Databricks](../../docs/providers/databricks)**
+ - Add enhanced authentication, security features, and custom user-agent support - [PR #18349](https://github.com/BerriAI/litellm/pull/18349)
+- **[MiniMax](../../docs/providers/minimax)**
+ - Add MiniMax chat completion support - [PR #18380](https://github.com/BerriAI/litellm/pull/18380)
+ - Add Anthropic native endpoint support for MiniMax - [PR #18377](https://github.com/BerriAI/litellm/pull/18377)
+ - Add support for MiniMax TTS - [PR #18334](https://github.com/BerriAI/litellm/pull/18334)
+ - Add MiniMax provider support to UI dashboard - [PR #18496](https://github.com/BerriAI/litellm/pull/18496)
+- **[Together AI](../../docs/providers/togetherai)**
+ - Add supports_response_schema to all supported Together AI models - [PR #18368](https://github.com/BerriAI/litellm/pull/18368)
+- **[OpenRouter](../../docs/providers/openrouter)**
+ - Add OpenRouter embeddings API support - [PR #18391](https://github.com/BerriAI/litellm/pull/18391)
+- **[Anthropic](../../docs/providers/anthropic)**
+ - Pass server_tool_use and tool_search_tool_result blocks - [PR #18770](https://github.com/BerriAI/litellm/pull/18770)
+ - Add Anthropic cache control option to image tool call results - [PR #18674](https://github.com/BerriAI/litellm/pull/18674)
+- **[Ollama](../../docs/providers/ollama)**
+ - Add dimensions for ollama embedding - [PR #18536](https://github.com/BerriAI/litellm/pull/18536)
+ - Extract pure base64 data from data URLs for Ollama - [PR #18465](https://github.com/BerriAI/litellm/pull/18465)
+- **[Watsonx](../../docs/providers/watsonx/index)**
+ - Add Watsonx fields support - [PR #18569](https://github.com/BerriAI/litellm/pull/18569)
+ - Fix Watsonx Audio Transcription - filter model field - [PR #18810](https://github.com/BerriAI/litellm/pull/18810)
+- **[SAP](../../docs/providers/sap)**
+ - Add SAP creds for list in proxy UI - [PR #18375](https://github.com/BerriAI/litellm/pull/18375)
+ - Pass through extra params from allowed_openai_params - [PR #18432](https://github.com/BerriAI/litellm/pull/18432)
+ - Add client header for SAP AI Core Tracking - [PR #18714](https://github.com/BerriAI/litellm/pull/18714)
+- **[Fireworks AI](../../docs/providers/fireworks_ai)**
+ - Correct deepseek-v3p2 pricing - [PR #18483](https://github.com/BerriAI/litellm/pull/18483)
+- **[ZAI](../../docs/providers/zai)**
+ - Add GLM-4.7 model with reasoning support - [PR #18476](https://github.com/BerriAI/litellm/pull/18476)
+- **[Codestral](../../docs/providers/codestral)**
+ - Correctly route codestral chat and FIM endpoints - [PR #18467](https://github.com/BerriAI/litellm/pull/18467)
+- **[Azure AI](../../docs/providers/azure_ai)**
+ - Fix authentication errors at messages API via azure_ai - [PR #18500](https://github.com/BerriAI/litellm/pull/18500)
+
+#### New Provider Support
+
+- **[AWS Polly](../../docs/providers/aws_polly)** - Add AWS Polly API for TTS - [PR #18326](https://github.com/BerriAI/litellm/pull/18326)
+- **[GigaChat](../../docs/providers/gigachat)** - Add GigaChat provider support - [PR #18564](https://github.com/BerriAI/litellm/pull/18564)
+- **[LlamaGate](../../docs/providers/llamagate)** - Add LlamaGate as a new provider - [PR #18673](https://github.com/BerriAI/litellm/pull/18673)
+- **[Abliteration AI](../../docs/providers/abliteration)** - Add abliteration.ai provider - [PR #18678](https://github.com/BerriAI/litellm/pull/18678)
+- **[Manus](../../docs/providers/manus)** - Add Manus API support on /responses, GET /responses - [PR #18804](https://github.com/BerriAI/litellm/pull/18804)
+- **5 AI Providers via openai_like** - Add 5 AI providers using openai_like - [PR #18362](https://github.com/BerriAI/litellm/pull/18362)
+
+### Bug Fixes
+
+- **[Gemini](../../docs/providers/gemini)**
+ - Properly catch context window exceeded errors - [PR #18283](https://github.com/BerriAI/litellm/pull/18283)
+ - Remove prompt caching headers as support has been removed - [PR #18579](https://github.com/BerriAI/litellm/pull/18579)
+ - Fix generate content request with audio file id - [PR #18745](https://github.com/BerriAI/litellm/pull/18745)
+ - Fix google_genai streaming adapter provider handling - [PR #18845](https://github.com/BerriAI/litellm/pull/18845)
+- **[Groq](../../docs/providers/groq)**
+ - Remove deprecated Groq models and update model registry - [PR #18062](https://github.com/BerriAI/litellm/pull/18062)
+- **[Vertex AI](../../docs/providers/vertex)**
+ - Handle unsupported region for Vertex AI count tokens endpoint - [PR #18665](https://github.com/BerriAI/litellm/pull/18665)
+- **General**
+ - Fix request body for image embedding request - [PR #18336](https://github.com/BerriAI/litellm/pull/18336)
+ - Fix lost tool_calls when streaming has both text and tool_calls - [PR #18316](https://github.com/BerriAI/litellm/pull/18316)
+ - Add all resolution for gpt-image-1.5 - [PR #18586](https://github.com/BerriAI/litellm/pull/18586)
+ - Fix gpt-image-1 cost calculation using token-based pricing - [PR #17906](https://github.com/BerriAI/litellm/pull/17906)
+ - Fix response_format leaking into extra_body - [PR #18859](https://github.com/BerriAI/litellm/pull/18859)
+ - Align max_tokens with max_output_tokens for consistency - [PR #18820](https://github.com/BerriAI/litellm/pull/18820)
+
+---
+
+## LLM API Endpoints
+
+#### Features
+
+- **[Responses API](../../docs/response_api)**
+ - Add new compact endpoint (v1/responses/compact) - [PR #18697](https://github.com/BerriAI/litellm/pull/18697)
+ - Support more streaming callback hooks - [PR #18513](https://github.com/BerriAI/litellm/pull/18513)
+ - Add mapping for reasoning effort to summary param - [PR #18635](https://github.com/BerriAI/litellm/pull/18635)
+ - Add output_text property to ResponsesAPIResponse - [PR #18491](https://github.com/BerriAI/litellm/pull/18491)
+ - Add annotations to completions responses API bridge - [PR #18754](https://github.com/BerriAI/litellm/pull/18754)
+- **[Interactions API](../../docs/interactions)**
+ - Allow using all LiteLLM providers (interactions -> responses API bridge) - [PR #18373](https://github.com/BerriAI/litellm/pull/18373)
+- **[RAG Search API](../../docs/search/index)**
+ - Add RAG Search/Query endpoint - [PR #18376](https://github.com/BerriAI/litellm/pull/18376)
+- **[CountTokens API](../../docs/anthropic_count_tokens)**
+ - Add Bedrock as a new provider for `/v1/messages/count_tokens` - [PR #18858](https://github.com/BerriAI/litellm/pull/18858)
+- **[Generate Content](../../docs/providers/gemini)**
+ - Add generate content in LLM route - [PR #18405](https://github.com/BerriAI/litellm/pull/18405)
+- **General**
+ - Enable async_post_call_failure_hook to transform error responses - [PR #18348](https://github.com/BerriAI/litellm/pull/18348)
+ - Calculate total_tokens manually if missing and can be calculated - [PR #18445](https://github.com/BerriAI/litellm/pull/18445)
+ - Add custom llm provider to get_llm_provider when sent via UI - [PR #18638](https://github.com/BerriAI/litellm/pull/18638)
+
+#### Bugs
+
+- **General**
+ - Handle empty error objects in response conversion - [PR #18493](https://github.com/BerriAI/litellm/pull/18493)
+ - Preserve client error status codes in streaming mode - [PR #18698](https://github.com/BerriAI/litellm/pull/18698)
+ - Return json error response instead of SSE format for initial streaming errors - [PR #18757](https://github.com/BerriAI/litellm/pull/18757)
+ - Fix auth header for custom api base in generateContent request - [PR #18637](https://github.com/BerriAI/litellm/pull/18637)
+ - Tool content should be string for Deepinfra - [PR #18739](https://github.com/BerriAI/litellm/pull/18739)
+ - Fix incomplete usage in response object passed - [PR #18799](https://github.com/BerriAI/litellm/pull/18799)
+ - Unify model names to provider-defined names - [PR #18573](https://github.com/BerriAI/litellm/pull/18573)
+
+---
+
+## Management Endpoints / UI
+
+#### Features
+
+- **SSO Configuration**
+ - Add SSO Role Mapping feature - [PR #18090](https://github.com/BerriAI/litellm/pull/18090)
+ - Add SSO Settings Page - [PR #18600](https://github.com/BerriAI/litellm/pull/18600)
+ - Allow adding role mappings for SSO - [PR #18593](https://github.com/BerriAI/litellm/pull/18593)
+ - SSO Settings Page Add Role Mappings - [PR #18677](https://github.com/BerriAI/litellm/pull/18677)
+ - SSO Settings Loading State + Deprecate Previous SSO Flow - [PR #18617](https://github.com/BerriAI/litellm/pull/18617)
+- **Virtual Keys**
+ - Allow deleting key expiry - [PR #18278](https://github.com/BerriAI/litellm/pull/18278)
+ - Add optional query param "expand" to /key/list - [PR #18502](https://github.com/BerriAI/litellm/pull/18502)
+ - Key Table Loading Skeleton - [PR #18527](https://github.com/BerriAI/litellm/pull/18527)
+ - Allow column resizing on Keys Table - [PR #18424](https://github.com/BerriAI/litellm/pull/18424)
+ - Virtual Keys Table Loading State Between Pages - [PR #18619](https://github.com/BerriAI/litellm/pull/18619)
+ - Key and Team Router Setting - [PR #18790](https://github.com/BerriAI/litellm/pull/18790)
+ - Allow router_settings on Keys and Teams - [PR #18675](https://github.com/BerriAI/litellm/pull/18675)
+ - Use timedelta to calculate key expiry on generate - [PR #18666](https://github.com/BerriAI/litellm/pull/18666)
+- **Models + Endpoints**
+ - Add Model Clearer Flow For Team Admins - [PR #18532](https://github.com/BerriAI/litellm/pull/18532)
+ - Model Page Loading State - [PR #18574](https://github.com/BerriAI/litellm/pull/18574)
+ - Model Page Model Provider Select Performance - [PR #18425](https://github.com/BerriAI/litellm/pull/18425)
+ - Model Page Sorting Sorts Entire Set - [PR #18420](https://github.com/BerriAI/litellm/pull/18420)
+ - Refactor Model Hub Page - [PR #18568](https://github.com/BerriAI/litellm/pull/18568)
+ - Add request provider form on UI - [PR #18704](https://github.com/BerriAI/litellm/pull/18704)
+- **Organizations & Teams**
+ - Allow Organization Admins to See Organization Tab - [PR #18400](https://github.com/BerriAI/litellm/pull/18400)
+ - Resolve Organization Alias on Team Table - [PR #18401](https://github.com/BerriAI/litellm/pull/18401)
+ - Resolve Team Alias in Organization Info View - [PR #18404](https://github.com/BerriAI/litellm/pull/18404)
+ - Allow Organization Admins to View Their Organization Info - [PR #18417](https://github.com/BerriAI/litellm/pull/18417)
+ - Allow editing team_member_budget_duration in /team/update - [PR #18735](https://github.com/BerriAI/litellm/pull/18735)
+ - Reusable Duration Select + Team Update Member Budget Duration - [PR #18736](https://github.com/BerriAI/litellm/pull/18736)
+- **Usage & Spend**
+ - Add Error Code Filtering on Spend Logs - [PR #18359](https://github.com/BerriAI/litellm/pull/18359)
+ - Add Error Code Filtering on UI - [PR #18366](https://github.com/BerriAI/litellm/pull/18366)
+ - Usage Page User Max Budget fix - [PR #18555](https://github.com/BerriAI/litellm/pull/18555)
+ - Add endpoint to Daily Activity Tables - [PR #18729](https://github.com/BerriAI/litellm/pull/18729)
+ - Endpoint Activity in Usage - [PR #18798](https://github.com/BerriAI/litellm/pull/18798)
+- **Cost Estimator**
+ - Add Cost Estimator for AI Gateway - [PR #18643](https://github.com/BerriAI/litellm/pull/18643)
+ - Add view for estimating costs across requests - [PR #18645](https://github.com/BerriAI/litellm/pull/18645)
+ - Allow selecting many models for cost estimator - [PR #18653](https://github.com/BerriAI/litellm/pull/18653)
+- **CloudZero**
+ - Improve Create and Delete Path for CloudZero - [PR #18263](https://github.com/BerriAI/litellm/pull/18263)
+ - Add CloudZero UI Docs - [PR #18350](https://github.com/BerriAI/litellm/pull/18350)
+- **Playground**
+ - Add MCP test support to completions on Playground - [PR #18440](https://github.com/BerriAI/litellm/pull/18440)
+ - Add selectable MCP servers to the playground - [PR #18578](https://github.com/BerriAI/litellm/pull/18578)
+ - Add custom proxy base URL support to Playground - [PR #18661](https://github.com/BerriAI/litellm/pull/18661)
+- **General UI**
+ - UI styling improvements and fixes - [PR #18310](https://github.com/BerriAI/litellm/pull/18310)
+ - Add reusable "New" badge component for feature highlights - [PR #18537](https://github.com/BerriAI/litellm/pull/18537)
+ - Hide New Badges - [PR #18547](https://github.com/BerriAI/litellm/pull/18547)
+ - Change Budget page to Have Tabs - [PR #18576](https://github.com/BerriAI/litellm/pull/18576)
+ - Clicking on Logo Directs to Correct URL - [PR #18575](https://github.com/BerriAI/litellm/pull/18575)
+ - Add UI support for configuring meta URLs - [PR #18580](https://github.com/BerriAI/litellm/pull/18580)
+ - Expire Previous UI Session Tokens on Login - [PR #18557](https://github.com/BerriAI/litellm/pull/18557)
+ - Add license endpoint - [PR #18311](https://github.com/BerriAI/litellm/pull/18311)
+ - Router Fields Endpoint + React Query for Router Fields - [PR #18880](https://github.com/BerriAI/litellm/pull/18880)
+
+#### Bugs
+
+- **UI Fixes**
+ - Fix Key Creation MCP Settings Submit Form Unintentionally - [PR #18355](https://github.com/BerriAI/litellm/pull/18355)
+ - Fix UI Disappears in Development Environments - [PR #18399](https://github.com/BerriAI/litellm/pull/18399)
+ - Fix Disable Admin UI Flag - [PR #18397](https://github.com/BerriAI/litellm/pull/18397)
+ - Remove Model Analytics From Model Page - [PR #18552](https://github.com/BerriAI/litellm/pull/18552)
+ - Useful Links Remove Modal on Adding Links - [PR #18602](https://github.com/BerriAI/litellm/pull/18602)
+ - SSO Edit Modal Clear Role Mapping Values on Provider Change - [PR #18680](https://github.com/BerriAI/litellm/pull/18680)
+ - UI Login Case Sensitivity fix - [PR #18877](https://github.com/BerriAI/litellm/pull/18877)
+- **API Fixes**
+ - Fix User Invite & Key Generation Email Notification Logic - [PR #18524](https://github.com/BerriAI/litellm/pull/18524)
+ - Normalize Proxy Config Callback - [PR #18775](https://github.com/BerriAI/litellm/pull/18775)
+ - Return empty data array instead of 500 when no models configured - [PR #18556](https://github.com/BerriAI/litellm/pull/18556)
+ - Enforce org level max budget - [PR #18813](https://github.com/BerriAI/litellm/pull/18813)
+
+---
+
+## AI Integrations
+
+### New Integrations (4 new integrations)
+
+| Integration | Type | Description |
+| ----------- | ---- | ----------- |
+| [Focus](../../docs/observability/focus) | Logging | Focus export support for observability - [PR #18802](https://github.com/BerriAI/litellm/pull/18802) |
+| [SigNoz](../../docs/observability/signoz) | Logging | SigNoz integration for observability - [PR #18726](https://github.com/BerriAI/litellm/pull/18726) |
+| [Qualifire](../../docs/proxy/guardrails/qualifire) | Guardrails | Qualifire guardrails and eval webhook - [PR #18594](https://github.com/BerriAI/litellm/pull/18594) |
+| [Levo AI](../../docs/observability/levo_integration) | Guardrails | Levo AI integration for security - [PR #18529](https://github.com/BerriAI/litellm/pull/18529) |
+
+### Logging
+
+- **[DataDog](../../docs/proxy/logging#datadog)**
+ - Fix span kind fallback when parent_id missing - [PR #18418](https://github.com/BerriAI/litellm/pull/18418)
+- **[Langfuse](../../docs/proxy/logging#langfuse)**
+ - Map Gemini cached_tokens to Langfuse cache_read_input_tokens - [PR #18614](https://github.com/BerriAI/litellm/pull/18614)
+- **[Prometheus](../../docs/proxy/logging#prometheus)**
+ - Align prometheus metric names with DEFINED_PROMETHEUS_METRICS - [PR #18463](https://github.com/BerriAI/litellm/pull/18463)
+ - Add Prometheus metrics for request queue time and guardrails - [PR #17973](https://github.com/BerriAI/litellm/pull/17973)
+ - Add caching metrics for cache hits, misses, and tokens - [PR #18755](https://github.com/BerriAI/litellm/pull/18755)
+ - Skip metrics for invalid API key requests - [PR #18788](https://github.com/BerriAI/litellm/pull/18788)
+- **[Braintrust](../../docs/proxy/logging#braintrust)**
+ - Pass span_attributes in async logging and skip tags on non-root spans - [PR #18409](https://github.com/BerriAI/litellm/pull/18409)
+- **[CloudZero](../../docs/proxy/logging#cloudzero)**
+ - Add user email to CloudZero - [PR #18584](https://github.com/BerriAI/litellm/pull/18584)
+- **[OpenTelemetry](../../docs/proxy/logging#opentelemetry)**
+ - Use already configured opentelemetry providers - [PR #18279](https://github.com/BerriAI/litellm/pull/18279)
+ - Prevent LiteLLM from closing external OTEL spans - [PR #18553](https://github.com/BerriAI/litellm/pull/18553)
+ - Allow configuring arize project name for OpenTelemetry service name - [PR #18738](https://github.com/BerriAI/litellm/pull/18738)
+- **[LangSmith](../../docs/proxy/logging#langsmith)**
+ - Add support for LangSmith organization-scoped API keys with tenant ID - [PR #18623](https://github.com/BerriAI/litellm/pull/18623)
+- **[Generic API Logger](../../docs/proxy/logging#generic-api-logger)**
+ - Add log_format option to GenericAPILogger - [PR #18587](https://github.com/BerriAI/litellm/pull/18587)
+
+### Guardrails
+
+- **[Content Filter](../../docs/proxy/guardrails/litellm_content_filter)**
+ - Add content filter logs page - [PR #18335](https://github.com/BerriAI/litellm/pull/18335)
+ - Log actual event type for guardrails - [PR #18489](https://github.com/BerriAI/litellm/pull/18489)
+- **[Qualifire](../../docs/proxy/guardrails/qualifire)**
+ - Add Qualifire eval webhook - [PR #18836](https://github.com/BerriAI/litellm/pull/18836)
+- **[Lasso Security](../../docs/proxy/guardrails/lasso_security)**
+ - Add Lasso guardrail API docs - [PR #18652](https://github.com/BerriAI/litellm/pull/18652)
+- **[Noma Security](../../docs/proxy/guardrails/noma_security)**
+ - Add MCP guardrail support for Noma - [PR #18668](https://github.com/BerriAI/litellm/pull/18668)
+- **[Bedrock Guardrails](../../docs/proxy/guardrails/bedrock)**
+ - Remove redundant Bedrock guardrail block handling - [PR #18634](https://github.com/BerriAI/litellm/pull/18634)
+- **General**
+ - Generic guardrail API update - [PR #18647](https://github.com/BerriAI/litellm/pull/18647)
+ - Prevent proxy startup failures from case-sensitive tool permission guardrail validation - [PR #18662](https://github.com/BerriAI/litellm/pull/18662)
+ - Extend case normalization to ALL guardrail types - [PR #18664](https://github.com/BerriAI/litellm/pull/18664)
+ - Fix MCP handling in unified guardrail - [PR #18630](https://github.com/BerriAI/litellm/pull/18630)
+ - Fix embeddings calltype for guardrail precallhook - [PR #18740](https://github.com/BerriAI/litellm/pull/18740)
+
+---
+
+## Spend Tracking, Budgets and Rate Limiting
+
+- **Platform Fee / Margins** - Add support for Platform Fee / Margins - [PR #18427](https://github.com/BerriAI/litellm/pull/18427)
+- **Negative Budget Validation** - Add validation for negative budget - [PR #18583](https://github.com/BerriAI/litellm/pull/18583)
+- **Cost Calculation Fixes**
+ - Correct cost calculation when reasoning_tokens are without text_tokens - [PR #18607](https://github.com/BerriAI/litellm/pull/18607)
+ - Fix background cost tracking tests - [PR #18588](https://github.com/BerriAI/litellm/pull/18588)
+- **Tag Routing** - Support toggling tag matching between ANY and ALL - [PR #18776](https://github.com/BerriAI/litellm/pull/18776)
+
+---
+
+## MCP Gateway
+
+- **MCP Global Mode** - Add MCP global mode - [PR #18639](https://github.com/BerriAI/litellm/pull/18639)
+- **MCP Server Visibility** - Add configurable MCP server visibility - [PR #18681](https://github.com/BerriAI/litellm/pull/18681)
+- **MCP Registry** - Add MCP registry - [PR #18850](https://github.com/BerriAI/litellm/pull/18850)
+- **MCP Stdio Header** - Support MCP stdio header env overrides - [PR #18324](https://github.com/BerriAI/litellm/pull/18324)
+- **Parallel Tool Fetching** - Parallelize tool fetching from multiple MCP servers - [PR #18627](https://github.com/BerriAI/litellm/pull/18627)
+- **Optimize MCP Server Listing** - Separate health checks for optimized listing - [PR #18530](https://github.com/BerriAI/litellm/pull/18530)
+- **Auth Improvements**
+ - Require auth for MCP connection test endpoint - [PR #18290](https://github.com/BerriAI/litellm/pull/18290)
+ - Fix MCP gateway OAuth2 auth issues and ClosedResourceError - [PR #18281](https://github.com/BerriAI/litellm/pull/18281)
+- **Bug Fixes**
+ - Fix MCP server health status reporting - [PR #18443](https://github.com/BerriAI/litellm/pull/18443)
+ - Fix OpenAPI to MCP tool conversion - [PR #18597](https://github.com/BerriAI/litellm/pull/18597)
+ - Remove exec() usage and handle invalid OpenAPI parameter names for security - [PR #18480](https://github.com/BerriAI/litellm/pull/18480)
+ - Fix MCP error when using multiple servers simultaneously - [PR #18855](https://github.com/BerriAI/litellm/pull/18855)
+- **Migrate MCP Fetching Logic to React Query** - [PR #18352](https://github.com/BerriAI/litellm/pull/18352)
+
+---
+
+## Performance / Loadbalancing / Reliability improvements
+
+- **92.7% Faster Provider Config Lookup** - LiteLLM now stresses LLM providers 2.5x more - [PR #18867](https://github.com/BerriAI/litellm/pull/18867)
+- **Lazy Loading Improvements**
+ - Consolidate lazy import handlers with registry pattern - [PR #18389](https://github.com/BerriAI/litellm/pull/18389)
+ - Complete lazy loading migration for all 180+ LLM config classes - [PR #18392](https://github.com/BerriAI/litellm/pull/18392)
+ - Lazy load additional components (types, callbacks, utilities) - [PR #18396](https://github.com/BerriAI/litellm/pull/18396)
+ - Add lazy loading for get_llm_provider - [PR #18591](https://github.com/BerriAI/litellm/pull/18591)
+ - Lazy-load heavy audio library and loggers - [PR #18592](https://github.com/BerriAI/litellm/pull/18592)
+ - Lazy load 9 heavy imports in litellm/utils.py - [PR #18595](https://github.com/BerriAI/litellm/pull/18595)
+ - Lazy load heavy imports to improve import time and memory usage - [PR #18610](https://github.com/BerriAI/litellm/pull/18610)
+ - Implement lazy loading for provider configs, model info classes, streaming handlers - [PR #18611](https://github.com/BerriAI/litellm/pull/18611)
+ - Lazy load 15 additional imports - [PR #18613](https://github.com/BerriAI/litellm/pull/18613)
+ - Lazy load 15+ unused imports - [PR #18616](https://github.com/BerriAI/litellm/pull/18616)
+ - Lazy load DatadogLLMObsInitParams - [PR #18658](https://github.com/BerriAI/litellm/pull/18658)
+ - Migrate utils.py lazy imports to registry pattern - [PR #18657](https://github.com/BerriAI/litellm/pull/18657)
+ - Lazy load get_llm_provider and remove_index_from_tool_calls - [PR #18608](https://github.com/BerriAI/litellm/pull/18608)
+- **Router Improvements**
+ - Validate routing_strategy at startup to fail fast with helpful error - [PR #18624](https://github.com/BerriAI/litellm/pull/18624)
+ - Correct num_retries tracking in retry logic - [PR #18712](https://github.com/BerriAI/litellm/pull/18712)
+ - Improve error messages and validation for wildcard routing with multiple credentials - [PR #18629](https://github.com/BerriAI/litellm/pull/18629)
+- **Memory Improvements**
+ - Add memory pattern detection test and fix bad memory patterns - [PR #18589](https://github.com/BerriAI/litellm/pull/18589)
+ - Add unbounded data structure detection to memory test - [PR #18590](https://github.com/BerriAI/litellm/pull/18590)
+ - Add memory leak detection tests with CI integration - [PR #18881](https://github.com/BerriAI/litellm/pull/18881)
+- **Database**
+ - Add idx on LOWER(user_email) for faster duplicate email checks - [PR #18828](https://github.com/BerriAI/litellm/pull/18828)
+ - Proactive RDS IAM token refresh to prevent 15-min connection failed - [PR #18795](https://github.com/BerriAI/litellm/pull/18795)
+ - Clarify database_connection_pool_limit applies per worker - [PR #18780](https://github.com/BerriAI/litellm/pull/18780)
+ - Make base_connection_pool_limit default value the same - [PR #18721](https://github.com/BerriAI/litellm/pull/18721)
+- **Docker**
+ - Add libsndfile to database Docker image for audio processing - [PR #18612](https://github.com/BerriAI/litellm/pull/18612)
+ - Add line_profiler support for performance analysis and fix Windows CRLF issues - [PR #18773](https://github.com/BerriAI/litellm/pull/18773)
+- **Helm**
+ - Add lifecycle support to Helm charts - [PR #18517](https://github.com/BerriAI/litellm/pull/18517)
+- **Authentication**
+ - Add Kubernetes ServiceAccount JWT authentication support - [PR #18055](https://github.com/BerriAI/litellm/pull/18055)
+ - Use async anthropic client to prevent event loop blocking - [PR #18435](https://github.com/BerriAI/litellm/pull/18435)
+- **Logging Worker**
+ - Handle event loop changes in multiprocessing - [PR #18423](https://github.com/BerriAI/litellm/pull/18423)
+- **Security**
+ - Prevent expired key plaintext leak in error response - [PR #18860](https://github.com/BerriAI/litellm/pull/18860)
+ - Mask extra header secrets in model info - [PR #18822](https://github.com/BerriAI/litellm/pull/18822)
+ - Prevent duplicate User-Agent tags in request_tags - [PR #18723](https://github.com/BerriAI/litellm/pull/18723)
+ - Properly use litellm api keys - [PR #18832](https://github.com/BerriAI/litellm/pull/18832)
+- **Misc**
+ - Remove double imports in main.py - [PR #18406](https://github.com/BerriAI/litellm/pull/18406)
+ - Add LITELLM_DISABLE_LAZY_LOADING env var to fix VCR cassette creation issue - [PR #18725](https://github.com/BerriAI/litellm/pull/18725)
+ - Add xiaomi_mimo to LlmProviders enum to fix router support - [PR #18819](https://github.com/BerriAI/litellm/pull/18819)
+ - Allow installation with current grpcio on old Python - [PR #18473](https://github.com/BerriAI/litellm/pull/18473)
+ - Add Custom CA certificates to boto3 clients - [PR #18852](https://github.com/BerriAI/litellm/pull/18852)
+ - Fix bedrock_cache, metadata and max_model_budget - [PR #18872](https://github.com/BerriAI/litellm/pull/18872)
+ - Fix LiteLLM SDK embedding headers missing field - [PR #18844](https://github.com/BerriAI/litellm/pull/18844)
+ - Put automatic reasoning summary inclusion behind feat flag - [PR #18688](https://github.com/BerriAI/litellm/pull/18688)
+ - turn_off_message_logging Does Not Redact Request Messages in proxy_server_request Field - [PR #18897](https://github.com/BerriAI/litellm/pull/18897)
+
+---
+
+## Documentation Updates
+
+- **Provider Documentation**
+ - Update MiniMax docs to be in proper format - [PR #18403](https://github.com/BerriAI/litellm/pull/18403)
+ - Add docs for 5 AI providers - [PR #18388](https://github.com/BerriAI/litellm/pull/18388)
+ - Fix gpt-5-mini reasoning_effort supported values - [PR #18346](https://github.com/BerriAI/litellm/pull/18346)
+ - Fix PDF documentation inconsistency in Anthropic page - [PR #18816](https://github.com/BerriAI/litellm/pull/18816)
+ - Update OpenRouter docs to include embedding support - [PR #18874](https://github.com/BerriAI/litellm/pull/18874)
+ - Add LITELLM_REASONING_AUTO_SUMMARY in doc - [PR #18705](https://github.com/BerriAI/litellm/pull/18705)
+- **MCP Documentation**
+ - Agentcore MCP server docs - [PR #18603](https://github.com/BerriAI/litellm/pull/18603)
+ - Mention MCP prompt/resources types in overview - [PR #18669](https://github.com/BerriAI/litellm/pull/18669)
+ - Add Focus docs - [PR #18837](https://github.com/BerriAI/litellm/pull/18837)
+- **Guardrails Documentation**
+ - Qualifire docs hotfix - [PR #18724](https://github.com/BerriAI/litellm/pull/18724)
+- **Infrastructure Documentation**
+ - IAM Roles Anywhere docs - [PR #18559](https://github.com/BerriAI/litellm/pull/18559)
+ - Fix formatting in proxy configs documentation - [PR #18498](https://github.com/BerriAI/litellm/pull/18498)
+ - Fix GCS cache docs missing for proxy mode - [PR #13328](https://github.com/BerriAI/litellm/pull/13328)
+ - Fix how to execute cloudzero sql - [PR #18841](https://github.com/BerriAI/litellm/pull/18841)
+- **General**
+ - LiteLLM adopters section - [PR #18605](https://github.com/BerriAI/litellm/pull/18605)
+ - Remove redundant comments about setting litellm.callbacks - [PR #18711](https://github.com/BerriAI/litellm/pull/18711)
+ - Update header to be markdown bold by removing space - [PR #18846](https://github.com/BerriAI/litellm/pull/18846)
+ - Manus docs - new provider - [PR #18817](https://github.com/BerriAI/litellm/pull/18817)
+
+---
+
+## New Contributors
+
+* @prasadkona made their first contribution in [PR #18349](https://github.com/BerriAI/litellm/pull/18349)
+* @lucasrothman made their first contribution in [PR #18283](https://github.com/BerriAI/litellm/pull/18283)
+* @aggeentik made their first contribution in [PR #18317](https://github.com/BerriAI/litellm/pull/18317)
+* @mihidumh made their first contribution in [PR #18361](https://github.com/BerriAI/litellm/pull/18361)
+* @Prazeina made their first contribution in [PR #18498](https://github.com/BerriAI/litellm/pull/18498)
+* @systec-dk made their first contribution in [PR #18500](https://github.com/BerriAI/litellm/pull/18500)
+* @xuan07t2 made their first contribution in [PR #18514](https://github.com/BerriAI/litellm/pull/18514)
+* @RensDimmendaal made their first contribution in [PR #18190](https://github.com/BerriAI/litellm/pull/18190)
+* @yurekami made their first contribution in [PR #18483](https://github.com/BerriAI/litellm/pull/18483)
+* @agertz7 made their first contribution in [PR #18556](https://github.com/BerriAI/litellm/pull/18556)
+* @yudelevi made their first contribution in [PR #18550](https://github.com/BerriAI/litellm/pull/18550)
+* @smallp made their first contribution in [PR #18536](https://github.com/BerriAI/litellm/pull/18536)
+* @kevinpauer made their first contribution in [PR #18569](https://github.com/BerriAI/litellm/pull/18569)
+* @cansakiroglu made their first contribution in [PR #18517](https://github.com/BerriAI/litellm/pull/18517)
+* @dee-walia20 made their first contribution in [PR #18432](https://github.com/BerriAI/litellm/pull/18432)
+* @luxinfeng made their first contribution in [PR #18477](https://github.com/BerriAI/litellm/pull/18477)
+* @cantalupo555 made their first contribution in [PR #18476](https://github.com/BerriAI/litellm/pull/18476)
+* @andersk made their first contribution in [PR #18473](https://github.com/BerriAI/litellm/pull/18473)
+* @majiayu000 made their first contribution in [PR #18467](https://github.com/BerriAI/litellm/pull/18467)
+* @amangupta-20 made their first contribution in [PR #18529](https://github.com/BerriAI/litellm/pull/18529)
+* @hamzaq453 made their first contribution in [PR #18480](https://github.com/BerriAI/litellm/pull/18480)
+* @ktsaou made their first contribution in [PR #18627](https://github.com/BerriAI/litellm/pull/18627)
+* @FlibbertyGibbitz made their first contribution in [PR #18624](https://github.com/BerriAI/litellm/pull/18624)
+* @drorIvry made their first contribution in [PR #18594](https://github.com/BerriAI/litellm/pull/18594)
+* @urainshah made their first contribution in [PR #18524](https://github.com/BerriAI/litellm/pull/18524)
+* @mangabits made their first contribution in [PR #18279](https://github.com/BerriAI/litellm/pull/18279)
+* @0717376 made their first contribution in [PR #18564](https://github.com/BerriAI/litellm/pull/18564)
+* @nmgarza5 made their first contribution in [PR #17330](https://github.com/BerriAI/litellm/pull/17330)
+* @wileykestner made their first contribution in [PR #18445](https://github.com/BerriAI/litellm/pull/18445)
+* @minijeong-log made their first contribution in [PR #14440](https://github.com/BerriAI/litellm/pull/14440)
+* @Isaac4real made their first contribution in [PR #18710](https://github.com/BerriAI/litellm/pull/18710)
+* @marukaz made their first contribution in [PR #18711](https://github.com/BerriAI/litellm/pull/18711)
+* @rohitravirane made their first contribution in [PR #18712](https://github.com/BerriAI/litellm/pull/18712)
+* @lizzzcai made their first contribution in [PR #18714](https://github.com/BerriAI/litellm/pull/18714)
+* @hkd987 made their first contribution in [PR #18673](https://github.com/BerriAI/litellm/pull/18673)
+* @Mr-Pepe made their first contribution in [PR #18674](https://github.com/BerriAI/litellm/pull/18674)
+* @gkarthi-signoz made their first contribution in [PR #18726](https://github.com/BerriAI/litellm/pull/18726)
+* @Tianduo16 made their first contribution in [PR #18723](https://github.com/BerriAI/litellm/pull/18723)
+* @wilsonjr made their first contribution in [PR #18721](https://github.com/BerriAI/litellm/pull/18721)
+* @abliteration-ai made their first contribution in [PR #18678](https://github.com/BerriAI/litellm/pull/18678)
+* @danialkhan02 made their first contribution in [PR #18770](https://github.com/BerriAI/litellm/pull/18770)
+* @ihower made their first contribution in [PR #18409](https://github.com/BerriAI/litellm/pull/18409)
+* @elkkhan made their first contribution in [PR #18391](https://github.com/BerriAI/litellm/pull/18391)
+* @runixer made their first contribution in [PR #18435](https://github.com/BerriAI/litellm/pull/18435)
+* @choby-shun made their first contribution in [PR #18776](https://github.com/BerriAI/litellm/pull/18776)
+* @jutaz made their first contribution in [PR #18853](https://github.com/BerriAI/litellm/pull/18853)
+* @sjmatta made their first contribution in [PR #18250](https://github.com/BerriAI/litellm/pull/18250)
+* @andres-ortizl made their first contribution in [PR #18856](https://github.com/BerriAI/litellm/pull/18856)
+* @gauthiermartin made their first contribution in [PR #18844](https://github.com/BerriAI/litellm/pull/18844)
+* @mel2oo made their first contribution in [PR #18845](https://github.com/BerriAI/litellm/pull/18845)
+* @DominikHallab made their first contribution in [PR #18846](https://github.com/BerriAI/litellm/pull/18846)
+* @ji-chuan-che made their first contribution in [PR #18540](https://github.com/BerriAI/litellm/pull/18540)
+* @raghav-stripe made their first contribution in [PR #18858](https://github.com/BerriAI/litellm/pull/18858)
+* @akraines made their first contribution in [PR #18629](https://github.com/BerriAI/litellm/pull/18629)
+* @otaviofbrito made their first contribution in [PR #18665](https://github.com/BerriAI/litellm/pull/18665)
+* @chetanchoudhary-sumo made their first contribution in [PR #18587](https://github.com/BerriAI/litellm/pull/18587)
+* @pascalwhoop made their first contribution in [PR #13328](https://github.com/BerriAI/litellm/pull/13328)
+* @orgersh92 made their first contribution in [PR #18652](https://github.com/BerriAI/litellm/pull/18652)
+* @DevajMody made their first contribution in [PR #18497](https://github.com/BerriAI/litellm/pull/18497)
+* @matt-greathouse made their first contribution in [PR #18247](https://github.com/BerriAI/litellm/pull/18247)
+* @emerzon made their first contribution in [PR #18290](https://github.com/BerriAI/litellm/pull/18290)
+* @Eric84626 made their first contribution in [PR #18281](https://github.com/BerriAI/litellm/pull/18281)
+* @LukasdeBoer made their first contribution in [PR #18055](https://github.com/BerriAI/litellm/pull/18055)
+* @LingXuanYin made their first contribution in [PR #18513](https://github.com/BerriAI/litellm/pull/18513)
+* @krisxia0506 made their first contribution in [PR #18698](https://github.com/BerriAI/litellm/pull/18698)
+* @LouisShark made their first contribution in [PR #18414](https://github.com/BerriAI/litellm/pull/18414)
+
+---
+
+## Full Changelog
+
+**[View complete changelog on GitHub](https://github.com/BerriAI/litellm/compare/v1.80.11.rc.1...v1.80.14.rc.1)**
+
+
diff --git a/litellm/files/main.py b/litellm/files/main.py
index a7c82290c29..913ec84626d 100644
--- a/litellm/files/main.py
+++ b/litellm/files/main.py
@@ -8,6 +8,7 @@ https://platform.openai.com/docs/api-reference/files
import asyncio
import contextvars
import os
+import time
from functools import partial
from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast
@@ -60,7 +61,7 @@ async def acreate_file(
file: FileTypes,
purpose: Literal["assistants", "batch", "fine-tune"],
expires_after: Optional[FileExpiresAfter] = None,
- custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -105,7 +106,7 @@ def create_file(
file: FileTypes,
purpose: Literal["assistants", "batch", "fine-tune"],
expires_after: Optional[FileExpiresAfter] = None,
- custom_llm_provider: Optional[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"]] = None,
+ custom_llm_provider: Optional[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "manus"]] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -274,7 +275,7 @@ def create_file(
)
else:
raise litellm.exceptions.BadRequestError(
- message="LiteLLM doesn't support {} for 'create_file'. Only ['openai', 'azure', 'vertex_ai'] are supported.".format(
+ message="LiteLLM doesn't support {} for 'create_file'. Only ['openai', 'azure', 'vertex_ai', 'manus'] are supported.".format(
custom_llm_provider
),
model="n/a",
@@ -293,7 +294,7 @@ def create_file(
@client
async def afile_retrieve(
file_id: str,
- custom_llm_provider: Literal["openai", "azure", "hosted_vllm"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "hosted_vllm", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -334,7 +335,7 @@ async def afile_retrieve(
@client
def file_retrieve(
file_id: str,
- custom_llm_provider: Literal["openai", "azure", "hosted_vllm"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "hosted_vllm", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -428,18 +429,60 @@ def file_retrieve(
file_id=file_id,
)
else:
- raise litellm.exceptions.BadRequestError(
- message="LiteLLM doesn't support {} for 'file_retrieve'. Only 'openai' and 'azure' are supported.".format(
- custom_llm_provider
- ),
- model="n/a",
- llm_provider=custom_llm_provider,
- response=httpx.Response(
- status_code=400,
- content="Unsupported provider",
- request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
- ),
+ # Try using provider config pattern (for Manus, Bedrock, etc.)
+ provider_config = ProviderConfigManager.get_provider_files_config(
+ model="",
+ provider=LlmProviders(custom_llm_provider),
)
+ if provider_config is not None:
+ litellm_params_dict = get_litellm_params(**kwargs)
+ litellm_params_dict["api_key"] = optional_params.api_key
+ litellm_params_dict["api_base"] = optional_params.api_base
+
+ logging_obj = kwargs.get("litellm_logging_obj")
+ if logging_obj is None:
+ from litellm.litellm_core_utils.litellm_logging import (
+ Logging as LiteLLMLoggingObj,
+ )
+ logging_obj = LiteLLMLoggingObj(
+ model="",
+ messages=[],
+ stream=False,
+ call_type="afile_retrieve" if _is_async else "file_retrieve",
+ start_time=time.time(),
+ litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())),
+ function_id=str(kwargs.get("id") or ""),
+ )
+
+ client = kwargs.get("client")
+ response = base_llm_http_handler.retrieve_file(
+ file_id=file_id,
+ provider_config=provider_config,
+ litellm_params=litellm_params_dict,
+ headers=extra_headers or {},
+ logging_obj=logging_obj,
+ _is_async=_is_async,
+ client=(
+ client
+ if client is not None
+ and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
+ else None
+ ),
+ timeout=timeout,
+ )
+ else:
+ raise litellm.exceptions.BadRequestError(
+ message="LiteLLM doesn't support {} for 'file_retrieve'. Only 'openai', 'azure', and 'manus' are supported.".format(
+ custom_llm_provider
+ ),
+ model="n/a",
+ llm_provider=custom_llm_provider,
+ response=httpx.Response(
+ status_code=400,
+ content="Unsupported provider",
+ request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
+ ),
+ )
return cast(FileObject, response)
except Exception as e:
@@ -450,7 +493,7 @@ def file_retrieve(
@client
async def afile_delete(
file_id: str,
- custom_llm_provider: Literal["openai", "azure"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -494,7 +537,7 @@ async def afile_delete(
def file_delete(
file_id: str,
model: Optional[str] = None,
- custom_llm_provider: Union[Literal["openai", "azure"], str] = "openai",
+ custom_llm_provider: Union[Literal["openai", "azure", "manus"], str] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -596,18 +639,58 @@ def file_delete(
litellm_params=litellm_params_dict,
)
else:
- raise litellm.exceptions.BadRequestError(
- message="LiteLLM doesn't support {} for 'delete_batch'. Only 'openai' is supported.".format(
- custom_llm_provider
- ),
- model="n/a",
- llm_provider=custom_llm_provider,
- response=httpx.Response(
- status_code=400,
- content="Unsupported provider",
- request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
- ),
+ # Try using provider config pattern (for Manus, Bedrock, etc.)
+ provider_config = ProviderConfigManager.get_provider_files_config(
+ model="",
+ provider=LlmProviders(custom_llm_provider),
)
+ if provider_config is not None:
+ litellm_params_dict["api_key"] = optional_params.api_key
+ litellm_params_dict["api_base"] = optional_params.api_base
+
+ logging_obj = kwargs.get("litellm_logging_obj")
+ if logging_obj is None:
+ from litellm.litellm_core_utils.litellm_logging import (
+ Logging as LiteLLMLoggingObj,
+ )
+ logging_obj = LiteLLMLoggingObj(
+ model="",
+ messages=[],
+ stream=False,
+ call_type="afile_delete" if _is_async else "file_delete",
+ start_time=time.time(),
+ litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())),
+ function_id=str(kwargs.get("id") or ""),
+ )
+
+ response = base_llm_http_handler.delete_file(
+ file_id=file_id,
+ provider_config=provider_config,
+ litellm_params=litellm_params_dict,
+ headers=extra_headers or {},
+ logging_obj=logging_obj,
+ _is_async=_is_async,
+ client=(
+ client
+ if client is not None
+ and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
+ else None
+ ),
+ timeout=timeout,
+ )
+ else:
+ raise litellm.exceptions.BadRequestError(
+ message="LiteLLM doesn't support {} for 'file_delete'. Only 'openai', 'azure', and 'manus' are supported.".format(
+ custom_llm_provider
+ ),
+ model="n/a",
+ llm_provider=custom_llm_provider,
+ response=httpx.Response(
+ status_code=400,
+ content="Unsupported provider",
+ request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
+ ),
+ )
return cast(FileDeleted, response)
except Exception as e:
raise e
@@ -616,7 +699,7 @@ def file_delete(
# List files
@client
async def afile_list(
- custom_llm_provider: Literal["openai", "azure"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "manus"] = "openai",
purpose: Optional[str] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
@@ -657,7 +740,7 @@ async def afile_list(
@client
def file_list(
- custom_llm_provider: Literal["openai", "azure"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "manus"] = "openai",
purpose: Optional[str] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
@@ -687,7 +770,50 @@ def file_list(
timeout = 600.0
_is_async = kwargs.pop("is_async", False) is True
- if custom_llm_provider in OPENAI_COMPATIBLE_BATCH_AND_FILES_PROVIDERS:
+
+ # Check if provider has a custom files config (e.g., Manus, Bedrock, Vertex AI)
+ provider_config = ProviderConfigManager.get_provider_files_config(
+ model="",
+ provider=LlmProviders(custom_llm_provider),
+ )
+ if provider_config is not None:
+ litellm_params_dict = get_litellm_params(**kwargs)
+ litellm_params_dict["api_key"] = optional_params.api_key
+ litellm_params_dict["api_base"] = optional_params.api_base
+
+ logging_obj = kwargs.get("litellm_logging_obj")
+ if logging_obj is None:
+ from litellm.litellm_core_utils.litellm_logging import (
+ Logging as LiteLLMLoggingObj,
+ )
+ logging_obj = LiteLLMLoggingObj(
+ model="",
+ messages=[],
+ stream=False,
+ call_type="afile_list" if _is_async else "file_list",
+ start_time=time.time(),
+ litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())),
+ function_id=str(kwargs.get("id", "")),
+ )
+
+ client = kwargs.get("client")
+ response = base_llm_http_handler.list_files(
+ purpose=purpose,
+ provider_config=provider_config,
+ litellm_params=litellm_params_dict,
+ headers=extra_headers or {},
+ logging_obj=logging_obj,
+ _is_async=_is_async,
+ client=(
+ client
+ if client is not None
+ and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
+ else None
+ ),
+ timeout=timeout,
+ )
+ return response
+ elif custom_llm_provider in OPENAI_COMPATIBLE_BATCH_AND_FILES_PROVIDERS:
# for deepinfra/perplexity/anyscale/groq we check in get_llm_provider and pass in the api base from there
api_base = (
optional_params.api_base
@@ -752,7 +878,7 @@ def file_list(
)
else:
raise litellm.exceptions.BadRequestError(
- message="LiteLLM doesn't support {} for 'file_list'. Only 'openai' and 'azure' are supported.".format(
+ message="LiteLLM doesn't support {} for 'file_list'. Only 'openai', 'azure', and 'manus' are supported.".format(
custom_llm_provider
),
model="n/a",
@@ -771,7 +897,7 @@ def file_list(
@client
async def afile_content(
file_id: str,
- custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic"] = "openai",
+ custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@@ -816,7 +942,7 @@ def file_content(
file_id: str,
model: Optional[str] = None,
custom_llm_provider: Optional[
- Union[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic"], str]
+ Union[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic", "manus"], str]
] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
@@ -977,7 +1103,7 @@ def file_content(
)
else:
raise litellm.exceptions.BadRequestError(
- message="LiteLLM doesn't support {} for 'custom_llm_provider'. Supported providers are 'openai', 'azure', 'vertex_ai', 'bedrock'.".format(
+ message="LiteLLM doesn't support {} for 'file_content'. Supported providers are 'openai', 'azure', 'vertex_ai', 'bedrock', 'manus'.".format(
custom_llm_provider
),
model="n/a",
diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py
index c32a7b75c51..bb9d6166b01 100644
--- a/litellm/integrations/prometheus.py
+++ b/litellm/integrations/prometheus.py
@@ -875,16 +875,7 @@ class PrometheusLogger(CustomLogger):
# Include top-level metadata fields (excluding nested dictionaries)
# This allows accessing fields like requester_ip_address from top-level metadata
top_level_metadata = standard_logging_payload.get("metadata", {})
- top_level_fields: Dict[str, Any] = {}
- if isinstance(top_level_metadata, dict):
- top_level_fields = {
- k: v
- for k, v in top_level_metadata.items()
- if not isinstance(v, dict) # Exclude nested dicts to avoid conflicts
- }
-
combined_metadata: Dict[str, Any] = {
- **top_level_fields, # Include top-level fields first
**(_requester_metadata if _requester_metadata else {}),
**(user_api_key_auth_metadata if user_api_key_auth_metadata else {}),
}
diff --git a/litellm/llms/base_llm/files/transformation.py b/litellm/llms/base_llm/files/transformation.py
index 35b76479cdc..7b0a1868f19 100644
--- a/litellm/llms/base_llm/files/transformation.py
+++ b/litellm/llms/base_llm/files/transformation.py
@@ -2,11 +2,14 @@ from abc import ABC, abstractmethod
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union
import httpx
+from openai.types.file_deleted import FileDeleted
from litellm.proxy._types import UserAPIKeyAuth
+from litellm.types.files import TwoStepFileUploadConfig
from litellm.types.llms.openai import (
AllMessageValues,
CreateFileRequest,
+ FileContentRequest,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
OpenAIFilesPurpose,
@@ -75,7 +78,15 @@ class BaseFilesConfig(BaseConfig):
create_file_data: CreateFileRequest,
optional_params: dict,
litellm_params: dict,
- ) -> Union[dict, str, bytes]:
+ ) -> Union[dict, str, bytes, "TwoStepFileUploadConfig"]:
+ """
+ Transform OpenAI-style file creation request into provider-specific format.
+
+ Returns:
+ - dict: For pre-signed single-step uploads (e.g., Bedrock S3)
+ - str/bytes: For traditional file uploads
+ - TwoStepFileUploadConfig: For two-step upload process (e.g., Manus, GCS)
+ """
pass
@abstractmethod
@@ -88,6 +99,86 @@ class BaseFilesConfig(BaseConfig):
) -> OpenAIFileObject:
pass
+ @abstractmethod
+ def transform_retrieve_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Transform file retrieve request into provider-specific format."""
+ pass
+
+ @abstractmethod
+ def transform_retrieve_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> OpenAIFileObject:
+ """Transform file retrieve response into OpenAI format."""
+ pass
+
+ @abstractmethod
+ def transform_delete_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Transform file delete request into provider-specific format."""
+ pass
+
+ @abstractmethod
+ def transform_delete_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> "FileDeleted":
+ """Transform file delete response into OpenAI format."""
+ pass
+
+ @abstractmethod
+ def transform_list_files_request(
+ self,
+ purpose: Optional[str],
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Transform file list request into provider-specific format."""
+ pass
+
+ @abstractmethod
+ def transform_list_files_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> List[OpenAIFileObject]:
+ """Transform file list response into OpenAI format."""
+ pass
+
+ @abstractmethod
+ def transform_file_content_request(
+ self,
+ file_content_request: "FileContentRequest",
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Transform file content request into provider-specific format."""
+ pass
+
+ @abstractmethod
+ def transform_file_content_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> "HttpxBinaryResponseContent":
+ """Transform file content response into OpenAI format."""
+ pass
+
def transform_request(
self,
model: str,
diff --git a/litellm/llms/bedrock/base_aws_llm.py b/litellm/llms/bedrock/base_aws_llm.py
index 18e9deb53b0..e9cea23ea4a 100644
--- a/litellm/llms/bedrock/base_aws_llm.py
+++ b/litellm/llms/bedrock/base_aws_llm.py
@@ -74,41 +74,6 @@ class BaseAWSLLM:
"aws_external_id",
]
- def _get_ssl_verify(self):
- """
- Get SSL verification setting for boto3 clients.
-
- This ensures that custom CA certificates are properly used for all AWS API calls,
- including STS and Bedrock services.
-
- Returns:
- Union[bool, str]: SSL verification setting - False to disable, True to enable,
- or a string path to a CA bundle file
- """
- import litellm
- from litellm.secret_managers.main import str_to_bool
-
- # Check environment variable first (highest priority)
- ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify)
-
- # Convert string "False"/"True" to boolean
- if isinstance(ssl_verify, str):
- # Check if it's a file path
- if os.path.exists(ssl_verify):
- return ssl_verify
- # Otherwise try to convert to boolean
- ssl_verify_bool = str_to_bool(ssl_verify)
- if ssl_verify_bool is not None:
- ssl_verify = ssl_verify_bool
-
- # Check SSL_CERT_FILE environment variable for custom CA bundle
- if ssl_verify is True or ssl_verify == "True":
- ssl_cert_file = os.getenv("SSL_CERT_FILE")
- if ssl_cert_file and os.path.exists(ssl_cert_file):
- return ssl_cert_file
-
- return ssl_verify
-
def get_cache_key(self, credential_args: Dict[str, Optional[str]]) -> str:
"""
Generate a unique cache key based on the credential arguments.
@@ -604,7 +569,6 @@ class BaseAWSLLM:
"sts",
region_name=aws_region_name,
endpoint_url=sts_endpoint,
- verify=self._get_ssl_verify(),
)
# https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html
@@ -661,7 +625,7 @@ class BaseAWSLLM:
# Create an STS client without credentials
with tracer.trace("boto3.client(sts) for manual IRSA"):
- sts_client = boto3.client("sts", region_name=region, verify=self._get_ssl_verify())
+ sts_client = boto3.client("sts", region_name=region)
# Manually assume the IRSA role with the session name
verbose_logger.debug(
@@ -684,7 +648,6 @@ class BaseAWSLLM:
aws_access_key_id=irsa_creds["AccessKeyId"],
aws_secret_access_key=irsa_creds["SecretAccessKey"],
aws_session_token=irsa_creds["SessionToken"],
- verify=self._get_ssl_verify(),
)
# Get current caller identity for debugging
@@ -723,7 +686,7 @@ class BaseAWSLLM:
verbose_logger.debug("Same account role assumption, using automatic IRSA")
with tracer.trace("boto3.client(sts) with automatic IRSA"):
- sts_client = boto3.client("sts", region_name=region, verify=self._get_ssl_verify())
+ sts_client = boto3.client("sts", region_name=region)
# Get current caller identity for debugging
try:
@@ -846,7 +809,7 @@ class BaseAWSLLM:
# This allows the web identity token to work automatically
if aws_access_key_id is None and aws_secret_access_key is None:
with tracer.trace("boto3.client(sts)"):
- sts_client = boto3.client("sts", verify=self._get_ssl_verify())
+ sts_client = boto3.client("sts")
else:
with tracer.trace("boto3.client(sts)"):
sts_client = boto3.client(
@@ -854,7 +817,6 @@ class BaseAWSLLM:
aws_access_key_id=aws_access_key_id,
aws_secret_access_key=aws_secret_access_key,
aws_session_token=aws_session_token,
- verify=self._get_ssl_verify(),
)
assume_role_params = {
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index f4b5de8f7c0..d62a8bae425 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -260,7 +260,7 @@ def init_bedrock_client(
status_code=401,
)
- sts_client = boto3.client("sts", verify=ssl_verify)
+ sts_client = boto3.client("sts")
# https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html
# https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts/client/assume_role_with_web_identity.html
diff --git a/litellm/llms/bedrock/files/handler.py b/litellm/llms/bedrock/files/handler.py
index 0350271dc44..d6177e090d5 100644
--- a/litellm/llms/bedrock/files/handler.py
+++ b/litellm/llms/bedrock/files/handler.py
@@ -142,7 +142,6 @@ class BedrockFilesHandler(BaseAWSLLM):
aws_secret_access_key=credentials.secret_key,
aws_session_token=credentials.token,
region_name=aws_region_name,
- verify=self._get_ssl_verify(),
)
# Download file from S3
diff --git a/litellm/llms/bedrock/files/transformation.py b/litellm/llms/bedrock/files/transformation.py
index 0a95cf9168f..fdcbe1a8242 100644
--- a/litellm/llms/bedrock/files/transformation.py
+++ b/litellm/llms/bedrock/files/transformation.py
@@ -1,12 +1,14 @@
import json
import os
import time
-from litellm._uuid import uuid
from typing import Any, Dict, List, Optional, Tuple, Union
+import httpx
from httpx import Headers, Response
+from openai.types.file_deleted import FileDeleted
from litellm._logging import verbose_logger
+from litellm._uuid import uuid
from litellm.files.utils import FilesAPIUtils
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@@ -18,6 +20,7 @@ from litellm.types.llms.openai import (
AllMessageValues,
CreateFileRequest,
FileTypes,
+ HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
PathLike,
@@ -539,6 +542,70 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
status_code=status_code, message=error_message, headers=headers
)
+ def transform_retrieve_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("BedrockFilesConfig does not support file retrieval")
+
+ def transform_retrieve_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> OpenAIFileObject:
+ raise NotImplementedError("BedrockFilesConfig does not support file retrieval")
+
+ def transform_delete_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("BedrockFilesConfig does not support file deletion")
+
+ def transform_delete_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> FileDeleted:
+ raise NotImplementedError("BedrockFilesConfig does not support file deletion")
+
+ def transform_list_files_request(
+ self,
+ purpose: Optional[str],
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("BedrockFilesConfig does not support file listing")
+
+ def transform_list_files_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> List[OpenAIFileObject]:
+ raise NotImplementedError("BedrockFilesConfig does not support file listing")
+
+ def transform_file_content_request(
+ self,
+ file_content_request,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("BedrockFilesConfig does not support file content retrieval")
+
+ def transform_file_content_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> HttpxBinaryResponseContent:
+ raise NotImplementedError("BedrockFilesConfig does not support file content retrieval")
+
class BedrockJsonlFilesTransformation:
"""
diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py
index ea740400664..1da6e61252f 100644
--- a/litellm/llms/custom_httpx/llm_http_handler.py
+++ b/litellm/llms/custom_httpx/llm_http_handler.py
@@ -14,6 +14,7 @@ from typing import (
)
import httpx # type: ignore
+from openai.types.file_deleted import FileDeleted
import litellm
import litellm.litellm_core_utils
@@ -71,6 +72,7 @@ from litellm.types.containers.main import (
ContainerObject,
DeleteContainerResult,
)
+from litellm.types.files import TwoStepFileUploadConfig
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
@@ -82,6 +84,7 @@ from litellm.types.llms.anthropic_skills import (
from litellm.types.llms.openai import (
CreateBatchRequest,
CreateFileRequest,
+ FileContentRequest,
HttpxBinaryResponseContent,
OpenAIFileObject,
ResponseInputParam,
@@ -2782,6 +2785,38 @@ class BaseLLMHTTPHandler:
logging_obj=logging_obj,
)
+ def _extract_upload_url_from_response(
+ self,
+ response: httpx.Response,
+ upload_url_location: str,
+ upload_url_key: str = "upload_url",
+ ) -> tuple[Optional[str], Optional[dict]]:
+ """
+ Extract upload URL from initial file creation response.
+
+ Args:
+ response: HTTP response from initial file creation request
+ upload_url_location: Where to find URL ('headers' or 'body')
+ upload_url_key: Key name for URL in response body (default: 'upload_url')
+
+ Returns:
+ Tuple of (upload_url, response_data)
+ - upload_url: The extracted upload URL, or None if not found
+ - response_data: Parsed response body (for 'body' location), or None
+ """
+ if upload_url_location == "headers":
+ # Google Cloud Storage style - URL in X-Goog-Upload-URL header
+ upload_url = response.headers.get("X-Goog-Upload-URL")
+ return upload_url, None
+ else:
+ # Response body style (e.g., Manus, S3 presigned URLs)
+ try:
+ response_data = response.json()
+ upload_url = response_data.get(upload_url_key)
+ return upload_url, response_data if upload_url else None
+ except Exception:
+ return None, None
+
def create_file(
self,
create_file_data: CreateFileRequest,
@@ -2844,14 +2879,58 @@ class BaseLLMHTTPHandler:
else:
sync_httpx_client = client
- if isinstance(transformed_request, dict) and "method" in transformed_request:
+ if isinstance(transformed_request, dict) and "initial_request" in transformed_request:
+ # Handle two-step uploads (TwoStepFileUploadConfig)
+ # Used by providers like Manus, Google Cloud Storage
+ try:
+ # Step 1: Initial request to get upload URL
+ initial_response = sync_httpx_client.post(
+ url=api_base,
+ headers={
+ **headers,
+ **transformed_request["initial_request"]["headers"],
+ },
+ data=json.dumps(transformed_request["initial_request"]["data"]),
+ timeout=timeout,
+ )
+
+ # Extract upload URL from response
+ upload_url, initial_response_data = self._extract_upload_url_from_response(
+ response=initial_response,
+ upload_url_location=transformed_request.get("upload_url_location", "headers"),
+ upload_url_key=transformed_request.get("upload_url_key", "upload_url"),
+ )
+
+ if not upload_url:
+ raise ValueError("Failed to get upload URL from initial request")
+
+ # Step 2: Upload the actual file
+ upload_method = transformed_request["upload_request"].get("method", "POST").lower()
+ upload_response = getattr(sync_httpx_client, upload_method)(
+ url=upload_url,
+ headers=transformed_request["upload_request"]["headers"],
+ data=transformed_request["upload_request"]["data"],
+ timeout=timeout,
+ )
+
+ # Store initial response for transformation
+ if initial_response_data:
+ litellm_params["initial_file_response"] = initial_response_data
+ except Exception as e:
+ raise self._handle_error(
+ e=e,
+ provider_config=provider_config,
+ )
+ elif isinstance(transformed_request, dict) and "method" in transformed_request and "initial_request" not in transformed_request:
# Handle pre-signed requests (e.g., from Bedrock S3 uploads)
+ # Type narrowing: this is a plain dict, not TwoStepFileUploadConfig
+ presigned_request = cast(Dict[str, Any], transformed_request)
upload_response = getattr(
- sync_httpx_client, transformed_request["method"].lower()
+ sync_httpx_client, presigned_request["method"].lower()
)(
- url=transformed_request["url"],
- headers=transformed_request["headers"],
- data=transformed_request["data"],
+ url=presigned_request["url"],
+ headers=presigned_request["headers"],
+ data=presigned_request["data"],
timeout=timeout,
)
elif isinstance(transformed_request, str) or isinstance(
@@ -2879,36 +2958,7 @@ class BaseLLMHTTPHandler:
timeout=timeout,
)
else:
- try:
- # Step 1: Initial request to get upload URL
- initial_response = sync_httpx_client.post(
- url=api_base,
- headers={
- **headers,
- **transformed_request["initial_request"]["headers"],
- },
- data=json.dumps(transformed_request["initial_request"]["data"]),
- timeout=timeout,
- )
-
- # Extract upload URL from response headers
- upload_url = initial_response.headers.get("X-Goog-Upload-URL")
-
- if not upload_url:
- raise ValueError("Failed to get upload URL from initial request")
-
- # Step 2: Upload the actual file
- upload_response = sync_httpx_client.post(
- url=upload_url,
- headers=transformed_request["upload_request"]["headers"],
- data=transformed_request["upload_request"]["data"],
- timeout=timeout,
- )
- except Exception as e:
- raise self._handle_error(
- e=e,
- provider_config=provider_config,
- )
+ raise ValueError(f"Unsupported transformed_request type: {type(transformed_request)}")
# Store the upload URL in litellm_params for the transformation method
litellm_params_with_url = dict(litellm_params)
@@ -2923,7 +2973,7 @@ class BaseLLMHTTPHandler:
async def async_create_file(
self,
- transformed_request: Union[bytes, str, dict],
+ transformed_request: Union[bytes, str, dict, "TwoStepFileUploadConfig"],
litellm_params: dict,
provider_config: BaseFilesConfig,
headers: dict,
@@ -2955,14 +3005,59 @@ class BaseLLMHTTPHandler:
},
)
- if isinstance(transformed_request, dict) and "method" in transformed_request:
+ if isinstance(transformed_request, dict) and "initial_request" in transformed_request:
+ # Handle two-step uploads (TwoStepFileUploadConfig)
+ # Used by providers like Manus, Google Cloud Storage
+ try:
+ # Step 1: Initial request to get upload URL
+ initial_response = await async_httpx_client.post(
+ url=api_base,
+ headers={
+ **headers,
+ **transformed_request["initial_request"]["headers"],
+ },
+ data=json.dumps(transformed_request["initial_request"]["data"]),
+ timeout=timeout,
+ )
+
+ # Extract upload URL from response
+ upload_url, initial_response_data = self._extract_upload_url_from_response(
+ response=initial_response,
+ upload_url_location=transformed_request.get("upload_url_location", "headers"),
+ upload_url_key=transformed_request.get("upload_url_key", "upload_url"),
+ )
+
+ if not upload_url:
+ raise ValueError("Failed to get upload URL from initial request")
+
+ # Step 2: Upload the actual file
+ upload_method = transformed_request["upload_request"].get("method", "POST").lower()
+ upload_response = await getattr(async_httpx_client, upload_method)(
+ url=upload_url,
+ headers=transformed_request["upload_request"]["headers"],
+ data=transformed_request["upload_request"]["data"],
+ timeout=timeout,
+ )
+
+ # Store initial response for transformation
+ if initial_response_data:
+ litellm_params["initial_file_response"] = initial_response_data
+ except Exception as e:
+ verbose_logger.exception(f"Error creating file: {e}")
+ raise self._handle_error(
+ e=e,
+ provider_config=provider_config,
+ )
+ elif isinstance(transformed_request, dict) and "method" in transformed_request and "initial_request" not in transformed_request:
# Handle pre-signed requests (e.g., from Bedrock S3 uploads)
+ # Type narrowing: this is a plain dict, not TwoStepFileUploadConfig
+ presigned_request = cast(Dict[str, Any], transformed_request)
upload_response = await getattr(
- async_httpx_client, transformed_request["method"].lower()
+ async_httpx_client, presigned_request["method"].lower()
)(
- url=transformed_request["url"],
- headers=transformed_request["headers"],
- data=transformed_request["data"],
+ url=presigned_request["url"],
+ headers=presigned_request["headers"],
+ data=presigned_request["data"],
timeout=timeout,
)
elif isinstance(transformed_request, str) or isinstance(
@@ -2990,37 +3085,7 @@ class BaseLLMHTTPHandler:
timeout=timeout,
)
else:
- try:
- # Step 1: Initial request to get upload URL
- initial_response = await async_httpx_client.post(
- url=api_base,
- headers={
- **headers,
- **transformed_request["initial_request"]["headers"],
- },
- data=json.dumps(transformed_request["initial_request"]["data"]),
- timeout=timeout,
- )
-
- # Extract upload URL from response headers
- upload_url = initial_response.headers.get("X-Goog-Upload-URL")
-
- if not upload_url:
- raise ValueError("Failed to get upload URL from initial request")
-
- # Step 2: Upload the actual file
- upload_response = await async_httpx_client.post(
- url=upload_url,
- headers=transformed_request["upload_request"]["headers"],
- data=transformed_request["upload_request"]["data"],
- timeout=timeout,
- )
- except Exception as e:
- verbose_logger.exception(f"Error creating file: {e}")
- raise self._handle_error(
- e=e,
- provider_config=provider_config,
- )
+ raise ValueError(f"Unsupported transformed_request type: {type(transformed_request)}")
return provider_config.transform_create_file_response(
model=None,
@@ -3734,29 +3799,525 @@ class BaseLLMHTTPHandler:
logging_obj=logging_obj,
)
- def list_files(self):
+ def retrieve_file(
+ self,
+ file_id: str,
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ _is_async: bool = False,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> Union[OpenAIFileObject, Coroutine[Any, Any, OpenAIFileObject]]:
"""
- Lists all files
+ Retrieve file metadata by ID
"""
- pass
+ if _is_async:
+ return self.async_retrieve_file(
+ file_id=file_id,
+ provider_config=provider_config,
+ litellm_params=litellm_params,
+ headers=headers,
+ logging_obj=logging_obj,
+ client=client,
+ timeout=timeout,
+ )
- def delete_file(self):
- """
- Deletes a file
- """
- pass
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client()
+ else:
+ sync_httpx_client = client
- def retrieve_file(self):
- """
- Returns the metadata of the file
- """
- pass
+ # Get URL and params from provider config
+ url, params = provider_config.transform_retrieve_file_request(
+ file_id=file_id,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
- def retrieve_file_content(self):
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "file_id": file_id,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_retrieve_file_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ async def async_retrieve_file(
+ self,
+ file_id: str,
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> OpenAIFileObject:
"""
- Returns the content of the file
+ Async retrieve file metadata by ID
"""
- pass
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=provider_config.custom_llm_provider
+ )
+ else:
+ async_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_retrieve_file_request(
+ file_id=file_id,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "file_id": file_id,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_retrieve_file_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ def delete_file(
+ self,
+ file_id: str,
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ _is_async: bool = False,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> Union["FileDeleted", Coroutine[Any, Any, "FileDeleted"]]:
+ """
+ Delete a file by ID
+ """
+ if _is_async:
+ return self.async_delete_file(
+ file_id=file_id,
+ provider_config=provider_config,
+ litellm_params=litellm_params,
+ headers=headers,
+ logging_obj=logging_obj,
+ client=client,
+ timeout=timeout,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client()
+ else:
+ sync_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_delete_file_request(
+ file_id=file_id,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "file_id": file_id,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.delete(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_delete_file_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ async def async_delete_file(
+ self,
+ file_id: str,
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> "FileDeleted":
+ """
+ Async delete a file by ID
+ """
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=provider_config.custom_llm_provider
+ )
+ else:
+ async_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_delete_file_request(
+ file_id=file_id,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "file_id": file_id,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.delete(
+ url=url, headers=headers, params=params, timeout=timeout
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_delete_file_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ def list_files(
+ self,
+ purpose: Optional[str],
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ _is_async: bool = False,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> Union[List[OpenAIFileObject], Coroutine[Any, Any, List[OpenAIFileObject]]]:
+ """
+ List all files
+ """
+ if _is_async:
+ return self.async_list_files(
+ purpose=purpose,
+ provider_config=provider_config,
+ litellm_params=litellm_params,
+ headers=headers,
+ logging_obj=logging_obj,
+ client=client,
+ timeout=timeout,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client()
+ else:
+ sync_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_list_files_request(
+ purpose=purpose,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "purpose": purpose,
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_list_files_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ async def async_list_files(
+ self,
+ purpose: Optional[str],
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> List[OpenAIFileObject]:
+ """
+ Async list all files
+ """
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=provider_config.custom_llm_provider
+ )
+ else:
+ async_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_list_files_request(
+ purpose=purpose,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "purpose": purpose,
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_list_files_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ def retrieve_file_content(
+ self,
+ file_content_request: "FileContentRequest",
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ _is_async: bool = False,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> Union["HttpxBinaryResponseContent", Coroutine[Any, Any, "HttpxBinaryResponseContent"]]:
+ """
+ Retrieve file content by ID
+ """
+ if _is_async:
+ return self.async_retrieve_file_content(
+ file_content_request=file_content_request,
+ provider_config=provider_config,
+ litellm_params=litellm_params,
+ headers=headers,
+ logging_obj=logging_obj,
+ client=client,
+ timeout=timeout,
+ )
+
+ if client is None or not isinstance(client, HTTPHandler):
+ sync_httpx_client = _get_httpx_client()
+ else:
+ sync_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_file_content_request(
+ file_content_request=file_content_request,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "file_id": file_content_request.get("file_id"),
+ },
+ )
+
+ try:
+ response = sync_httpx_client.get(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_file_content_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ async def async_retrieve_file_content(
+ self,
+ file_content_request: "FileContentRequest",
+ provider_config: BaseFilesConfig,
+ litellm_params: dict,
+ headers: dict,
+ logging_obj: LiteLLMLoggingObj,
+ client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
+ timeout: Optional[Union[float, httpx.Timeout]] = None,
+ ) -> "HttpxBinaryResponseContent":
+ """
+ Async retrieve file content by ID
+ """
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ async_httpx_client = get_async_httpx_client(
+ llm_provider=provider_config.custom_llm_provider
+ )
+ else:
+ async_httpx_client = client
+
+ # Get URL and params from provider config
+ url, params = provider_config.transform_file_content_request(
+ file_content_request=file_content_request,
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ # Validate environment and get headers
+ headers = provider_config.validate_environment(
+ api_key=litellm_params.get("api_key"),
+ headers=headers,
+ model="",
+ messages=[],
+ optional_params={},
+ litellm_params=litellm_params,
+ )
+
+ logging_obj.pre_call(
+ input="",
+ api_key="",
+ additional_args={
+ "api_base": url,
+ "headers": headers,
+ "file_id": file_content_request.get("file_id"),
+ },
+ )
+
+ try:
+ response = await async_httpx_client.get(
+ url=url, headers=headers, params=params
+ )
+ except Exception as e:
+ raise self._handle_error(e=e, provider_config=provider_config)
+
+ return provider_config.transform_file_content_response(
+ raw_response=response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
def _prepare_fake_stream_request(
self,
diff --git a/litellm/llms/gemini/common_utils.py b/litellm/llms/gemini/common_utils.py
index 30c5b4f17c5..e53829d3329 100644
--- a/litellm/llms/gemini/common_utils.py
+++ b/litellm/llms/gemini/common_utils.py
@@ -150,15 +150,6 @@ def get_api_key_from_env() -> Optional[str]:
return get_secret_str("GOOGLE_API_KEY") or get_secret_str("GEMINI_API_KEY")
-def get_vertex_api_key_from_env() -> Optional[str]:
- """
- Get API key from environment for Vertex AI.
- Checks VERTEXAI_API_KEY and VERTEX_API_KEY environment variables.
- This allows using Vertex AI with API keys instead of service account credentials.
- """
- return get_secret_str("VERTEXAI_API_KEY") or get_secret_str("VERTEX_API_KEY")
-
-
class GoogleAIStudioTokenCounter(BaseTokenCounter):
"""Token counter implementation for Google AI Studio provider."""
def should_use_token_counting_api(
diff --git a/litellm/llms/gemini/files/transformation.py b/litellm/llms/gemini/files/transformation.py
index e98e76dabc8..d9ebf69a97a 100644
--- a/litellm/llms/gemini/files/transformation.py
+++ b/litellm/llms/gemini/files/transformation.py
@@ -7,6 +7,7 @@ import time
from typing import List, Optional
import httpx
+from openai.types.file_deleted import FileDeleted
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
@@ -17,6 +18,7 @@ from litellm.llms.base_llm.files.transformation import (
from litellm.types.llms.gemini import GeminiCreateFilesResponseObject
from litellm.types.llms.openai import (
CreateFileRequest,
+ HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
)
@@ -171,3 +173,67 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig):
except Exception as e:
verbose_logger.exception(f"Error parsing file upload response: {str(e)}")
raise ValueError(f"Error parsing file upload response: {str(e)}")
+
+ def transform_retrieve_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file retrieval")
+
+ def transform_retrieve_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> OpenAIFileObject:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file retrieval")
+
+ def transform_delete_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file deletion")
+
+ def transform_delete_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> FileDeleted:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file deletion")
+
+ def transform_list_files_request(
+ self,
+ purpose: Optional[str],
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file listing")
+
+ def transform_list_files_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> List[OpenAIFileObject]:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file listing")
+
+ def transform_file_content_request(
+ self,
+ file_content_request,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file content retrieval")
+
+ def transform_file_content_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> HttpxBinaryResponseContent:
+ raise NotImplementedError("GoogleAIStudioFilesHandler does not support file content retrieval")
diff --git a/litellm/llms/manus/files/__init__.py b/litellm/llms/manus/files/__init__.py
new file mode 100644
index 00000000000..66d23ca0340
--- /dev/null
+++ b/litellm/llms/manus/files/__init__.py
@@ -0,0 +1,2 @@
+# Manus Files API implementation
+
diff --git a/litellm/llms/manus/files/transformation.py b/litellm/llms/manus/files/transformation.py
new file mode 100644
index 00000000000..a7965011969
--- /dev/null
+++ b/litellm/llms/manus/files/transformation.py
@@ -0,0 +1,439 @@
+"""
+Manus Files API implementation.
+
+Manus has an OpenAI-compatible Files API with some differences:
+- Uses API_KEY header instead of Authorization: Bearer
+- File upload is a two-step process:
+ 1. Create file record to get upload URL
+ 2. Upload file content to the upload URL
+
+Reference: https://open.manus.im/docs/openai-compatibility#file-management
+"""
+
+import time
+from typing import Any, Dict, List, Optional, Union
+
+import httpx
+from openai.types.file_deleted import FileDeleted
+
+import litellm
+from litellm._logging import verbose_logger
+from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
+from litellm.llms.base_llm.chat.transformation import BaseLLMException
+from litellm.llms.base_llm.files.transformation import (
+ BaseFilesConfig,
+ LiteLLMLoggingObj,
+)
+from litellm.llms.openai.common_utils import OpenAIError
+from litellm.secret_managers.main import get_secret_str
+from litellm.types.files import TwoStepFileUploadConfig, TwoStepFileUploadRequest
+from litellm.types.llms.openai import (
+ CreateFileRequest,
+ FileContentRequest,
+ HttpxBinaryResponseContent,
+ OpenAICreateFileRequestOptionalParams,
+ OpenAIFileObject,
+)
+from litellm.types.utils import LlmProviders
+
+MANUS_API_BASE = "https://api.manus.im"
+
+
+class ManusFilesConfig(BaseFilesConfig):
+ """
+ Configuration for Manus Files API.
+
+ Manus uses:
+ - API_KEY header for authentication (not Authorization: Bearer)
+ - Two-step file upload process
+ - Content-Type: application/json for all requests
+
+ Reference: https://open.manus.im/docs/openai-compatibility#file-management
+ """
+
+ def __init__(self):
+ pass
+
+ @property
+ def custom_llm_provider(self) -> LlmProviders:
+ return LlmProviders.MANUS
+
+ def validate_environment(
+ self,
+ headers: dict,
+ model: str,
+ messages: list,
+ optional_params: dict,
+ litellm_params: dict,
+ api_key: Optional[str] = None,
+ api_base: Optional[str] = None,
+ ) -> dict:
+ """
+ Validate environment and set up headers for Manus API.
+
+ Manus uses API_KEY header instead of Authorization: Bearer.
+ For file uploads, don't set Content-Type - httpx will set it for multipart.
+ """
+ api_key = (
+ api_key
+ or litellm.api_key
+ or get_secret_str("MANUS_API_KEY")
+ )
+
+ if not api_key:
+ raise ValueError(
+ "Manus API key is required. Set MANUS_API_KEY environment variable or pass api_key parameter."
+ )
+
+ # Manus uses API_KEY header, not Authorization: Bearer
+ # Manus requires Content-Type: application/json for all requests (even GET)
+ headers.update(
+ {
+ "API_KEY": api_key,
+ "Content-Type": "application/json",
+ }
+ )
+ return headers
+
+ def get_supported_openai_params(
+ self, model: str
+ ) -> List[OpenAICreateFileRequestOptionalParams]:
+ """
+ Return supported OpenAI file creation parameters for Manus.
+ Manus supports the standard 'purpose' parameter.
+ """
+ return ["purpose"]
+
+ def map_openai_params(
+ self,
+ non_default_params: dict,
+ optional_params: dict,
+ model: str,
+ drop_params: bool,
+ ) -> dict:
+ """
+ Map OpenAI parameters to Manus-specific parameters.
+ Manus is OpenAI-compatible, so no special mapping needed.
+ """
+ return optional_params
+
+ def get_complete_url(
+ self,
+ api_base: Optional[str],
+ api_key: Optional[str],
+ model: str,
+ optional_params: dict,
+ litellm_params: dict,
+ stream: Optional[bool] = None,
+ ) -> str:
+ """
+ Get the complete URL for Manus Files API endpoint.
+
+ Returns:
+ str: The full URL for the Manus /v1/files endpoint
+ """
+ api_base = (
+ api_base
+ or litellm.api_base
+ or get_secret_str("MANUS_API_BASE")
+ or MANUS_API_BASE
+ )
+
+ # Remove trailing slashes
+ api_base = api_base.rstrip("/")
+
+ # Manus API uses /v1/files endpoint
+ if api_base.endswith("/v1"):
+ return f"{api_base}/files"
+ return f"{api_base}/v1/files"
+
+ def get_error_class(
+ self,
+ error_message: str,
+ status_code: int,
+ headers: Union[dict, httpx.Headers],
+ ) -> BaseLLMException:
+ """
+ Return the appropriate error class for Manus API errors.
+ Uses OpenAIError since Manus is OpenAI-compatible.
+ """
+ return OpenAIError(
+ status_code=status_code,
+ message=error_message,
+ headers=headers,
+ )
+
+ def transform_create_file_request(
+ self,
+ model: str,
+ create_file_data: CreateFileRequest,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> TwoStepFileUploadConfig:
+ """
+ Transform OpenAI-style file creation request into Manus's two-step format.
+
+ Manus API spec (https://open.manus.im/docs/openai-compatibility#file-management):
+ 1. POST /v1/files with JSON {"filename": "..."} ā returns {"id": "...", "upload_url": "..."}
+ 2. PUT to upload_url with raw file content
+ """
+ # Extract file data
+ file_data = create_file_data.get("file")
+ if file_data is None:
+ raise ValueError("File data is required")
+
+ extracted_data = extract_file_data(file_data)
+ filename = extracted_data["filename"] or f"file_{int(time.time())}"
+ content = extracted_data["content"]
+
+ # Get API base URL
+ api_base = self.get_complete_url(
+ api_base=litellm_params.get("api_base"),
+ api_key=litellm_params.get("api_key"),
+ model=model,
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ )
+
+ # Get API key
+ api_key = (
+ litellm_params.get("api_key")
+ or litellm.api_key
+ or get_secret_str("MANUS_API_KEY")
+ )
+
+ if not api_key:
+ raise ValueError(
+ "Manus API key is required. Set MANUS_API_KEY environment variable or pass api_key parameter."
+ )
+
+ # Build typed two-step upload config
+ return TwoStepFileUploadConfig(
+ initial_request=TwoStepFileUploadRequest(
+ method="POST",
+ url=api_base,
+ headers={
+ "API_KEY": api_key,
+ "Content-Type": "application/json",
+ },
+ data={"filename": filename},
+ ),
+ upload_request=TwoStepFileUploadRequest(
+ method="PUT",
+ url="", # Will be populated from initial_request response
+ headers={},
+ data=content,
+ ),
+ upload_url_location="body",
+ upload_url_key="upload_url",
+ )
+
+ def transform_create_file_response(
+ self,
+ model: Optional[str],
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> OpenAIFileObject:
+ """
+ Transform Manus's file upload response into OpenAI-style FileObject.
+
+ For two-step uploads, the handler stores the initial response in litellm_params.
+ We need to return the file object from the initial POST, not the final PUT.
+
+ Manus initial response format:
+ {
+ "id": "file-abc123xyz",
+ "object": "file",
+ "filename": "document.pdf",
+ "status": "pending",
+ "upload_url": "https://...",
+ "upload_expires_at": "...",
+ "created_at": "..."
+ }
+ """
+ try:
+ # For two-step uploads, get the initial response from litellm_params
+ initial_response_data = litellm_params.get("initial_file_response")
+ if initial_response_data:
+ response_json = initial_response_data
+ else:
+ # Log raw response for debugging
+ verbose_logger.debug(f"Manus raw response text: {raw_response.text}")
+ response_json = raw_response.json()
+
+ verbose_logger.debug(f"Manus file response: {response_json}")
+
+ # Parse created_at timestamp
+ created_at_str = response_json.get("created_at", "")
+ if created_at_str:
+ try:
+ # Try parsing ISO format
+ created_at = int(
+ time.mktime(
+ time.strptime(
+ created_at_str.replace("Z", "+00:00")[:19],
+ "%Y-%m-%dT%H:%M:%S",
+ )
+ )
+ )
+ except (ValueError, TypeError):
+ created_at = int(time.time())
+ else:
+ created_at = int(time.time())
+
+ return OpenAIFileObject(
+ id=response_json.get("id", ""),
+ bytes=response_json.get("bytes", 0),
+ created_at=created_at,
+ filename=response_json.get("filename", ""),
+ object="file",
+ purpose=response_json.get("purpose", "assistants"),
+ status="uploaded", # After successful upload, status is uploaded
+ status_details=response_json.get("status_details"),
+ )
+ except Exception as e:
+ verbose_logger.exception(f"Error parsing Manus file response: {str(e)}")
+ raise ValueError(f"Error parsing Manus file response: {str(e)}")
+
+ def transform_retrieve_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Get URL and params for retrieving a file."""
+ api_base = self.get_complete_url(
+ api_base=litellm_params.get("api_base"),
+ api_key=litellm_params.get("api_key"),
+ model="",
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ )
+ return f"{api_base}/{file_id}", {}
+
+ def transform_retrieve_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> OpenAIFileObject:
+ """Transform retrieve file response."""
+ return self.transform_create_file_response(
+ model=None,
+ raw_response=raw_response,
+ logging_obj=logging_obj,
+ litellm_params=litellm_params,
+ )
+
+ def transform_delete_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Get URL and params for deleting a file."""
+ api_base = self.get_complete_url(
+ api_base=litellm_params.get("api_base"),
+ api_key=litellm_params.get("api_key"),
+ model="",
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ )
+ return f"{api_base}/{file_id}", {}
+
+ def transform_delete_file_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> FileDeleted:
+ """Transform delete file response."""
+ response_json = raw_response.json()
+ return FileDeleted(**response_json)
+
+ def transform_list_files_request(
+ self,
+ purpose: Optional[str],
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Get URL and params for listing files."""
+ api_base = self.get_complete_url(
+ api_base=litellm_params.get("api_base"),
+ api_key=litellm_params.get("api_key"),
+ model="",
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ )
+ params = {}
+ if purpose:
+ params["purpose"] = purpose
+ return api_base, params
+
+ def transform_list_files_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> List[OpenAIFileObject]:
+ """Transform list files response."""
+ response_json = raw_response.json()
+ files_data = response_json.get("data", [])
+ return [self._parse_file_dict(f) for f in files_data]
+
+ def _parse_file_dict(self, file_dict: Dict[str, Any]) -> OpenAIFileObject:
+ """Parse a file dict into OpenAIFileObject."""
+ created_at_str = file_dict.get("created_at", "")
+ if created_at_str:
+ try:
+ created_at = int(
+ time.mktime(
+ time.strptime(
+ created_at_str.replace("Z", "+00:00")[:19],
+ "%Y-%m-%dT%H:%M:%S",
+ )
+ )
+ )
+ except (ValueError, TypeError):
+ created_at = int(time.time())
+ else:
+ created_at = int(time.time())
+
+ return OpenAIFileObject(
+ id=file_dict.get("id", ""),
+ bytes=file_dict.get("bytes", 0),
+ created_at=created_at,
+ filename=file_dict.get("filename", ""),
+ object="file",
+ purpose=file_dict.get("purpose", "assistants"),
+ status=file_dict.get("status", "uploaded"),
+ status_details=file_dict.get("status_details"),
+ )
+
+ def transform_file_content_request(
+ self,
+ file_content_request: FileContentRequest,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ """Get URL and params for retrieving file content."""
+ file_id = file_content_request.get("file_id")
+ api_base = self.get_complete_url(
+ api_base=litellm_params.get("api_base"),
+ api_key=litellm_params.get("api_key"),
+ model="",
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ )
+ return f"{api_base}/{file_id}/content", {}
+
+ def transform_file_content_response(
+ self,
+ raw_response: httpx.Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> HttpxBinaryResponseContent:
+ """Transform file content response."""
+ return HttpxBinaryResponseContent(response=raw_response)
+
diff --git a/litellm/llms/manus/responses/transformation.py b/litellm/llms/manus/responses/transformation.py
index 8e0a4a5a30f..fbbed19f8d4 100644
--- a/litellm/llms/manus/responses/transformation.py
+++ b/litellm/llms/manus/responses/transformation.py
@@ -1,3 +1,4 @@
+import uuid
from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple, Union
import httpx
@@ -227,6 +228,12 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig):
total_tokens=0,
)
+ # Ensure id is present - failed responses may not include it
+ if "id" not in raw_response_json or raw_response_json.get("id") is None:
+ # Generate a placeholder id for failed responses
+ # This allows the response object to be created even when the API doesn't return an id
+ raw_response_json["id"] = f"unknown-{uuid.uuid4().hex[:8]}"
+
try:
response = ResponsesAPIResponse(**raw_response_json)
except Exception:
@@ -296,6 +303,28 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig):
raw_response_headers = dict(raw_response.headers)
processed_headers = process_response_headers(raw_response_headers)
+ # Ensure reasoning, text, output, and usage are present with defaults
+ if "reasoning" not in raw_response_json or raw_response_json.get("reasoning") is None:
+ raw_response_json["reasoning"] = {}
+
+ if "text" not in raw_response_json or raw_response_json.get("text") is None:
+ raw_response_json["text"] = {}
+
+ if "output" not in raw_response_json or raw_response_json.get("output") is None:
+ raw_response_json["output"] = []
+
+ if "usage" not in raw_response_json or raw_response_json.get("usage") is None:
+ raw_response_json["usage"] = ResponseAPIUsage(
+ input_tokens=0,
+ output_tokens=0,
+ total_tokens=0,
+ )
+
+ # Ensure id is present - failed responses may not include it
+ if "id" not in raw_response_json or raw_response_json.get("id") is None:
+ # Generate a placeholder id for failed responses
+ raw_response_json["id"] = f"unknown-{uuid.uuid4().hex[:8]}"
+
try:
response = ResponsesAPIResponse(**raw_response_json)
except Exception:
diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py
index 01f6c86fd4d..b3612113ec2 100644
--- a/litellm/llms/vertex_ai/files/transformation.py
+++ b/litellm/llms/vertex_ai/files/transformation.py
@@ -1,11 +1,12 @@
import json
import os
import time
-from litellm._uuid import uuid
from typing import Any, Dict, List, Optional, Tuple, Union
from httpx import Headers, Response
+from openai.types.file_deleted import FileDeleted
+from litellm._uuid import uuid
from litellm.files.utils import FilesAPIUtils
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@@ -24,6 +25,7 @@ from litellm.types.llms.openai import (
AllMessageValues,
CreateFileRequest,
FileTypes,
+ HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
PathLike,
@@ -333,6 +335,70 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
status_code=status_code, message=error_message, headers=headers
)
+ def transform_retrieve_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("VertexAIFilesConfig does not support file retrieval")
+
+ def transform_retrieve_file_response(
+ self,
+ raw_response: Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> OpenAIFileObject:
+ raise NotImplementedError("VertexAIFilesConfig does not support file retrieval")
+
+ def transform_delete_file_request(
+ self,
+ file_id: str,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("VertexAIFilesConfig does not support file deletion")
+
+ def transform_delete_file_response(
+ self,
+ raw_response: Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> FileDeleted:
+ raise NotImplementedError("VertexAIFilesConfig does not support file deletion")
+
+ def transform_list_files_request(
+ self,
+ purpose: Optional[str],
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("VertexAIFilesConfig does not support file listing")
+
+ def transform_list_files_response(
+ self,
+ raw_response: Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> List[OpenAIFileObject]:
+ raise NotImplementedError("VertexAIFilesConfig does not support file listing")
+
+ def transform_file_content_request(
+ self,
+ file_content_request,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> tuple[str, dict]:
+ raise NotImplementedError("VertexAIFilesConfig does not support file content retrieval")
+
+ def transform_file_content_response(
+ self,
+ raw_response: Response,
+ logging_obj: LiteLLMLoggingObj,
+ litellm_params: dict,
+ ) -> HttpxBinaryResponseContent:
+ raise NotImplementedError("VertexAIFilesConfig does not support file content retrieval")
+
class VertexAIJsonlFilesTransformation(VertexGeminiConfig):
"""
diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py
index a3606ff9deb..826f151df35 100644
--- a/litellm/llms/vertex_ai/vertex_llm_base.py
+++ b/litellm/llms/vertex_ai/vertex_llm_base.py
@@ -388,10 +388,6 @@ class VertexBase:
Internal function. Returns the token and url for the call.
Handles logic if it's google ai studio vs. vertex ai.
-
- For Vertex AI:
- - If gemini_api_key is provided, use API key authentication (x-goog-api-key header)
- - Otherwise, use service account credentials (OAuth2 Bearer token)
Returns
token, url
@@ -404,7 +400,7 @@ class VertexBase:
stream=stream,
gemini_api_key=gemini_api_key,
)
- auth_header = None # this field is not used for gemini
+ auth_header = None # this field is not used for gemin
else:
vertex_location = self.get_vertex_region(
vertex_region=vertex_location,
@@ -413,32 +409,14 @@ class VertexBase:
### SET RUNTIME ENDPOINT ###
version = "v1beta1" if should_use_v1beta1_features is True else "v1"
-
- # Check if using API key authentication for Vertex AI
- if gemini_api_key and not vertex_credentials:
- # When using API key with Vertex AI, use the Google AI Studio endpoint
- # This is because Vertex AI API keys work with generativelanguage.googleapis.com
- verbose_logger.debug(
- f"Using Vertex AI API key authentication for model: {model} - routing to Google AI Studio endpoint"
- )
- url, endpoint = _get_gemini_url(
- mode=mode,
- model=model,
- stream=stream,
- gemini_api_key=gemini_api_key,
- )
- # API key is already included in the URL by _get_gemini_url
- auth_header = None
- else:
- # Use OAuth2 Bearer token authentication (traditional Vertex AI)
- url, endpoint = _get_vertex_url(
- mode=mode,
- model=model,
- stream=stream,
- vertex_project=vertex_project,
- vertex_location=vertex_location,
- vertex_api_version=version,
- )
+ url, endpoint = _get_vertex_url(
+ mode=mode,
+ model=model,
+ stream=stream,
+ vertex_project=vertex_project,
+ vertex_location=vertex_location,
+ vertex_api_version=version,
+ )
return self._check_custom_proxy(
api_base=api_base,
diff --git a/litellm/main.py b/litellm/main.py
index 10e3bcac04b..6905d8f6a87 100644
--- a/litellm/main.py
+++ b/litellm/main.py
@@ -189,7 +189,7 @@ from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from .llms.custom_llm import CustomLLM, custom_chat_llm_router
from .llms.databricks.embed.handler import DatabricksEmbeddingHandler
from .llms.deprecated_providers import aleph_alpha, palm
-from .llms.gemini.common_utils import get_api_key_from_env, get_vertex_api_key_from_env
+from .llms.gemini.common_utils import get_api_key_from_env
from .llms.groq.chat.handler import GroqChatCompletion
from .llms.heroku.chat.transformation import HerokuChatConfig
from .llms.huggingface.embedding.handler import HuggingFaceEmbedding
@@ -3230,12 +3230,6 @@ def completion( # type: ignore # noqa: PLR0915
or get_secret("VERTEXAI_CREDENTIALS")
)
- vertex_api_key = (
- api_key
- or get_vertex_api_key_from_env()
- or litellm.api_key
- )
-
api_base = api_base or litellm.api_base or get_secret("VERTEXAI_API_BASE")
new_params = safe_deep_copy(optional_params or {})
@@ -3277,7 +3271,7 @@ def completion( # type: ignore # noqa: PLR0915
vertex_location=vertex_ai_location,
vertex_project=vertex_ai_project,
vertex_credentials=vertex_credentials,
- gemini_api_key=vertex_api_key, # Support for Vertex AI API Key
+ gemini_api_key=None,
logging_obj=logging,
acompletion=acompletion,
timeout=timeout,
diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py
index 26778ece60e..843da0e3a0d 100644
--- a/litellm/proxy/auth/auth_checks.py
+++ b/litellm/proxy/auth/auth_checks.py
@@ -1794,11 +1794,20 @@ async def get_org_object(
user_api_key_cache: DualCache,
parent_otel_span: Optional[Span] = None,
proxy_logging_obj: Optional[ProxyLogging] = None,
+ include_budget_table: bool = False,
) -> Optional[LiteLLM_OrganizationTable]:
"""
- Check if org id in proxy Org Table
- if valid, return LiteLLM_OrganizationTable object
- if not, then raise an error
+
+ Args:
+ org_id: Organization ID to look up
+ prisma_client: Database client
+ user_api_key_cache: Cache for storing results
+ parent_otel_span: Optional OpenTelemetry span
+ proxy_logging_obj: Optional proxy logging object
+ include_budget_table: If True, includes litellm_budget_table in the query
"""
if prisma_client is None:
raise Exception(
@@ -1807,8 +1816,13 @@ async def get_org_object(
if not isinstance(org_id, str):
return None
+ # Use different cache key if budget table is included
+ cache_key = "org_id:{}".format(org_id)
+ if include_budget_table:
+ cache_key = "org_id:{}:with_budget".format(org_id)
+
# check if in cache
- cached_org_obj = user_api_key_cache.async_get_cache(key="org_id:{}".format(org_id))
+ cached_org_obj = user_api_key_cache.async_get_cache(key=cache_key)
if cached_org_obj is not None:
if isinstance(cached_org_obj, dict):
return LiteLLM_OrganizationTable(**cached_org_obj)
@@ -1816,13 +1830,24 @@ async def get_org_object(
return cached_org_obj
# else, check db
try:
+ query_kwargs = {"where": {"organization_id": org_id}}
+ if include_budget_table:
+ query_kwargs["include"] = {"litellm_budget_table": True}
+
response = await prisma_client.db.litellm_organizationtable.find_unique(
- where={"organization_id": org_id}
+ **query_kwargs
)
if response is None:
raise Exception
+ # Cache the result
+ await user_api_key_cache.async_set_cache(
+ key=cache_key,
+ value=response.model_dump() if hasattr(response, "model_dump") else response,
+ ttl=DEFAULT_IN_MEMORY_TTL,
+ )
+
return response
except Exception:
raise Exception(
@@ -2344,11 +2369,14 @@ async def _organization_max_budget_check(
if org_id is None:
return
- # Get organization object with budget table to check current spend and max budget
+ # Get organization object with budget table - use get_org_object so it can be mocked in tests
try:
- org_table = await prisma_client.db.litellm_organizationtable.find_unique(
- where={"organization_id": org_id},
- include={"litellm_budget_table": True},
+ org_table = await get_org_object(
+ org_id=org_id,
+ prisma_client=prisma_client,
+ user_api_key_cache=user_api_key_cache,
+ proxy_logging_obj=proxy_logging_obj,
+ include_budget_table=True,
)
except Exception:
# If organization lookup fails, skip the check
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index 171898b1631..9ea2ea7d5c9 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -4479,21 +4479,35 @@ def validate_model_access(
) -> None:
"""
Validate that a model is accessible to the user.
+ Supports batch requests with comma-separated model IDs.
Args:
- model_id: The model ID to validate
+ model_id: The model ID to validate (can be comma-separated for batch requests)
available_models: List of models available to the user
Raises:
HTTPException: If the model is not accessible
"""
- if model_id not in available_models:
- raise HTTPException(
- status_code=404,
- detail="The model `{}` does not exist or is not accessible".format(
- model_id
- ),
- )
+ # Handle batch requests with comma-separated models
+ if "," in model_id:
+ models = [m.strip() for m in model_id.split(",")]
+ inaccessible_models = [m for m in models if m not in available_models]
+ if inaccessible_models:
+ raise HTTPException(
+ status_code=404,
+ detail="The following model(s) do not exist or are not accessible: {}".format(
+ ", ".join(inaccessible_models)
+ ),
+ )
+ else:
+ # Single model validation
+ if model_id not in available_models:
+ raise HTTPException(
+ status_code=404,
+ detail="The model `{}` does not exist or is not accessible".format(
+ model_id
+ ),
+ )
def _path_matches_pattern(path: str, pattern: str) -> bool:
diff --git a/litellm/responses/utils.py b/litellm/responses/utils.py
index 7667d1bad84..49b8e123fe2 100644
--- a/litellm/responses/utils.py
+++ b/litellm/responses/utils.py
@@ -198,11 +198,14 @@ class ResponsesAPIRequestUtils:
model_id = model_info.get("id")
# access the response id based on the object type
- response_id = (
- responses_api_response["id"]
- if isinstance(responses_api_response, dict)
- else responses_api_response.id
- )
+ if isinstance(responses_api_response, dict):
+ response_id = responses_api_response.get("id")
+ else:
+ response_id = getattr(responses_api_response, "id", None)
+
+ # If no response_id, return the response as-is (likely an error response)
+ if response_id is None:
+ return responses_api_response
updated_id = ResponsesAPIRequestUtils._build_responses_api_response_id(
model_id=model_id,
diff --git a/litellm/router.py b/litellm/router.py
index 364e6719300..638df49ac05 100644
--- a/litellm/router.py
+++ b/litellm/router.py
@@ -4486,21 +4486,9 @@ class Router:
if hasattr(original_exception, "message"):
# add the available fallbacks to the exception
- deployment_info = ""
- if kwargs is not None:
- metadata = kwargs.get('metadata', {})
- if metadata and 'deployment' in metadata:
- deployment_info = f"\nUsed Deployment: {metadata['deployment']}"
- if 'model_info' in metadata:
- model_info = metadata['model_info']
- if isinstance(model_info, dict):
- deployment_info += f"\nDeployment ID: {model_info.get('id', 'unknown')}"
-
- original_exception.message += ( # type: ignore
- f". Received Model Group={model_group}"
- f"\nAvailable Model Group Fallbacks={fallback_model_group}"
- f"{deployment_info}"
- f"\n\nš” Tip: If using wildcard patterns (e.g., 'openai/*'), ensure all matching deployments have credentials with access to this model."
+ original_exception.message += ". Received Model Group={}\nAvailable Model Group Fallbacks={}".format( # type: ignore
+ model_group,
+ fallback_model_group,
)
if len(fallback_failure_exception_str) > 0:
original_exception.message += ( # type: ignore
@@ -7679,10 +7667,6 @@ class Router:
)
if pattern_deployments:
- verbose_router_logger.debug(
- f"Pattern match for model='{model}': Found {len(pattern_deployments)} deployments. "
- f"Deployment IDs: {[d.get('model_info', {}).get('id', 'unknown') for d in pattern_deployments]}"
- )
return model, pattern_deployments
if (
diff --git a/litellm/types/files.py b/litellm/types/files.py
index 600ad806e22..8b87b33cd1b 100644
--- a/litellm/types/files.py
+++ b/litellm/types/files.py
@@ -1,6 +1,8 @@
from enum import Enum
from types import MappingProxyType
-from typing import List, Set, Mapping
+from typing import Any, Dict, List, Literal, Mapping, Set, Union
+
+from typing_extensions import Required, TypedDict
"""
Base Enums/Consts
@@ -281,3 +283,41 @@ GEMINI_1_5_ACCEPTED_FILE_TYPES: Set[FileType] = {
def is_gemini_1_5_accepted_file_type(file_type: FileType) -> bool:
return file_type in GEMINI_1_5_ACCEPTED_FILE_TYPES
+
+
+"""
+Two-Step File Upload Types
+"""
+
+
+class TwoStepFileUploadRequest(TypedDict):
+ """
+ Request structure for two-step file upload process.
+
+ Step 1: Initial request to get upload URL
+ Step 2: Upload file content to the upload URL
+
+ Used by providers like Manus and Google Cloud Storage.
+ """
+
+ method: Required[str]
+ url: Required[str]
+ headers: Required[Dict[str, str]]
+ data: Required[Union[str, bytes, Dict[str, Any]]]
+
+
+class TwoStepFileUploadConfig(TypedDict, total=False):
+ """
+ Configuration for two-step file upload process.
+
+ Properties:
+ initial_request: Request to create file record and get upload URL
+ upload_request: Request to upload actual file content
+ upload_url_location: Where to find upload URL ('headers' or 'body')
+ upload_url_key: Key name for upload URL in response (default: 'upload_url')
+ """
+
+ initial_request: Required[TwoStepFileUploadRequest]
+ upload_request: Required[TwoStepFileUploadRequest]
+ upload_url_location: Required[Literal["headers", "body"]]
+ upload_url_key: str
diff --git a/litellm/utils.py b/litellm/utils.py
index 42d4a1ac372..1393f1f5cb5 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -48,6 +48,7 @@ from tokenizers import Tokenizer
import litellm
import litellm.litellm_core_utils
+
# audio_utils.utils is lazy-loaded - only imported when needed for transcription calls
import litellm.litellm_core_utils.json_validation_rule
from litellm._lazy_imports import (
@@ -71,8 +72,6 @@ from litellm.constants import (
TOOL_CHOICE_OBJECT_TOKEN_COUNT,
)
-
-
_CachingHandlerResponse = None
_LLMCachingHandler = None
_CustomGuardrail = None
@@ -7959,6 +7958,10 @@ class ProviderConfigManager:
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
return BedrockFilesConfig()
+ elif LlmProviders.MANUS == provider:
+ from litellm.llms.manus.files.transformation import ManusFilesConfig
+
+ return ManusFilesConfig()
return None
@staticmethod
diff --git a/poetry.lock b/poetry.lock
index 0a4ef10d09f..3bafdb157ca 100644
--- a/poetry.lock
+++ b/poetry.lock
@@ -3081,15 +3081,15 @@ files = [
[[package]]
name = "litellm-proxy-extras"
-version = "0.4.18"
+version = "0.4.21"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
optional = true
python-versions = "!=2.7.*,!=3.0.*,!=3.1.*,!=3.2.*,!=3.3.*,!=3.4.*,!=3.5.*,!=3.6.*,!=3.7.*,>=3.8"
groups = ["main"]
markers = "extra == \"proxy\""
files = [
- {file = "litellm_proxy_extras-0.4.18-py3-none-any.whl", hash = "sha256:c3edee68bf8eb073c6158dcf7df05727dfc829e63c03a617fcb48853d11490df"},
- {file = "litellm_proxy_extras-0.4.18.tar.gz", hash = "sha256:898b28e3e74acdc29142906b84787ab05a90e30aa3c0c8aee849915e3a16adb3"},
+ {file = "litellm_proxy_extras-0.4.21-py3-none-any.whl", hash = "sha256:83a1734e9773610945230606012e602bbcbfba1c60fde836d51102c1a296f166"},
+ {file = "litellm_proxy_extras-0.4.21.tar.gz", hash = "sha256:fa0e012984aa8e5114f88f4bad53d6abb589e5ca3eab445f74f8ddeceb62d848"},
]
[[package]]
@@ -7981,4 +7981,4 @@ utils = ["numpydoc"]
[metadata]
lock-version = "2.1"
python-versions = ">=3.9,<4.0"
-content-hash = "e9fd12b5ccc703ec156d98877452417083e3ac18b5970cb3a58c3bde09d267bb"
+content-hash = "ea62b77c662ab9fc486e421c576f0868bcde16d62a24703ee1f4916a0465ffb2"
diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json
index 8432ce4e874..db11ec8a0de 100644
--- a/provider_endpoints_support.json
+++ b/provider_endpoints_support.json
@@ -2335,6 +2335,7 @@
"audio_speech": false,
"moderations": false,
"batches": false,
+ "files": true,
"rerank": false,
"a2a": true,
"interactions": true
diff --git a/pyproject.toml b/pyproject.toml
index 00e7bff78b6..742f743b54d 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -167,7 +167,7 @@ requires = ["poetry-core", "wheel"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
-version = "1.80.13"
+version = "1.80.14"
version_files = [
"pyproject.toml:^version"
]
diff --git a/tests/batches_tests/test_manus_files_all_methods.py b/tests/batches_tests/test_manus_files_all_methods.py
new file mode 100644
index 00000000000..49322bdcb17
--- /dev/null
+++ b/tests/batches_tests/test_manus_files_all_methods.py
@@ -0,0 +1,72 @@
+"""
+E2E test for all Manus Files API methods.
+"""
+
+import os
+import pytest
+import litellm
+
+
+@pytest.mark.asyncio
+async def test_manus_files_api_e2e_all_methods():
+ """
+ E2E test for Manus Files API: create, retrieve, list, delete.
+ """
+ litellm._turn_on_debug()
+
+ api_key = os.getenv("MANUS_API_KEY")
+ if api_key is None:
+ pytest.skip("MANUS_API_KEY not set")
+
+ # Create a simple test file content
+ test_content = b"This is a test file for Manus Files API - all methods test."
+ test_filename = "test_file_all_methods.txt"
+
+ # Step 1: Create file
+ print("Step 1: Creating file...")
+ created_file = await litellm.acreate_file(
+ file=(test_filename, test_content),
+ purpose="assistants",
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Created file: {created_file}")
+ assert created_file.filename == test_filename
+ assert created_file.status == "uploaded"
+ # Note: Manus doesn't return bytes in initial response
+ file_id = created_file.id
+
+ # Step 2: Retrieve file
+ print(f"\nStep 2: Retrieving file {file_id}...")
+ retrieved_file = await litellm.afile_retrieve(
+ file_id=file_id,
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Retrieved file: {retrieved_file}")
+ assert retrieved_file.id == file_id
+ assert retrieved_file.filename == test_filename
+
+ # Step 3: List files
+ print("\nStep 3: Listing files...")
+ files_list = await litellm.afile_list(
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Files list: {files_list}")
+ assert isinstance(files_list, list)
+ assert any(f.id == file_id for f in files_list)
+
+ # Step 4: Delete file
+ print(f"\nStep 4: Deleting file {file_id}...")
+ deleted_file = await litellm.afile_delete(
+ file_id=file_id,
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Deleted file: {deleted_file}")
+ assert deleted_file.id == file_id
+ assert deleted_file.deleted is True
+
+ print("\nā
All Manus Files API methods working!")
+
diff --git a/tests/code_coverage_tests/recursive_detector.py b/tests/code_coverage_tests/recursive_detector.py
index 20e6a381e5b..2a460f621b2 100644
--- a/tests/code_coverage_tests/recursive_detector.py
+++ b/tests/code_coverage_tests/recursive_detector.py
@@ -35,6 +35,7 @@ IGNORE_FUNCTIONS = [
"_fix_enum_types", # max depth set.
"_collect_argument_paths", # max depth set.
"_split_text", # max depth set.
+ "_mask_sequence", # max depth set.
"_delete_nested_value_custom", # max depth set (bounded by number of path segments).
"filter_exceptions_from_params", # max depth set (default 20) to prevent infinite recursion.
"__getattr__", # lazy loading pattern in litellm/__init__.py with proper caching to prevent infinite recursion.
diff --git a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py
index 2f92afb3824..e8fe4dd3393 100644
--- a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py
+++ b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py
@@ -1124,124 +1124,6 @@ def test_get_custom_labels_from_metadata_tags(monkeypatch):
assert get_custom_labels_from_metadata(metadata) == {}
-def test_get_custom_labels_from_top_level_metadata(monkeypatch):
- """
- Test that get_custom_labels_from_metadata can extract fields from top-level metadata,
- such as requester_ip_address, not just from nested dictionaries like requester_metadata.
- """
- monkeypatch.setattr(
- "litellm.custom_prometheus_metadata_labels",
- ["requester_ip_address", "user_api_key_alias"],
- )
- # Simulate metadata structure with top-level fields
- metadata = {
- "requester_ip_address": "10.48.203.20", # Top-level field
- "user_api_key_alias": "TestAlias", # Top-level field
- "requester_metadata": {"nested_field": "nested_value"}, # Nested dict (excluded)
- "user_api_key_auth_metadata": {"another_nested": "value"}, # Nested dict (excluded)
- }
- result = get_custom_labels_from_metadata(metadata)
- assert result == {
- "requester_ip_address": "10.48.203.20",
- "user_api_key_alias": "TestAlias",
- }
-
-
-def test_get_custom_labels_from_top_level_and_nested_metadata(monkeypatch):
- """
- Test that get_custom_labels_from_metadata can extract fields from both top-level
- and nested metadata (requester_metadata, user_api_key_auth_metadata).
- """
- monkeypatch.setattr(
- "litellm.custom_prometheus_metadata_labels",
- [
- "requester_ip_address", # Top-level
- "metadata.foo", # From requester_metadata
- "metadata.bar", # From user_api_key_auth_metadata
- ],
- )
- # Simulate combined_metadata structure as it would appear after merging
- # This is what gets passed to get_custom_labels_from_metadata
- combined_metadata = {
- "requester_ip_address": "10.48.203.20", # Top-level field
- "foo": "bar_value", # From requester_metadata (spread)
- "bar": "baz_value", # From user_api_key_auth_metadata (spread)
- }
- result = get_custom_labels_from_metadata(combined_metadata)
- assert result == {
- "requester_ip_address": "10.48.203.20",
- "metadata_foo": "bar_value",
- "metadata_bar": "baz_value",
- }
-
-
-async def test_async_log_success_event_with_top_level_metadata(prometheus_logger, monkeypatch):
- """
- Test that async_log_success_event correctly extracts custom labels from top-level metadata
- fields like requester_ip_address, not just from nested dictionaries.
- """
- # Configure custom metadata labels to extract requester_ip_address
- monkeypatch.setattr(
- "litellm.custom_prometheus_metadata_labels", ["requester_ip_address"]
- )
-
- # Create standard logging payload with requester_ip_address at top-level metadata
- standard_logging_object = create_standard_logging_payload()
- standard_logging_object["metadata"]["requester_ip_address"] = "10.48.203.20"
- standard_logging_object["metadata"]["requester_metadata"] = {} # Empty nested dict
- standard_logging_object["metadata"]["user_api_key_auth_metadata"] = {} # Empty nested dict
-
- kwargs = {
- "model": "gpt-3.5-turbo",
- "stream": True,
- "litellm_params": {
- "metadata": {
- "user_api_key": "test_key",
- "user_api_key_user_id": "test_user",
- "user_api_key_team_id": "test_team",
- "user_api_key_end_user_id": "test_end_user",
- }
- },
- "start_time": datetime.now(),
- "completion_start_time": datetime.now(),
- "api_call_start_time": datetime.now(),
- "end_time": datetime.now() + timedelta(seconds=1),
- "standard_logging_object": standard_logging_object,
- }
- response_obj = MagicMock()
-
- # Mock the prometheus client methods
- prometheus_logger.litellm_requests_metric = MagicMock()
- prometheus_logger.litellm_spend_metric = MagicMock()
- prometheus_logger.litellm_tokens_metric = MagicMock()
- prometheus_logger.litellm_input_tokens_metric = MagicMock()
- prometheus_logger.litellm_output_tokens_metric = MagicMock()
- prometheus_logger.litellm_remaining_team_budget_metric = MagicMock()
- prometheus_logger.litellm_remaining_api_key_budget_metric = MagicMock()
- prometheus_logger.litellm_remaining_api_key_requests_for_model = MagicMock()
- prometheus_logger.litellm_remaining_api_key_tokens_for_model = MagicMock()
- prometheus_logger.litellm_llm_api_time_to_first_token_metric = MagicMock()
- prometheus_logger.litellm_llm_api_latency_metric = MagicMock()
- prometheus_logger.litellm_request_total_latency_metric = MagicMock()
-
- await prometheus_logger.async_log_success_event(
- kwargs, response_obj, kwargs["start_time"], kwargs["end_time"]
- )
-
- # Verify that the metrics were called with labels including requester_ip_address
- # Check that labels() was called - the actual labels dict should include requester_ip_address
- assert prometheus_logger.litellm_requests_metric.labels.called
- assert prometheus_logger.litellm_spend_metric.labels.called
-
- # Get the actual call arguments to verify requester_ip_address is included
- # The custom labels should be extracted and included in the label factory
- call_args = prometheus_logger.litellm_requests_metric.labels.call_args
- assert call_args is not None
- # The labels() method receives a dict with label names and values
- # We can't easily assert the exact values without checking the internal implementation,
- # but we've verified the function is called, which means the extraction happened
-
-
def test_get_custom_labels_from_tags(monkeypatch):
from litellm.integrations.prometheus import get_custom_labels_from_tags
diff --git a/tests/llm_responses_api_testing/test_manus_files_all_methods.py b/tests/llm_responses_api_testing/test_manus_files_all_methods.py
new file mode 100644
index 00000000000..39311441f59
--- /dev/null
+++ b/tests/llm_responses_api_testing/test_manus_files_all_methods.py
@@ -0,0 +1,71 @@
+"""
+E2E test for all Manus Files API methods.
+"""
+
+import os
+import pytest
+import litellm
+
+
+@pytest.mark.asyncio
+async def test_manus_files_api_e2e_all_methods():
+ """
+ E2E test for Manus Files API: create, retrieve, list, delete.
+ """
+ litellm._turn_on_debug()
+
+ api_key = os.getenv("MANUS_API_KEY")
+ if api_key is None:
+ pytest.skip("MANUS_API_KEY not set")
+
+ # Create a simple test file content
+ test_content = b"This is a test file for Manus Files API - all methods test."
+ test_filename = "test_file_all_methods.txt"
+
+ # Step 1: Create file
+ print("Step 1: Creating file...")
+ created_file = await litellm.acreate_file(
+ file=(test_filename, test_content),
+ purpose="assistants",
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Created file: {created_file}")
+ assert created_file.filename == test_filename
+ assert created_file.status == "uploaded"
+ # Note: Manus doesn't return bytes in initial response
+ file_id = created_file.id
+
+ # Step 2: Retrieve file
+ print(f"\nStep 2: Retrieving file {file_id}...")
+ retrieved_file = await litellm.afile_retrieve(
+ file_id=file_id,
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Retrieved file: {retrieved_file}")
+ assert retrieved_file.id == file_id
+ assert retrieved_file.filename == test_filename
+
+ # Step 3: List files
+ print("\nStep 3: Listing files...")
+ files_list = await litellm.afile_list(
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Files list: {files_list}")
+ assert isinstance(files_list, list)
+ assert any(f.id == file_id for f in files_list)
+
+ # Step 4: Delete file
+ print(f"\nStep 4: Deleting file {file_id}...")
+ deleted_file = await litellm.afile_delete(
+ file_id=file_id,
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Deleted file: {deleted_file}")
+ assert deleted_file.id == file_id
+ assert deleted_file.deleted is True
+
+ print("\nā
All Manus Files API methods working!")
diff --git a/tests/llm_responses_api_testing/test_manus_responses_api.py b/tests/llm_responses_api_testing/test_manus_responses_api.py
index 4f64980d261..6256363f407 100644
--- a/tests/llm_responses_api_testing/test_manus_responses_api.py
+++ b/tests/llm_responses_api_testing/test_manus_responses_api.py
@@ -38,7 +38,11 @@ async def test_manus_responses_api_with_agent_profile():
print("Manus response=", json.dumps(response, indent=4, default=str))
## Get the status of the response
- got_response = await litellm.aget_responses(response_id=response.id)
+ got_response = await litellm.aget_responses(
+ response_id=response.id,
+ custom_llm_provider="manus",
+ api_key=os.getenv("MANUS_API_KEY"),
+ )
print("GET API MANUS RESPONSE=", json.dumps(got_response, indent=4, default=str))
if got_response.status == "completed":
assert got_response.output is not None
@@ -47,6 +51,95 @@ async def test_manus_responses_api_with_agent_profile():
# Manus can return "running" or "pending" status
assert got_response.status in ["running", "pending"]
assert got_response.id is not None
+
+
+@pytest.mark.asyncio
+async def test_manus_responses_api_with_file_upload():
+ """
+ Test that uploads a file via Files API and then passes it to Responses API.
+ """
+ litellm._turn_on_debug()
+
+ api_key = os.getenv("MANUS_API_KEY")
+ if api_key is None:
+ pytest.skip("MANUS_API_KEY not set")
+
+ # Step 1: Upload a file
+ test_content = b"Warren Buffett's 2023 Letter to Shareholders\n\nKey Points:\n1. Long-term value creation\n2. Capital allocation strategy\n3. Market volatility perspective"
+ test_filename = "buffett_letter_summary.txt"
+
+ print("Step 1: Uploading file...")
+ uploaded_file = await litellm.acreate_file(
+ file=(test_filename, test_content),
+ purpose="assistants",
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Uploaded file: {uploaded_file}")
+ assert uploaded_file.id is not None
+ file_id = uploaded_file.id
+
+ # Step 2: Create a response with the uploaded file
+ print(f"\nStep 2: Creating response with file {file_id}...")
+ response = await litellm.aresponses(
+ model="manus/manus-1.6-lite",
+ input=[
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "input_text",
+ "text": "Summarize the key points from this letter.",
+ },
+ {
+ "type": "input_file",
+ "file_id": file_id,
+ },
+ ],
+ },
+ ],
+ api_key=api_key,
+ max_output_tokens=100,
+ )
+
+ print(f"Response created: {response}")
+ print(f"Response type: {type(response)}")
+ print(f"Response has id: {hasattr(response, 'id')}")
+
+ # Handle both dict and ResponsesAPIResponse object
+ if isinstance(response, dict):
+ response_id = response.get("id")
+ else:
+ response_id = getattr(response, "id", None)
+
+ assert response_id is not None, f"Response ID is None. Response: {response}"
+
+ # Step 3: Get the response status
+ print(f"\nStep 3: Getting response status...")
+ got_response = await litellm.aget_responses(
+ response_id=response_id,
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Response status: {got_response}")
+ got_response_id = getattr(got_response, "id", None)
+ got_response_status = getattr(got_response, "status", None)
+
+ assert got_response_id == response_id
+ assert got_response_status in ["completed", "running", "pending"]
+
+ # Step 4: Clean up - delete the file
+ print(f"\nStep 4: Cleaning up - deleting file {file_id}...")
+ deleted_file = await litellm.afile_delete(
+ file_id=file_id,
+ custom_llm_provider="manus",
+ api_key=api_key,
+ )
+ print(f"Deleted file: {deleted_file}")
+ assert deleted_file.deleted is True
+
+ print("\nā
File upload and responses API integration test passed!")
+
diff --git a/tests/proxy_unit_tests/test_proxy_server.py b/tests/proxy_unit_tests/test_proxy_server.py
index 8d3b0fee48f..0db1240544a 100644
--- a/tests/proxy_unit_tests/test_proxy_server.py
+++ b/tests/proxy_unit_tests/test_proxy_server.py
@@ -642,8 +642,8 @@ def test_embedding(mock_aembedding, client_no_auth):
during_call_kwargs = mock_during_hook.await_args_list[0].kwargs
assert (
- during_call_kwargs.get("call_type") == "embeddings"
- ), f"expected during_call_hook to receive call_type='embeddings', got {during_call_kwargs.get('call_type')}"
+ during_call_kwargs.get("call_type") == "embedding"
+ ), f"expected during_call_hook to receive call_type='embedding', got {during_call_kwargs.get('call_type')}"
except Exception as e:
pytest.fail(f"LiteLLM Proxy test failed. Exception - {str(e)}")
@@ -2185,6 +2185,10 @@ async def test_proxy_server_prisma_setup():
mock_client._set_spend_logs_row_count_in_proxy_state = (
AsyncMock()
) # Mock the _set_spend_logs_row_count_in_proxy_state method
+ # Mock the db attribute with start_token_refresh_task for RDS IAM token refresh
+ mock_db = MagicMock()
+ mock_db.start_token_refresh_task = AsyncMock()
+ mock_client.db = mock_db
await ProxyStartupEvent._setup_prisma_client(
database_url=os.getenv("DATABASE_URL"),
diff --git a/tests/proxy_unit_tests/test_proxy_utils.py b/tests/proxy_unit_tests/test_proxy_utils.py
index 5c3c3948920..64f1ec24234 100644
--- a/tests/proxy_unit_tests/test_proxy_utils.py
+++ b/tests/proxy_unit_tests/test_proxy_utils.py
@@ -1574,6 +1574,10 @@ async def test_health_check_not_called_when_disabled(monkeypatch):
mock_prisma.health_check = AsyncMock()
mock_prisma.check_view_exists = AsyncMock()
mock_prisma._set_spend_logs_row_count_in_proxy_state = AsyncMock()
+ # Mock the db attribute with start_token_refresh_task for RDS IAM token refresh
+ mock_db = MagicMock()
+ mock_db.start_token_refresh_task = AsyncMock()
+ mock_prisma.db = mock_db
# Mock PrismaClient constructor
monkeypatch.setattr(
"litellm.proxy.proxy_server.PrismaClient", lambda **kwargs: mock_prisma
diff --git a/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py b/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py
index 9f0a1ae8ffe..b9f17aa4c7f 100644
--- a/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py
+++ b/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py
@@ -108,14 +108,16 @@ def test_lists_with_sensitive_keys_are_masked():
"""
masker = SensitiveDataMasker()
data = {
- "api_key": ["sk-123", "sk-456"],
+ "api_key": ["sk-1234567890abcdef", "sk-9876543210fedcba"],
"tags": ["prod", "test"],
}
masked = masker.mask_dict(data)
# sensitive key list entries should be masked
- assert masked["api_key"][0] != "sk-123"
+ assert masked["api_key"][0] != "sk-1234567890abcdef"
assert "*" in masked["api_key"][0]
+ assert masked["api_key"][1] != "sk-9876543210fedcba"
+ assert "*" in masked["api_key"][1]
# non-sensitive list should remain unchanged
assert masked["tags"] == ["prod", "test"]
diff --git a/tests/test_litellm/llms/bedrock/test_bedrock_ssl_verify.py b/tests/test_litellm/llms/bedrock/test_bedrock_ssl_verify.py
deleted file mode 100644
index 9142de295ea..00000000000
--- a/tests/test_litellm/llms/bedrock/test_bedrock_ssl_verify.py
+++ /dev/null
@@ -1,349 +0,0 @@
-"""
-Test SSL verification for AWS Bedrock boto3 clients.
-
-This test ensures that custom CA certificates are properly passed to all boto3 clients
-(STS and Bedrock services) to support internal certificate authorities.
-
-Issue: https://github.com/BerriAI/litellm/issues/XXXX
-User reported that SSL_CERT_FILE environment variable and ssl_verify config were not
-being applied to boto3 clients, causing "certificate verify failed" errors.
-"""
-
-import os
-import sys
-import tempfile
-from unittest.mock import MagicMock, Mock, patch
-
-import pytest
-
-sys.path.insert(0, os.path.abspath("../.."))
-
-import litellm
-from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
-from litellm.llms.bedrock.common_utils import init_bedrock_client
-
-
-class TestBedrockSSLVerify:
- """Test suite for SSL verification in Bedrock boto3 clients."""
-
- def test_base_aws_llm_get_ssl_verify_default(self):
- """Test that _get_ssl_verify returns default value when no custom config is set."""
- base_aws = BaseAWSLLM()
-
- # Clear any environment variables
- os.environ.pop("SSL_VERIFY", None)
- os.environ.pop("SSL_CERT_FILE", None)
-
- # Reset litellm.ssl_verify to default
- litellm.ssl_verify = True
-
- ssl_verify = base_aws._get_ssl_verify()
- assert ssl_verify is True
-
- def test_base_aws_llm_get_ssl_verify_false(self):
- """Test that _get_ssl_verify returns False when SSL verification is disabled."""
- base_aws = BaseAWSLLM()
-
- # Set SSL_VERIFY to False via environment
- os.environ["SSL_VERIFY"] = "False"
-
- ssl_verify = base_aws._get_ssl_verify()
- assert ssl_verify is False
-
- # Clean up
- os.environ.pop("SSL_VERIFY", None)
-
- def test_base_aws_llm_get_ssl_verify_custom_ca_bundle(self):
- """Test that _get_ssl_verify returns custom CA bundle path when SSL_CERT_FILE is set."""
- base_aws = BaseAWSLLM()
-
- # Create a temporary CA bundle file
- with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
- f.write("-----BEGIN CERTIFICATE-----\n")
- f.write("FAKE CERTIFICATE FOR TESTING\n")
- f.write("-----END CERTIFICATE-----\n")
- ca_bundle_path = f.name
-
- try:
- # Set SSL_CERT_FILE environment variable
- os.environ["SSL_CERT_FILE"] = ca_bundle_path
- os.environ.pop("SSL_VERIFY", None)
- litellm.ssl_verify = True
-
- ssl_verify = base_aws._get_ssl_verify()
- assert ssl_verify == ca_bundle_path
- finally:
- # Clean up
- os.environ.pop("SSL_CERT_FILE", None)
- os.unlink(ca_bundle_path)
-
- def test_base_aws_llm_get_ssl_verify_litellm_config(self):
- """Test that _get_ssl_verify uses litellm.ssl_verify when set."""
- base_aws = BaseAWSLLM()
-
- # Clear environment variables
- os.environ.pop("SSL_VERIFY", None)
- os.environ.pop("SSL_CERT_FILE", None)
-
- # Create a temporary CA bundle file
- with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
- f.write("-----BEGIN CERTIFICATE-----\n")
- f.write("FAKE CERTIFICATE FOR TESTING\n")
- f.write("-----END CERTIFICATE-----\n")
- ca_bundle_path = f.name
-
- try:
- # Set litellm.ssl_verify to custom CA bundle
- litellm.ssl_verify = ca_bundle_path
-
- ssl_verify = base_aws._get_ssl_verify()
- # When ssl_verify is a path, it should be returned directly
- assert ssl_verify == ca_bundle_path
- finally:
- # Clean up
- litellm.ssl_verify = True
- os.unlink(ca_bundle_path)
-
- @patch("boto3.client")
- def test_init_bedrock_client_passes_ssl_verify_to_sts(self, mock_boto3_client):
- """Test that init_bedrock_client passes ssl_verify to STS client."""
- # Create a temporary CA bundle file
- with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
- f.write("-----BEGIN CERTIFICATE-----\n")
- f.write("FAKE CERTIFICATE FOR TESTING\n")
- f.write("-----END CERTIFICATE-----\n")
- ca_bundle_path = f.name
-
- try:
- # Set SSL_CERT_FILE environment variable
- os.environ["SSL_CERT_FILE"] = ca_bundle_path
- litellm.ssl_verify = True
-
- # Mock the STS client and Bedrock client
- mock_sts_client = MagicMock()
- mock_sts_response = {
- "Credentials": {
- "AccessKeyId": "test_access_key",
- "SecretAccessKey": "test_secret_key",
- "SessionToken": "test_session_token",
- }
- }
- mock_sts_client.assume_role.return_value = mock_sts_response
-
- mock_bedrock_client = MagicMock()
-
- # Configure mock to return different clients based on service name
- def side_effect(service_name=None, **kwargs):
- if service_name == "sts":
- return mock_sts_client
- elif service_name == "bedrock-runtime":
- return mock_bedrock_client
- return MagicMock()
-
- mock_boto3_client.side_effect = side_effect
-
- # Call init_bedrock_client with role assumption
- client = init_bedrock_client(
- aws_region_name="us-west-2",
- aws_access_key_id="test_key",
- aws_secret_access_key="test_secret",
- aws_role_name="arn:aws:iam::123456789012:role/test-role",
- aws_session_name="test-session",
- )
-
- # Verify that boto3.client was called with verify parameter for STS
- sts_calls = [
- call for call in mock_boto3_client.call_args_list
- if (len(call[0]) > 0 and call[0][0] == "sts") or
- ("service_name" not in call[1]) # STS calls don't use service_name kwarg
- ]
-
- assert len(sts_calls) > 0, "STS client should have been created"
-
- # Check that verify parameter was passed to STS client
- sts_call = sts_calls[0]
- assert "verify" in sts_call[1], "verify parameter should be passed to STS client"
- assert sts_call[1]["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {sts_call[1]['verify']}"
-
- # Verify that boto3.client was called with verify parameter for Bedrock
- bedrock_calls = [
- call for call in mock_boto3_client.call_args_list
- if "service_name" in call[1] and call[1]["service_name"] == "bedrock-runtime"
- ]
-
- assert len(bedrock_calls) > 0, "Bedrock client should have been created"
-
- bedrock_call = bedrock_calls[0]
- assert "verify" in bedrock_call[1], "verify parameter should be passed to Bedrock client"
- assert bedrock_call[1]["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {bedrock_call[1]['verify']}"
-
- finally:
- # Clean up
- os.environ.pop("SSL_CERT_FILE", None)
- os.unlink(ca_bundle_path)
-
- @patch("boto3.client")
- def test_base_aws_llm_auth_with_role_passes_ssl_verify(self, mock_boto3_client):
- """Test that _auth_with_aws_role passes ssl_verify to STS client."""
- base_aws = BaseAWSLLM()
-
- # Create a temporary CA bundle file
- with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
- f.write("-----BEGIN CERTIFICATE-----\n")
- f.write("FAKE CERTIFICATE FOR TESTING\n")
- f.write("-----END CERTIFICATE-----\n")
- ca_bundle_path = f.name
-
- try:
- # Set SSL_CERT_FILE environment variable
- os.environ["SSL_CERT_FILE"] = ca_bundle_path
- litellm.ssl_verify = True
-
- # Mock the STS client
- mock_sts_client = MagicMock()
- mock_sts_response = {
- "Credentials": {
- "AccessKeyId": "test_access_key",
- "SecretAccessKey": "test_secret_key",
- "SessionToken": "test_session_token",
- "Expiration": "2025-01-10T00:00:00Z",
- }
- }
-
- # Convert Expiration to datetime
- from datetime import datetime, timezone
- mock_sts_response["Credentials"]["Expiration"] = datetime.now(timezone.utc)
-
- mock_sts_client.assume_role.return_value = mock_sts_response
- mock_boto3_client.return_value = mock_sts_client
-
- # Call _auth_with_aws_role
- credentials, ttl = base_aws._auth_with_aws_role(
- aws_access_key_id="test_key",
- aws_secret_access_key="test_secret",
- aws_session_token=None,
- aws_role_name="arn:aws:iam::123456789012:role/test-role",
- aws_session_name="test-session",
- )
-
- # Verify that boto3.client was called with verify parameter
- assert mock_boto3_client.called, "boto3.client should have been called"
-
- call_kwargs = mock_boto3_client.call_args[1]
- assert "verify" in call_kwargs, "verify parameter should be passed to STS client"
- assert call_kwargs["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {call_kwargs['verify']}"
-
- finally:
- # Clean up
- os.environ.pop("SSL_CERT_FILE", None)
- os.unlink(ca_bundle_path)
-
- @patch("litellm.llms.bedrock.base_aws_llm.get_secret")
- @patch("boto3.client")
- def test_base_aws_llm_auth_with_web_identity_passes_ssl_verify(self, mock_boto3_client, mock_get_secret):
- """Test that _auth_with_web_identity_token passes ssl_verify to STS client."""
- base_aws = BaseAWSLLM()
-
- # Create a temporary CA bundle file
- with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
- f.write("-----BEGIN CERTIFICATE-----\n")
- f.write("FAKE CERTIFICATE FOR TESTING\n")
- f.write("-----END CERTIFICATE-----\n")
- ca_bundle_path = f.name
-
- try:
- # Set SSL_CERT_FILE environment variable
- os.environ["SSL_CERT_FILE"] = ca_bundle_path
- litellm.ssl_verify = True
-
- # Mock get_secret to return the token
- mock_get_secret.return_value = "mocked_oidc_token"
-
- # Mock the STS client
- mock_sts_client = MagicMock()
- mock_sts_response = {
- "Credentials": {
- "AccessKeyId": "test_access_key",
- "SecretAccessKey": "test_secret_key",
- "SessionToken": "test_session_token",
- },
- "PackedPolicySize": 100,
- }
-
- mock_sts_client.assume_role_with_web_identity.return_value = mock_sts_response
-
- # Mock boto3.Session
- mock_session = MagicMock()
- mock_credentials = MagicMock()
- mock_session.get_credentials.return_value = mock_credentials
-
- mock_boto3_client.return_value = mock_sts_client
-
- with patch("boto3.Session", return_value=mock_session):
- # Call _auth_with_web_identity_token
- credentials, ttl = base_aws._auth_with_web_identity_token(
- aws_web_identity_token="test_token",
- aws_role_name="arn:aws:iam::123456789012:role/test-role",
- aws_session_name="test-session",
- aws_region_name="us-west-2",
- aws_sts_endpoint=None,
- )
-
- # Verify that boto3.client was called with verify parameter
- assert mock_boto3_client.called, "boto3.client should have been called"
-
- call_kwargs = mock_boto3_client.call_args[1]
- assert "verify" in call_kwargs, "verify parameter should be passed to STS client"
- assert call_kwargs["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {call_kwargs['verify']}"
-
- finally:
- # Clean up
- os.environ.pop("SSL_CERT_FILE", None)
- os.unlink(ca_bundle_path)
-
- def test_ssl_verify_priority_env_over_litellm_config(self):
- """Test that SSL_VERIFY environment variable takes priority over litellm.ssl_verify."""
- base_aws = BaseAWSLLM()
-
- # Set litellm.ssl_verify to True
- litellm.ssl_verify = True
-
- # Set SSL_VERIFY environment variable to False
- os.environ["SSL_VERIFY"] = "False"
-
- try:
- ssl_verify = base_aws._get_ssl_verify()
- assert ssl_verify is False, "Environment variable should take priority"
- finally:
- # Clean up
- os.environ.pop("SSL_VERIFY", None)
- litellm.ssl_verify = True
-
- def test_ssl_cert_file_priority_over_default(self):
- """Test that SSL_CERT_FILE takes priority when ssl_verify is True."""
- base_aws = BaseAWSLLM()
-
- # Create a temporary CA bundle file
- with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
- f.write("-----BEGIN CERTIFICATE-----\n")
- f.write("FAKE CERTIFICATE FOR TESTING\n")
- f.write("-----END CERTIFICATE-----\n")
- ca_bundle_path = f.name
-
- try:
- # Set SSL_CERT_FILE environment variable
- os.environ["SSL_CERT_FILE"] = ca_bundle_path
- os.environ.pop("SSL_VERIFY", None)
- litellm.ssl_verify = True
-
- ssl_verify = base_aws._get_ssl_verify()
- assert ssl_verify == ca_bundle_path, "SSL_CERT_FILE should be used when ssl_verify is True"
- finally:
- # Clean up
- os.environ.pop("SSL_CERT_FILE", None)
- os.unlink(ca_bundle_path)
-
-
-if __name__ == "__main__":
- # Run tests
- pytest.main([__file__, "-v", "-s"])
diff --git a/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py b/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py
index 80d65991acb..389c8446135 100644
--- a/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py
+++ b/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py
@@ -13,7 +13,6 @@ sys.path.insert(
import litellm
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
-from litellm.llms.vertex_ai.common_utils import _get_gemini_url
def run_sync(coro):
@@ -1049,139 +1048,3 @@ class TestVertexBase:
MockCredentials.from_info.assert_called_once_with(json_obj)
mock_creds.with_scopes.assert_called_once_with(scopes)
assert result == "scoped_creds"
-
- def test_get_token_and_url_with_api_key(self):
- """Test that API key authentication routes to Google AI Studio endpoint"""
- vertex_base = VertexBase()
-
- # Test with API key and no credentials - should use Google AI Studio endpoint
- auth_header, url = vertex_base._get_token_and_url(
- model="gemini-2.0-flash-exp",
- auth_header=None,
- gemini_api_key="test-api-key-123",
- vertex_project="test-project",
- vertex_location="us-central1",
- vertex_credentials=None, # No service account credentials
- stream=False,
- custom_llm_provider="vertex_ai",
- api_base=None,
- should_use_v1beta1_features=False,
- mode="chat",
- )
-
- # Should route to Google AI Studio endpoint
- assert "generativelanguage.googleapis.com" in url
- assert "gemini-2.0-flash-exp" in url
- assert "key=test-api-key-123" in url
- assert auth_header is None # API key is in URL, not header
-
- def test_get_token_and_url_with_credentials(self):
- """Test that service account credentials route to Vertex AI endpoint"""
- vertex_base = VertexBase()
-
- mock_creds = MagicMock()
- mock_creds.token = "mock-bearer-token"
- mock_creds.expired = False
-
- with patch.object(
- vertex_base, "_ensure_access_token", return_value=("mock-bearer-token", "test-project")
- ):
- # Test with credentials - should use Vertex AI endpoint
- auth_header, url = vertex_base._get_token_and_url(
- model="gemini-2.0-flash-exp",
- auth_header="mock-bearer-token",
- gemini_api_key=None,
- vertex_project="test-project",
- vertex_location="us-central1",
- vertex_credentials={"type": "service_account"},
- stream=False,
- custom_llm_provider="vertex_ai",
- api_base=None,
- should_use_v1beta1_features=False,
- mode="chat",
- )
-
- # Should route to Vertex AI endpoint
- assert "aiplatform.googleapis.com" in url
- assert "projects/test-project" in url
- assert "locations/us-central1" in url
- assert auth_header == "mock-bearer-token"
-
- def test_get_token_and_url_api_key_with_streaming(self):
- """Test API key authentication with streaming enabled"""
- vertex_base = VertexBase()
-
- auth_header, url = vertex_base._get_token_and_url(
- model="gemini-2.0-flash-exp",
- auth_header=None,
- gemini_api_key="test-api-key-456",
- vertex_project="test-project",
- vertex_location="us-central1",
- vertex_credentials=None,
- stream=True, # Streaming enabled
- custom_llm_provider="vertex_ai",
- api_base=None,
- should_use_v1beta1_features=False,
- mode="chat",
- )
-
- # Should route to Google AI Studio endpoint with streaming
- assert "generativelanguage.googleapis.com" in url
- assert "streamGenerateContent" in url
- assert "key=test-api-key-456" in url
- assert "alt=sse" in url
- assert auth_header is None
-
- def test_get_token_and_url_api_key_priority(self):
- """Test that credentials take priority over API key when both are provided"""
- vertex_base = VertexBase()
-
- # When both API key and credentials are provided, credentials take priority
- mock_creds = MagicMock()
- mock_creds.token = "mock-bearer-token"
- mock_creds.expired = False
-
- with patch.object(
- vertex_base, "_ensure_access_token", return_value=("mock-bearer-token", "test-project")
- ):
- auth_header, url = vertex_base._get_token_and_url(
- model="gemini-2.0-flash-exp",
- auth_header="mock-bearer-token",
- gemini_api_key="test-api-key-789",
- vertex_project="test-project",
- vertex_location="us-central1",
- vertex_credentials={"type": "service_account"}, # Credentials provided
- stream=False,
- custom_llm_provider="vertex_ai",
- api_base=None,
- should_use_v1beta1_features=False,
- mode="chat",
- )
-
- # Should use Vertex AI endpoint with Bearer token (credentials take priority)
- assert "aiplatform.googleapis.com" in url
- assert auth_header == "mock-bearer-token"
-
- def test_get_token_and_url_with_embedding_mode(self):
- """Test API key authentication with embedding mode"""
- vertex_base = VertexBase()
-
- auth_header, url = vertex_base._get_token_and_url(
- model="text-embedding-004",
- auth_header=None,
- gemini_api_key="test-embedding-key",
- vertex_project="test-project",
- vertex_location="us-central1",
- vertex_credentials=None,
- stream=False,
- custom_llm_provider="vertex_ai",
- api_base=None,
- should_use_v1beta1_features=False,
- mode="embedding",
- )
-
- # Should route to Google AI Studio endpoint for embeddings
- assert "generativelanguage.googleapis.com" in url
- assert "embedContent" in url
- assert "key=test-embedding-key" in url
- assert auth_header is None
\ No newline at end of file