Merge branch 'main' into litellm_oom_tests_002

This commit is contained in:
Alexsander Hamir 2026-01-10 17:16:06 -08:00
commit f0d5c0b1d1
40 changed files with 2731 additions and 926 deletions

View file

@ -9,7 +9,7 @@ Use Manus AI agents through LiteLLM's OpenAI-compatible Responses API.
|----------|---------|
| Description | Manus is an AI agent platform for complex reasoning tasks, document analysis, and multi-step workflows with asynchronous task execution. |
| Provider Route on LiteLLM | `manus/{agent_profile}` |
| Supported Operations | `/responses` (Responses API) |
| Supported Operations | `/responses` (Responses API), `/files` (Files API) |
| Provider Doc | [Manus API ↗](https://open.manus.im/docs/openai-compatibility) |
## Model Format
@ -188,7 +188,182 @@ For production applications, use [webhooks](https://open.manus.im/docs/webhooks)
| `max_output_tokens` | ✅ | Limits response length |
| `previous_response_id` | ✅ | For multi-turn conversations |
## Files API
Manus supports file uploads for document analysis and processing. Files can be uploaded and then referenced in Responses API calls.
### LiteLLM Python SDK
```python showLineNumbers title="Upload, Use, Retrieve, and Delete Files"
import litellm
import os
# Set API key
os.environ["MANUS_API_KEY"] = "your-manus-api-key"
# Upload file
file_content = b"This is a document for analysis."
created_file = await litellm.acreate_file(
file=("document.txt", file_content),
purpose="assistants",
custom_llm_provider="manus",
)
print(f"Uploaded file: {created_file.id}")
# Use file with Responses API
response = await litellm.aresponses(
model="manus/manus-1.6",
input=[
{
"role": "user",
"content": [
{"type": "input_text", "text": "Summarize this document."},
{"type": "input_file", "file_id": created_file.id},
],
},
],
extra_body={"task_mode": "agent", "agent_profile": "manus-1.6-agent"},
)
print(f"Response: {response.id}")
# Retrieve file
retrieved_file = await litellm.afile_retrieve(
file_id=created_file.id,
custom_llm_provider="manus",
)
print(f"File details: {retrieved_file.filename}, {retrieved_file.bytes} bytes")
# Delete file
deleted_file = await litellm.afile_delete(
file_id=created_file.id,
custom_llm_provider="manus",
)
print(f"Deleted: {deleted_file.deleted}")
```
### LiteLLM AI Gateway
<Tabs>
<TabItem value="curl" label="cURL">
```bash showLineNumbers title="Upload File"
# Upload file
curl -X POST http://localhost:4000/v1/files \
-H "Authorization: Bearer your-proxy-key" \
-F "file=@document.txt" \
-F "purpose=assistants" \
-F "custom_llm_provider=manus"
# Response
{
"id": "file_abc123",
"object": "file",
"bytes": 1024,
"created_at": 1234567890,
"filename": "document.txt",
"purpose": "assistants",
"status": "uploaded"
}
```
```bash showLineNumbers title="Use File with Responses API"
# Create response with file
curl -X POST http://localhost:4000/responses \
-H "Authorization: Bearer your-proxy-key" \
-H "Content-Type: application/json" \
-d '{
"model": "manus-agent",
"input": [
{
"role": "user",
"content": [
{"type": "input_text", "text": "Summarize this document."},
{"type": "input_file", "file_id": "file_abc123"}
]
}
]
}'
```
```bash showLineNumbers title="Retrieve File"
# Get file details
curl http://localhost:4000/v1/files/file_abc123 \
-H "Authorization: Bearer your-proxy-key"
# Response
{
"id": "file_abc123",
"object": "file",
"bytes": 1024,
"created_at": 1234567890,
"filename": "document.txt",
"purpose": "assistants",
"status": "uploaded"
}
```
```bash showLineNumbers title="Delete File"
# Delete file
curl -X DELETE http://localhost:4000/v1/files/file_abc123 \
-H "Authorization: Bearer your-proxy-key"
# Response
{
"id": "file_abc123",
"object": "file",
"deleted": true
}
```
</TabItem>
<TabItem value="openai" label="OpenAI SDK">
```python showLineNumbers title="Upload, Use, Retrieve, and Delete Files"
import openai
client = openai.OpenAI(
base_url="http://localhost:4000",
api_key="your-proxy-key"
)
# Upload file
with open("document.txt", "rb") as f:
created_file = client.files.create(
file=f,
purpose="assistants",
extra_body={"custom_llm_provider": "manus"}
)
print(f"Uploaded file: {created_file.id}")
# Use file with Responses API
response = client.responses.create(
model="manus-agent",
input=[
{
"role": "user",
"content": [
{"type": "input_text", "text": "Summarize this document."},
{"type": "input_file", "file_id": created_file.id}
]
}
]
)
print(f"Response: {response.id}")
# Retrieve file
retrieved_file = client.files.retrieve(created_file.id)
print(f"File: {retrieved_file.filename}, {retrieved_file.bytes} bytes")
# Delete file
deleted_file = client.files.delete(created_file.id)
print(f"Deleted: {deleted_file.deleted}")
```
</TabItem>
</Tabs>
## Related Documentation
- [LiteLLM Responses API](/docs/response_api)
- [LiteLLM Files API](/docs/proxy/litellm_managed_files)
- [Manus OpenAI Compatibility](https://open.manus.im/docs/openai-compatibility)

View file

@ -35,8 +35,6 @@ import json
# !gcloud auth application-default login - run this to add vertex credentials to your env
## OR ##
file_path = 'path/to/vertex_ai_service_account.json'
## OR ##
export VERTEXAI_API_KEY="your-api-key"
# Load the JSON file
with open(file_path, 'r') as file:
@ -49,7 +47,7 @@ vertex_credentials_json = json.dumps(vertex_credentials)
response = completion(
model="vertex_ai/gemini-2.5-pro",
messages=[{ "content": "Hello, how are you?","role": "user"}],
vertex_credentials=vertex_credentials_json # Can remove this is added VERTEXAI_API_KEY in env
vertex_credentials=vertex_credentials_json
)
```
@ -1331,41 +1329,15 @@ Here's how to use Vertex AI with the LiteLLM Proxy Server
## Authentication - vertex_project, vertex_location, etc.
LiteLLM supports two authentication methods for Vertex AI:
1. **API Key Authentication** (Recommended for getting started)
2. **Service Account Credentials** (Recommended for production)
Set your vertex credentials via:
- dynamic params
OR
- env vars
### **Authentication Method 1:
The simplest way to authenticate with Vertex AI. You can set:
- `api_key` (str) - Your Vertex AI API key
### **Dynamic Params**
**Environment Variables:**
```bash
export VERTEXAI_API_KEY="your-api-key"
```
**Or pass as parameters:**
```python
from litellm import completion
response = completion(
model="vertex_ai/gemini-2.0-flash-exp",
messages=[{"role": "user", "content": "Hello!"}],
api_key="your-vertex-api-key",
)
```
### **Authentication Method 2: Service Account Credentials**
For production environments with fine-grained access control. You can set:
You can set:
- `vertex_credentials` (str) - can be a json string or filepath to your vertex ai service account.json
- `vertex_location` (str) - place where vertex model is deployed (us-central1, asia-southeast1, etc.). Some models support the global location, please see [Vertex AI documentation](https://cloud.google.com/vertex-ai/generative-ai/docs/learn/locations#supported_models)
- `vertex_project` Optional[str] - use if vertex project different from the one in vertex_credentials
@ -1420,16 +1392,7 @@ model_list:
### **Environment Variables**
#### For API Key Authentication:
- `VERTEXAI_API_KEY` or `VERTEX_API_KEY` - Your Vertex AI API key
```bash
export VERTEXAI_API_KEY="your-vertex-api-key"
```
#### For Service Account Authentication:
You can set:
- `GOOGLE_APPLICATION_CREDENTIALS` - store the filepath for your service_account.json in here (used by vertex sdk directly).
- VERTEXAI_LOCATION - place where vertex model is deployed (us-central1, asia-southeast1, etc.)
- VERTEXAI_PROJECT - Optional[str] - use if vertex project different from the one in vertex_credentials

Binary file not shown.

After

Width:  |  Height:  |  Size: 503 KiB

View file

@ -1,5 +1,5 @@
---
title: "[Preview] v1.80.11 - Google Interactions API"
title: "v1.80.11 - Google Interactions API"
slug: "v1-80-11"
date: 2025-12-20T10:00:00
authors:
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
docker.litellm.ai/berriai/litellm:v1.80.11.rc.1
docker.litellm.ai/berriai/litellm:v1.80.11-stable
```
</TabItem>

View file

@ -0,0 +1,589 @@
---
title: "v1.80.14 - Manus API Support"
slug: "v1-80-14"
date: 2026-01-10T10:00:00
authors:
- name: Krrish Dholakia
title: CEO, LiteLLM
url: https://www.linkedin.com/in/krish-d/
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
- name: Ishaan Jaff
title: CTO, LiteLLM
url: https://www.linkedin.com/in/reffajnaahsi/
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
hide_table_of_contents: false
---
import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
## Deploy this version
<Tabs>
<TabItem value="docker" label="Docker">
``` showLineNumbers title="docker run litellm"
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
docker.litellm.ai/berriai/litellm:v1.80.14-stable
```
</TabItem>
<TabItem value="pip" label="Pip">
``` showLineNumbers title="pip install litellm"
pip install litellm==1.80.14
```
</TabItem>
</Tabs>
---
## Key Highlights
- **Manus API Support** - [New provider support for Manus API on /responses and GET /responses endpoints](../../docs/providers/manus)
- **MiniMax Provider** - [Full support for MiniMax chat completions, TTS, and Anthropic native endpoint](../../docs/providers/minimax)
- **AWS Polly TTS** - [New TTS provider using AWS Polly API](../../docs/providers/aws_polly)
- **SSO Role Mapping** - Configure role mappings for SSO providers directly in the UI
- **Cost Estimator** - New UI tool for estimating costs across multiple models and requests
- **MCP Global Mode** - [Configure MCP servers globally with visibility controls](../../docs/mcp)
- **Interactions API Bridge** - [Use all LiteLLM providers with the Interactions API](../../docs/interactions)
- **RAG Query Endpoint** - [New RAG Search/Query endpoint for retrieval-augmented generation](../../docs/search/index)
- **92.7% Faster Provider Config Lookup** - Major performance improvement for provider configuration
- **UI Usage - Endpoint Activity** - Users can now see Endpoint Activity Metrics in the UI.
---
### UI Usage - Endpoint Activity
<Image
img={require('../../img/ui_endpoint_activity.png')}
style={{width: '100%', display: 'block', margin: '2rem auto'}}
/>
Users can now see Endpoint Activity Metrics in the UI.
---
## New Providers and Endpoints
### New Providers (11 new providers)
| Provider | Supported LiteLLM Endpoints | Description |
| -------- | ------------------- | ----------- |
| [Manus](../../docs/providers/manus) | `/responses` | Manus API for agentic workflows |
| [Manus](../../docs/providers/manus) | `GET /responses` | Manus API for retrieving responses |
| [Manus](../../docs/providers/manus) | `/files` | Manus API for file management |
| [MiniMax](../../docs/providers/minimax) | `/chat/completions` | MiniMax chat completions |
| [MiniMax](../../docs/providers/minimax) | `/audio/speech` | MiniMax text-to-speech |
| [AWS Polly](../../docs/providers/aws_polly) | `/audio/speech` | AWS Polly text-to-speech API |
| [GigaChat](../../docs/providers/gigachat) | `/chat/completions` | GigaChat provider for Russian language AI |
| [LlamaGate](../../docs/providers/llamagate) | `/chat/completions` | LlamaGate chat completions |
| [LlamaGate](../../docs/providers/llamagate) | `/embeddings` | LlamaGate embeddings |
| [Abliteration AI](../../docs/providers/abliteration) | `/chat/completions` | Abliteration.ai provider support |
| [Bedrock](../../docs/providers/bedrock) | `/v1/messages/count_tokens` | Bedrock as new provider for token counting |
### New LLM API Endpoints (3 new endpoints)
| Endpoint | Method | Description | Documentation |
| -------- | ------ | ----------- | ------------- |
| `/responses/compact` | POST | Compact responses API endpoint | [Docs](../../docs/response_api) |
| `/rag/query` | POST | RAG Search/Query endpoint | [Docs](../../docs/search/index) |
| `/containers/{id}/files` | POST | Upload files to containers | [Docs](../../docs/container_files) |
---
## New Models / Updated Models
#### New Model Support (100+ new models)
| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Features |
| -------- | ----- | -------------- | ------------------- | -------------------- | -------- |
| Azure | `azure/gpt-5.2` | 400K | $1.75 | $14.00 | Reasoning, vision, caching |
| Azure | `azure/gpt-5.2-chat` | 128K | $1.75 | $14.00 | Reasoning, vision |
| Azure | `azure/gpt-5.2-pro` | 400K | $21.00 | $168.00 | Reasoning, vision, web search |
| Azure | `azure/gpt-image-1.5` | - | Token-based | Token-based | Image generation/editing |
| Azure AI | `azure_ai/gpt-oss-120b` | 131K | $0.15 | $0.60 | Function calling |
| Azure AI | `azure_ai/flux.2-pro` | - | - | $0.04/image | Image generation |
| Azure AI | `azure_ai/deepseek-v3.2` | 164K | $0.58 | $1.68 | Reasoning, function calling |
| Bedrock | `amazon.nova-2-multimodal-embeddings-v1:0` | 8K | $0.135 | - | Multimodal embeddings |
| Bedrock | `writer.palmyra-x4-v1:0` | 128K | $2.50 | $10.00 | Function calling, PDF |
| Bedrock | `writer.palmyra-x5-v1:0` | 1M | $0.60 | $6.00 | Function calling, PDF |
| Bedrock | `moonshot.kimi-k2-v1:0` | - | - | - | Kimi K2 model |
| Cerebras | `cerebras/zai-glm-4.6` | 128K | $2.25 | $2.75 | Reasoning, function calling |
| GigaChat | `gigachat/GigaChat-2-Lite` | - | - | - | Chat completions |
| GigaChat | `gigachat/GigaChat-2-Max` | - | - | - | Chat completions |
| GigaChat | `gigachat/GigaChat-2-Pro` | - | - | - | Chat completions |
| Gemini | `gemini/veo-3.1-generate-001` | - | - | - | Video generation |
| Gemini | `gemini/veo-3.1-fast-generate-001` | - | - | - | Video generation |
| GitHub Copilot | 25+ models | Various | - | - | Chat completions |
| LlamaGate | 15+ models | Various | - | - | Chat, vision, embeddings |
| MiniMax | `minimax/abab7-chat-preview` | - | - | - | Chat completions |
| Novita | 80+ models | Various | Various | Various | Chat, vision, embeddings |
| OpenRouter | `openrouter/google/gemini-3-flash-preview` | - | - | - | Chat completions |
| Together AI | Multiple models | Various | Various | Various | Response schema support |
| Vertex AI | `vertex_ai/zai-glm-4.7` | - | - | - | GLM 4.7 support |
#### Features
- **[Gemini](../../docs/providers/gemini)**
- Add image tokens in chat completion - [PR #18327](https://github.com/BerriAI/litellm/pull/18327)
- Add usage object in image generation - [PR #18328](https://github.com/BerriAI/litellm/pull/18328)
- Add thought signature support via tool call id - [PR #18374](https://github.com/BerriAI/litellm/pull/18374)
- Add thought signature for non tool call requests - [PR #18581](https://github.com/BerriAI/litellm/pull/18581)
- Preserve system instructions - [PR #18585](https://github.com/BerriAI/litellm/pull/18585)
- Fix Gemini 3 images in tool response - [PR #18190](https://github.com/BerriAI/litellm/pull/18190)
- Support snake_case for google_search tool parameters - [PR #18451](https://github.com/BerriAI/litellm/pull/18451)
- Google GenAI adapter inline data support - [PR #18477](https://github.com/BerriAI/litellm/pull/18477)
- Add deprecation_date for discontinued Google models - [PR #18550](https://github.com/BerriAI/litellm/pull/18550)
- **[Vertex AI](../../docs/providers/vertex)**
- Add centralized get_vertex_base_url() helper for global location support - [PR #18410](https://github.com/BerriAI/litellm/pull/18410)
- Convert image URLs to base64 for Vertex AI Anthropic - [PR #18497](https://github.com/BerriAI/litellm/pull/18497)
- Separate Tool objects for each tool type per API spec - [PR #18514](https://github.com/BerriAI/litellm/pull/18514)
- Add thought_signatures to VertexGeminiConfig - [PR #18853](https://github.com/BerriAI/litellm/pull/18853)
- Add support for Vertex AI API keys - [PR #18806](https://github.com/BerriAI/litellm/pull/18806)
- Add zai glm-4.7 model support - [PR #18782](https://github.com/BerriAI/litellm/pull/18782)
- **[Azure](../../docs/providers/azure/azure)**
- Add Azure gpt-image-1.5 pricing to cost map - [PR #18347](https://github.com/BerriAI/litellm/pull/18347)
- Add azure/gpt-5.2-chat model - [PR #18361](https://github.com/BerriAI/litellm/pull/18361)
- Add support for image generation via Azure AD token - [PR #18413](https://github.com/BerriAI/litellm/pull/18413)
- Add logprobs support for Azure OpenAI GPT-5.2 model - [PR #18856](https://github.com/BerriAI/litellm/pull/18856)
- Add Azure BFL Flux 2 models for image generation and editing - [PR #18764](https://github.com/BerriAI/litellm/pull/18764), [PR #18766](https://github.com/BerriAI/litellm/pull/18766)
- **[Bedrock](../../docs/providers/bedrock)**
- Add Bedrock Kimi K2 model support - [PR #18797](https://github.com/BerriAI/litellm/pull/18797)
- Add support for model id in bedrock passthrough - [PR #18800](https://github.com/BerriAI/litellm/pull/18800)
- Fix Nova model detection for Bedrock provider - [PR #18250](https://github.com/BerriAI/litellm/pull/18250)
- Ensure toolUse.input is always a dict when converting from OpenAI format - [PR #18414](https://github.com/BerriAI/litellm/pull/18414)
- **[Databricks](../../docs/providers/databricks)**
- Add enhanced authentication, security features, and custom user-agent support - [PR #18349](https://github.com/BerriAI/litellm/pull/18349)
- **[MiniMax](../../docs/providers/minimax)**
- Add MiniMax chat completion support - [PR #18380](https://github.com/BerriAI/litellm/pull/18380)
- Add Anthropic native endpoint support for MiniMax - [PR #18377](https://github.com/BerriAI/litellm/pull/18377)
- Add support for MiniMax TTS - [PR #18334](https://github.com/BerriAI/litellm/pull/18334)
- Add MiniMax provider support to UI dashboard - [PR #18496](https://github.com/BerriAI/litellm/pull/18496)
- **[Together AI](../../docs/providers/togetherai)**
- Add supports_response_schema to all supported Together AI models - [PR #18368](https://github.com/BerriAI/litellm/pull/18368)
- **[OpenRouter](../../docs/providers/openrouter)**
- Add OpenRouter embeddings API support - [PR #18391](https://github.com/BerriAI/litellm/pull/18391)
- **[Anthropic](../../docs/providers/anthropic)**
- Pass server_tool_use and tool_search_tool_result blocks - [PR #18770](https://github.com/BerriAI/litellm/pull/18770)
- Add Anthropic cache control option to image tool call results - [PR #18674](https://github.com/BerriAI/litellm/pull/18674)
- **[Ollama](../../docs/providers/ollama)**
- Add dimensions for ollama embedding - [PR #18536](https://github.com/BerriAI/litellm/pull/18536)
- Extract pure base64 data from data URLs for Ollama - [PR #18465](https://github.com/BerriAI/litellm/pull/18465)
- **[Watsonx](../../docs/providers/watsonx/index)**
- Add Watsonx fields support - [PR #18569](https://github.com/BerriAI/litellm/pull/18569)
- Fix Watsonx Audio Transcription - filter model field - [PR #18810](https://github.com/BerriAI/litellm/pull/18810)
- **[SAP](../../docs/providers/sap)**
- Add SAP creds for list in proxy UI - [PR #18375](https://github.com/BerriAI/litellm/pull/18375)
- Pass through extra params from allowed_openai_params - [PR #18432](https://github.com/BerriAI/litellm/pull/18432)
- Add client header for SAP AI Core Tracking - [PR #18714](https://github.com/BerriAI/litellm/pull/18714)
- **[Fireworks AI](../../docs/providers/fireworks_ai)**
- Correct deepseek-v3p2 pricing - [PR #18483](https://github.com/BerriAI/litellm/pull/18483)
- **[ZAI](../../docs/providers/zai)**
- Add GLM-4.7 model with reasoning support - [PR #18476](https://github.com/BerriAI/litellm/pull/18476)
- **[Codestral](../../docs/providers/codestral)**
- Correctly route codestral chat and FIM endpoints - [PR #18467](https://github.com/BerriAI/litellm/pull/18467)
- **[Azure AI](../../docs/providers/azure_ai)**
- Fix authentication errors at messages API via azure_ai - [PR #18500](https://github.com/BerriAI/litellm/pull/18500)
#### New Provider Support
- **[AWS Polly](../../docs/providers/aws_polly)** - Add AWS Polly API for TTS - [PR #18326](https://github.com/BerriAI/litellm/pull/18326)
- **[GigaChat](../../docs/providers/gigachat)** - Add GigaChat provider support - [PR #18564](https://github.com/BerriAI/litellm/pull/18564)
- **[LlamaGate](../../docs/providers/llamagate)** - Add LlamaGate as a new provider - [PR #18673](https://github.com/BerriAI/litellm/pull/18673)
- **[Abliteration AI](../../docs/providers/abliteration)** - Add abliteration.ai provider - [PR #18678](https://github.com/BerriAI/litellm/pull/18678)
- **[Manus](../../docs/providers/manus)** - Add Manus API support on /responses, GET /responses - [PR #18804](https://github.com/BerriAI/litellm/pull/18804)
- **5 AI Providers via openai_like** - Add 5 AI providers using openai_like - [PR #18362](https://github.com/BerriAI/litellm/pull/18362)
### Bug Fixes
- **[Gemini](../../docs/providers/gemini)**
- Properly catch context window exceeded errors - [PR #18283](https://github.com/BerriAI/litellm/pull/18283)
- Remove prompt caching headers as support has been removed - [PR #18579](https://github.com/BerriAI/litellm/pull/18579)
- Fix generate content request with audio file id - [PR #18745](https://github.com/BerriAI/litellm/pull/18745)
- Fix google_genai streaming adapter provider handling - [PR #18845](https://github.com/BerriAI/litellm/pull/18845)
- **[Groq](../../docs/providers/groq)**
- Remove deprecated Groq models and update model registry - [PR #18062](https://github.com/BerriAI/litellm/pull/18062)
- **[Vertex AI](../../docs/providers/vertex)**
- Handle unsupported region for Vertex AI count tokens endpoint - [PR #18665](https://github.com/BerriAI/litellm/pull/18665)
- **General**
- Fix request body for image embedding request - [PR #18336](https://github.com/BerriAI/litellm/pull/18336)
- Fix lost tool_calls when streaming has both text and tool_calls - [PR #18316](https://github.com/BerriAI/litellm/pull/18316)
- Add all resolution for gpt-image-1.5 - [PR #18586](https://github.com/BerriAI/litellm/pull/18586)
- Fix gpt-image-1 cost calculation using token-based pricing - [PR #17906](https://github.com/BerriAI/litellm/pull/17906)
- Fix response_format leaking into extra_body - [PR #18859](https://github.com/BerriAI/litellm/pull/18859)
- Align max_tokens with max_output_tokens for consistency - [PR #18820](https://github.com/BerriAI/litellm/pull/18820)
---
## LLM API Endpoints
#### Features
- **[Responses API](../../docs/response_api)**
- Add new compact endpoint (v1/responses/compact) - [PR #18697](https://github.com/BerriAI/litellm/pull/18697)
- Support more streaming callback hooks - [PR #18513](https://github.com/BerriAI/litellm/pull/18513)
- Add mapping for reasoning effort to summary param - [PR #18635](https://github.com/BerriAI/litellm/pull/18635)
- Add output_text property to ResponsesAPIResponse - [PR #18491](https://github.com/BerriAI/litellm/pull/18491)
- Add annotations to completions responses API bridge - [PR #18754](https://github.com/BerriAI/litellm/pull/18754)
- **[Interactions API](../../docs/interactions)**
- Allow using all LiteLLM providers (interactions -> responses API bridge) - [PR #18373](https://github.com/BerriAI/litellm/pull/18373)
- **[RAG Search API](../../docs/search/index)**
- Add RAG Search/Query endpoint - [PR #18376](https://github.com/BerriAI/litellm/pull/18376)
- **[CountTokens API](../../docs/anthropic_count_tokens)**
- Add Bedrock as a new provider for `/v1/messages/count_tokens` - [PR #18858](https://github.com/BerriAI/litellm/pull/18858)
- **[Generate Content](../../docs/providers/gemini)**
- Add generate content in LLM route - [PR #18405](https://github.com/BerriAI/litellm/pull/18405)
- **General**
- Enable async_post_call_failure_hook to transform error responses - [PR #18348](https://github.com/BerriAI/litellm/pull/18348)
- Calculate total_tokens manually if missing and can be calculated - [PR #18445](https://github.com/BerriAI/litellm/pull/18445)
- Add custom llm provider to get_llm_provider when sent via UI - [PR #18638](https://github.com/BerriAI/litellm/pull/18638)
#### Bugs
- **General**
- Handle empty error objects in response conversion - [PR #18493](https://github.com/BerriAI/litellm/pull/18493)
- Preserve client error status codes in streaming mode - [PR #18698](https://github.com/BerriAI/litellm/pull/18698)
- Return json error response instead of SSE format for initial streaming errors - [PR #18757](https://github.com/BerriAI/litellm/pull/18757)
- Fix auth header for custom api base in generateContent request - [PR #18637](https://github.com/BerriAI/litellm/pull/18637)
- Tool content should be string for Deepinfra - [PR #18739](https://github.com/BerriAI/litellm/pull/18739)
- Fix incomplete usage in response object passed - [PR #18799](https://github.com/BerriAI/litellm/pull/18799)
- Unify model names to provider-defined names - [PR #18573](https://github.com/BerriAI/litellm/pull/18573)
---
## Management Endpoints / UI
#### Features
- **SSO Configuration**
- Add SSO Role Mapping feature - [PR #18090](https://github.com/BerriAI/litellm/pull/18090)
- Add SSO Settings Page - [PR #18600](https://github.com/BerriAI/litellm/pull/18600)
- Allow adding role mappings for SSO - [PR #18593](https://github.com/BerriAI/litellm/pull/18593)
- SSO Settings Page Add Role Mappings - [PR #18677](https://github.com/BerriAI/litellm/pull/18677)
- SSO Settings Loading State + Deprecate Previous SSO Flow - [PR #18617](https://github.com/BerriAI/litellm/pull/18617)
- **Virtual Keys**
- Allow deleting key expiry - [PR #18278](https://github.com/BerriAI/litellm/pull/18278)
- Add optional query param "expand" to /key/list - [PR #18502](https://github.com/BerriAI/litellm/pull/18502)
- Key Table Loading Skeleton - [PR #18527](https://github.com/BerriAI/litellm/pull/18527)
- Allow column resizing on Keys Table - [PR #18424](https://github.com/BerriAI/litellm/pull/18424)
- Virtual Keys Table Loading State Between Pages - [PR #18619](https://github.com/BerriAI/litellm/pull/18619)
- Key and Team Router Setting - [PR #18790](https://github.com/BerriAI/litellm/pull/18790)
- Allow router_settings on Keys and Teams - [PR #18675](https://github.com/BerriAI/litellm/pull/18675)
- Use timedelta to calculate key expiry on generate - [PR #18666](https://github.com/BerriAI/litellm/pull/18666)
- **Models + Endpoints**
- Add Model Clearer Flow For Team Admins - [PR #18532](https://github.com/BerriAI/litellm/pull/18532)
- Model Page Loading State - [PR #18574](https://github.com/BerriAI/litellm/pull/18574)
- Model Page Model Provider Select Performance - [PR #18425](https://github.com/BerriAI/litellm/pull/18425)
- Model Page Sorting Sorts Entire Set - [PR #18420](https://github.com/BerriAI/litellm/pull/18420)
- Refactor Model Hub Page - [PR #18568](https://github.com/BerriAI/litellm/pull/18568)
- Add request provider form on UI - [PR #18704](https://github.com/BerriAI/litellm/pull/18704)
- **Organizations & Teams**
- Allow Organization Admins to See Organization Tab - [PR #18400](https://github.com/BerriAI/litellm/pull/18400)
- Resolve Organization Alias on Team Table - [PR #18401](https://github.com/BerriAI/litellm/pull/18401)
- Resolve Team Alias in Organization Info View - [PR #18404](https://github.com/BerriAI/litellm/pull/18404)
- Allow Organization Admins to View Their Organization Info - [PR #18417](https://github.com/BerriAI/litellm/pull/18417)
- Allow editing team_member_budget_duration in /team/update - [PR #18735](https://github.com/BerriAI/litellm/pull/18735)
- Reusable Duration Select + Team Update Member Budget Duration - [PR #18736](https://github.com/BerriAI/litellm/pull/18736)
- **Usage & Spend**
- Add Error Code Filtering on Spend Logs - [PR #18359](https://github.com/BerriAI/litellm/pull/18359)
- Add Error Code Filtering on UI - [PR #18366](https://github.com/BerriAI/litellm/pull/18366)
- Usage Page User Max Budget fix - [PR #18555](https://github.com/BerriAI/litellm/pull/18555)
- Add endpoint to Daily Activity Tables - [PR #18729](https://github.com/BerriAI/litellm/pull/18729)
- Endpoint Activity in Usage - [PR #18798](https://github.com/BerriAI/litellm/pull/18798)
- **Cost Estimator**
- Add Cost Estimator for AI Gateway - [PR #18643](https://github.com/BerriAI/litellm/pull/18643)
- Add view for estimating costs across requests - [PR #18645](https://github.com/BerriAI/litellm/pull/18645)
- Allow selecting many models for cost estimator - [PR #18653](https://github.com/BerriAI/litellm/pull/18653)
- **CloudZero**
- Improve Create and Delete Path for CloudZero - [PR #18263](https://github.com/BerriAI/litellm/pull/18263)
- Add CloudZero UI Docs - [PR #18350](https://github.com/BerriAI/litellm/pull/18350)
- **Playground**
- Add MCP test support to completions on Playground - [PR #18440](https://github.com/BerriAI/litellm/pull/18440)
- Add selectable MCP servers to the playground - [PR #18578](https://github.com/BerriAI/litellm/pull/18578)
- Add custom proxy base URL support to Playground - [PR #18661](https://github.com/BerriAI/litellm/pull/18661)
- **General UI**
- UI styling improvements and fixes - [PR #18310](https://github.com/BerriAI/litellm/pull/18310)
- Add reusable "New" badge component for feature highlights - [PR #18537](https://github.com/BerriAI/litellm/pull/18537)
- Hide New Badges - [PR #18547](https://github.com/BerriAI/litellm/pull/18547)
- Change Budget page to Have Tabs - [PR #18576](https://github.com/BerriAI/litellm/pull/18576)
- Clicking on Logo Directs to Correct URL - [PR #18575](https://github.com/BerriAI/litellm/pull/18575)
- Add UI support for configuring meta URLs - [PR #18580](https://github.com/BerriAI/litellm/pull/18580)
- Expire Previous UI Session Tokens on Login - [PR #18557](https://github.com/BerriAI/litellm/pull/18557)
- Add license endpoint - [PR #18311](https://github.com/BerriAI/litellm/pull/18311)
- Router Fields Endpoint + React Query for Router Fields - [PR #18880](https://github.com/BerriAI/litellm/pull/18880)
#### Bugs
- **UI Fixes**
- Fix Key Creation MCP Settings Submit Form Unintentionally - [PR #18355](https://github.com/BerriAI/litellm/pull/18355)
- Fix UI Disappears in Development Environments - [PR #18399](https://github.com/BerriAI/litellm/pull/18399)
- Fix Disable Admin UI Flag - [PR #18397](https://github.com/BerriAI/litellm/pull/18397)
- Remove Model Analytics From Model Page - [PR #18552](https://github.com/BerriAI/litellm/pull/18552)
- Useful Links Remove Modal on Adding Links - [PR #18602](https://github.com/BerriAI/litellm/pull/18602)
- SSO Edit Modal Clear Role Mapping Values on Provider Change - [PR #18680](https://github.com/BerriAI/litellm/pull/18680)
- UI Login Case Sensitivity fix - [PR #18877](https://github.com/BerriAI/litellm/pull/18877)
- **API Fixes**
- Fix User Invite & Key Generation Email Notification Logic - [PR #18524](https://github.com/BerriAI/litellm/pull/18524)
- Normalize Proxy Config Callback - [PR #18775](https://github.com/BerriAI/litellm/pull/18775)
- Return empty data array instead of 500 when no models configured - [PR #18556](https://github.com/BerriAI/litellm/pull/18556)
- Enforce org level max budget - [PR #18813](https://github.com/BerriAI/litellm/pull/18813)
---
## AI Integrations
### New Integrations (4 new integrations)
| Integration | Type | Description |
| ----------- | ---- | ----------- |
| [Focus](../../docs/observability/focus) | Logging | Focus export support for observability - [PR #18802](https://github.com/BerriAI/litellm/pull/18802) |
| [SigNoz](../../docs/observability/signoz) | Logging | SigNoz integration for observability - [PR #18726](https://github.com/BerriAI/litellm/pull/18726) |
| [Qualifire](../../docs/proxy/guardrails/qualifire) | Guardrails | Qualifire guardrails and eval webhook - [PR #18594](https://github.com/BerriAI/litellm/pull/18594) |
| [Levo AI](../../docs/observability/levo_integration) | Guardrails | Levo AI integration for security - [PR #18529](https://github.com/BerriAI/litellm/pull/18529) |
### Logging
- **[DataDog](../../docs/proxy/logging#datadog)**
- Fix span kind fallback when parent_id missing - [PR #18418](https://github.com/BerriAI/litellm/pull/18418)
- **[Langfuse](../../docs/proxy/logging#langfuse)**
- Map Gemini cached_tokens to Langfuse cache_read_input_tokens - [PR #18614](https://github.com/BerriAI/litellm/pull/18614)
- **[Prometheus](../../docs/proxy/logging#prometheus)**
- Align prometheus metric names with DEFINED_PROMETHEUS_METRICS - [PR #18463](https://github.com/BerriAI/litellm/pull/18463)
- Add Prometheus metrics for request queue time and guardrails - [PR #17973](https://github.com/BerriAI/litellm/pull/17973)
- Add caching metrics for cache hits, misses, and tokens - [PR #18755](https://github.com/BerriAI/litellm/pull/18755)
- Skip metrics for invalid API key requests - [PR #18788](https://github.com/BerriAI/litellm/pull/18788)
- **[Braintrust](../../docs/proxy/logging#braintrust)**
- Pass span_attributes in async logging and skip tags on non-root spans - [PR #18409](https://github.com/BerriAI/litellm/pull/18409)
- **[CloudZero](../../docs/proxy/logging#cloudzero)**
- Add user email to CloudZero - [PR #18584](https://github.com/BerriAI/litellm/pull/18584)
- **[OpenTelemetry](../../docs/proxy/logging#opentelemetry)**
- Use already configured opentelemetry providers - [PR #18279](https://github.com/BerriAI/litellm/pull/18279)
- Prevent LiteLLM from closing external OTEL spans - [PR #18553](https://github.com/BerriAI/litellm/pull/18553)
- Allow configuring arize project name for OpenTelemetry service name - [PR #18738](https://github.com/BerriAI/litellm/pull/18738)
- **[LangSmith](../../docs/proxy/logging#langsmith)**
- Add support for LangSmith organization-scoped API keys with tenant ID - [PR #18623](https://github.com/BerriAI/litellm/pull/18623)
- **[Generic API Logger](../../docs/proxy/logging#generic-api-logger)**
- Add log_format option to GenericAPILogger - [PR #18587](https://github.com/BerriAI/litellm/pull/18587)
### Guardrails
- **[Content Filter](../../docs/proxy/guardrails/litellm_content_filter)**
- Add content filter logs page - [PR #18335](https://github.com/BerriAI/litellm/pull/18335)
- Log actual event type for guardrails - [PR #18489](https://github.com/BerriAI/litellm/pull/18489)
- **[Qualifire](../../docs/proxy/guardrails/qualifire)**
- Add Qualifire eval webhook - [PR #18836](https://github.com/BerriAI/litellm/pull/18836)
- **[Lasso Security](../../docs/proxy/guardrails/lasso_security)**
- Add Lasso guardrail API docs - [PR #18652](https://github.com/BerriAI/litellm/pull/18652)
- **[Noma Security](../../docs/proxy/guardrails/noma_security)**
- Add MCP guardrail support for Noma - [PR #18668](https://github.com/BerriAI/litellm/pull/18668)
- **[Bedrock Guardrails](../../docs/proxy/guardrails/bedrock)**
- Remove redundant Bedrock guardrail block handling - [PR #18634](https://github.com/BerriAI/litellm/pull/18634)
- **General**
- Generic guardrail API update - [PR #18647](https://github.com/BerriAI/litellm/pull/18647)
- Prevent proxy startup failures from case-sensitive tool permission guardrail validation - [PR #18662](https://github.com/BerriAI/litellm/pull/18662)
- Extend case normalization to ALL guardrail types - [PR #18664](https://github.com/BerriAI/litellm/pull/18664)
- Fix MCP handling in unified guardrail - [PR #18630](https://github.com/BerriAI/litellm/pull/18630)
- Fix embeddings calltype for guardrail precallhook - [PR #18740](https://github.com/BerriAI/litellm/pull/18740)
---
## Spend Tracking, Budgets and Rate Limiting
- **Platform Fee / Margins** - Add support for Platform Fee / Margins - [PR #18427](https://github.com/BerriAI/litellm/pull/18427)
- **Negative Budget Validation** - Add validation for negative budget - [PR #18583](https://github.com/BerriAI/litellm/pull/18583)
- **Cost Calculation Fixes**
- Correct cost calculation when reasoning_tokens are without text_tokens - [PR #18607](https://github.com/BerriAI/litellm/pull/18607)
- Fix background cost tracking tests - [PR #18588](https://github.com/BerriAI/litellm/pull/18588)
- **Tag Routing** - Support toggling tag matching between ANY and ALL - [PR #18776](https://github.com/BerriAI/litellm/pull/18776)
---
## MCP Gateway
- **MCP Global Mode** - Add MCP global mode - [PR #18639](https://github.com/BerriAI/litellm/pull/18639)
- **MCP Server Visibility** - Add configurable MCP server visibility - [PR #18681](https://github.com/BerriAI/litellm/pull/18681)
- **MCP Registry** - Add MCP registry - [PR #18850](https://github.com/BerriAI/litellm/pull/18850)
- **MCP Stdio Header** - Support MCP stdio header env overrides - [PR #18324](https://github.com/BerriAI/litellm/pull/18324)
- **Parallel Tool Fetching** - Parallelize tool fetching from multiple MCP servers - [PR #18627](https://github.com/BerriAI/litellm/pull/18627)
- **Optimize MCP Server Listing** - Separate health checks for optimized listing - [PR #18530](https://github.com/BerriAI/litellm/pull/18530)
- **Auth Improvements**
- Require auth for MCP connection test endpoint - [PR #18290](https://github.com/BerriAI/litellm/pull/18290)
- Fix MCP gateway OAuth2 auth issues and ClosedResourceError - [PR #18281](https://github.com/BerriAI/litellm/pull/18281)
- **Bug Fixes**
- Fix MCP server health status reporting - [PR #18443](https://github.com/BerriAI/litellm/pull/18443)
- Fix OpenAPI to MCP tool conversion - [PR #18597](https://github.com/BerriAI/litellm/pull/18597)
- Remove exec() usage and handle invalid OpenAPI parameter names for security - [PR #18480](https://github.com/BerriAI/litellm/pull/18480)
- Fix MCP error when using multiple servers simultaneously - [PR #18855](https://github.com/BerriAI/litellm/pull/18855)
- **Migrate MCP Fetching Logic to React Query** - [PR #18352](https://github.com/BerriAI/litellm/pull/18352)
---
## Performance / Loadbalancing / Reliability improvements
- **92.7% Faster Provider Config Lookup** - LiteLLM now stresses LLM providers 2.5x more - [PR #18867](https://github.com/BerriAI/litellm/pull/18867)
- **Lazy Loading Improvements**
- Consolidate lazy import handlers with registry pattern - [PR #18389](https://github.com/BerriAI/litellm/pull/18389)
- Complete lazy loading migration for all 180+ LLM config classes - [PR #18392](https://github.com/BerriAI/litellm/pull/18392)
- Lazy load additional components (types, callbacks, utilities) - [PR #18396](https://github.com/BerriAI/litellm/pull/18396)
- Add lazy loading for get_llm_provider - [PR #18591](https://github.com/BerriAI/litellm/pull/18591)
- Lazy-load heavy audio library and loggers - [PR #18592](https://github.com/BerriAI/litellm/pull/18592)
- Lazy load 9 heavy imports in litellm/utils.py - [PR #18595](https://github.com/BerriAI/litellm/pull/18595)
- Lazy load heavy imports to improve import time and memory usage - [PR #18610](https://github.com/BerriAI/litellm/pull/18610)
- Implement lazy loading for provider configs, model info classes, streaming handlers - [PR #18611](https://github.com/BerriAI/litellm/pull/18611)
- Lazy load 15 additional imports - [PR #18613](https://github.com/BerriAI/litellm/pull/18613)
- Lazy load 15+ unused imports - [PR #18616](https://github.com/BerriAI/litellm/pull/18616)
- Lazy load DatadogLLMObsInitParams - [PR #18658](https://github.com/BerriAI/litellm/pull/18658)
- Migrate utils.py lazy imports to registry pattern - [PR #18657](https://github.com/BerriAI/litellm/pull/18657)
- Lazy load get_llm_provider and remove_index_from_tool_calls - [PR #18608](https://github.com/BerriAI/litellm/pull/18608)
- **Router Improvements**
- Validate routing_strategy at startup to fail fast with helpful error - [PR #18624](https://github.com/BerriAI/litellm/pull/18624)
- Correct num_retries tracking in retry logic - [PR #18712](https://github.com/BerriAI/litellm/pull/18712)
- Improve error messages and validation for wildcard routing with multiple credentials - [PR #18629](https://github.com/BerriAI/litellm/pull/18629)
- **Memory Improvements**
- Add memory pattern detection test and fix bad memory patterns - [PR #18589](https://github.com/BerriAI/litellm/pull/18589)
- Add unbounded data structure detection to memory test - [PR #18590](https://github.com/BerriAI/litellm/pull/18590)
- Add memory leak detection tests with CI integration - [PR #18881](https://github.com/BerriAI/litellm/pull/18881)
- **Database**
- Add idx on LOWER(user_email) for faster duplicate email checks - [PR #18828](https://github.com/BerriAI/litellm/pull/18828)
- Proactive RDS IAM token refresh to prevent 15-min connection failed - [PR #18795](https://github.com/BerriAI/litellm/pull/18795)
- Clarify database_connection_pool_limit applies per worker - [PR #18780](https://github.com/BerriAI/litellm/pull/18780)
- Make base_connection_pool_limit default value the same - [PR #18721](https://github.com/BerriAI/litellm/pull/18721)
- **Docker**
- Add libsndfile to database Docker image for audio processing - [PR #18612](https://github.com/BerriAI/litellm/pull/18612)
- Add line_profiler support for performance analysis and fix Windows CRLF issues - [PR #18773](https://github.com/BerriAI/litellm/pull/18773)
- **Helm**
- Add lifecycle support to Helm charts - [PR #18517](https://github.com/BerriAI/litellm/pull/18517)
- **Authentication**
- Add Kubernetes ServiceAccount JWT authentication support - [PR #18055](https://github.com/BerriAI/litellm/pull/18055)
- Use async anthropic client to prevent event loop blocking - [PR #18435](https://github.com/BerriAI/litellm/pull/18435)
- **Logging Worker**
- Handle event loop changes in multiprocessing - [PR #18423](https://github.com/BerriAI/litellm/pull/18423)
- **Security**
- Prevent expired key plaintext leak in error response - [PR #18860](https://github.com/BerriAI/litellm/pull/18860)
- Mask extra header secrets in model info - [PR #18822](https://github.com/BerriAI/litellm/pull/18822)
- Prevent duplicate User-Agent tags in request_tags - [PR #18723](https://github.com/BerriAI/litellm/pull/18723)
- Properly use litellm api keys - [PR #18832](https://github.com/BerriAI/litellm/pull/18832)
- **Misc**
- Remove double imports in main.py - [PR #18406](https://github.com/BerriAI/litellm/pull/18406)
- Add LITELLM_DISABLE_LAZY_LOADING env var to fix VCR cassette creation issue - [PR #18725](https://github.com/BerriAI/litellm/pull/18725)
- Add xiaomi_mimo to LlmProviders enum to fix router support - [PR #18819](https://github.com/BerriAI/litellm/pull/18819)
- Allow installation with current grpcio on old Python - [PR #18473](https://github.com/BerriAI/litellm/pull/18473)
- Add Custom CA certificates to boto3 clients - [PR #18852](https://github.com/BerriAI/litellm/pull/18852)
- Fix bedrock_cache, metadata and max_model_budget - [PR #18872](https://github.com/BerriAI/litellm/pull/18872)
- Fix LiteLLM SDK embedding headers missing field - [PR #18844](https://github.com/BerriAI/litellm/pull/18844)
- Put automatic reasoning summary inclusion behind feat flag - [PR #18688](https://github.com/BerriAI/litellm/pull/18688)
- turn_off_message_logging Does Not Redact Request Messages in proxy_server_request Field - [PR #18897](https://github.com/BerriAI/litellm/pull/18897)
---
## Documentation Updates
- **Provider Documentation**
- Update MiniMax docs to be in proper format - [PR #18403](https://github.com/BerriAI/litellm/pull/18403)
- Add docs for 5 AI providers - [PR #18388](https://github.com/BerriAI/litellm/pull/18388)
- Fix gpt-5-mini reasoning_effort supported values - [PR #18346](https://github.com/BerriAI/litellm/pull/18346)
- Fix PDF documentation inconsistency in Anthropic page - [PR #18816](https://github.com/BerriAI/litellm/pull/18816)
- Update OpenRouter docs to include embedding support - [PR #18874](https://github.com/BerriAI/litellm/pull/18874)
- Add LITELLM_REASONING_AUTO_SUMMARY in doc - [PR #18705](https://github.com/BerriAI/litellm/pull/18705)
- **MCP Documentation**
- Agentcore MCP server docs - [PR #18603](https://github.com/BerriAI/litellm/pull/18603)
- Mention MCP prompt/resources types in overview - [PR #18669](https://github.com/BerriAI/litellm/pull/18669)
- Add Focus docs - [PR #18837](https://github.com/BerriAI/litellm/pull/18837)
- **Guardrails Documentation**
- Qualifire docs hotfix - [PR #18724](https://github.com/BerriAI/litellm/pull/18724)
- **Infrastructure Documentation**
- IAM Roles Anywhere docs - [PR #18559](https://github.com/BerriAI/litellm/pull/18559)
- Fix formatting in proxy configs documentation - [PR #18498](https://github.com/BerriAI/litellm/pull/18498)
- Fix GCS cache docs missing for proxy mode - [PR #13328](https://github.com/BerriAI/litellm/pull/13328)
- Fix how to execute cloudzero sql - [PR #18841](https://github.com/BerriAI/litellm/pull/18841)
- **General**
- LiteLLM adopters section - [PR #18605](https://github.com/BerriAI/litellm/pull/18605)
- Remove redundant comments about setting litellm.callbacks - [PR #18711](https://github.com/BerriAI/litellm/pull/18711)
- Update header to be markdown bold by removing space - [PR #18846](https://github.com/BerriAI/litellm/pull/18846)
- Manus docs - new provider - [PR #18817](https://github.com/BerriAI/litellm/pull/18817)
---
## New Contributors
* @prasadkona made their first contribution in [PR #18349](https://github.com/BerriAI/litellm/pull/18349)
* @lucasrothman made their first contribution in [PR #18283](https://github.com/BerriAI/litellm/pull/18283)
* @aggeentik made their first contribution in [PR #18317](https://github.com/BerriAI/litellm/pull/18317)
* @mihidumh made their first contribution in [PR #18361](https://github.com/BerriAI/litellm/pull/18361)
* @Prazeina made their first contribution in [PR #18498](https://github.com/BerriAI/litellm/pull/18498)
* @systec-dk made their first contribution in [PR #18500](https://github.com/BerriAI/litellm/pull/18500)
* @xuan07t2 made their first contribution in [PR #18514](https://github.com/BerriAI/litellm/pull/18514)
* @RensDimmendaal made their first contribution in [PR #18190](https://github.com/BerriAI/litellm/pull/18190)
* @yurekami made their first contribution in [PR #18483](https://github.com/BerriAI/litellm/pull/18483)
* @agertz7 made their first contribution in [PR #18556](https://github.com/BerriAI/litellm/pull/18556)
* @yudelevi made their first contribution in [PR #18550](https://github.com/BerriAI/litellm/pull/18550)
* @smallp made their first contribution in [PR #18536](https://github.com/BerriAI/litellm/pull/18536)
* @kevinpauer made their first contribution in [PR #18569](https://github.com/BerriAI/litellm/pull/18569)
* @cansakiroglu made their first contribution in [PR #18517](https://github.com/BerriAI/litellm/pull/18517)
* @dee-walia20 made their first contribution in [PR #18432](https://github.com/BerriAI/litellm/pull/18432)
* @luxinfeng made their first contribution in [PR #18477](https://github.com/BerriAI/litellm/pull/18477)
* @cantalupo555 made their first contribution in [PR #18476](https://github.com/BerriAI/litellm/pull/18476)
* @andersk made their first contribution in [PR #18473](https://github.com/BerriAI/litellm/pull/18473)
* @majiayu000 made their first contribution in [PR #18467](https://github.com/BerriAI/litellm/pull/18467)
* @amangupta-20 made their first contribution in [PR #18529](https://github.com/BerriAI/litellm/pull/18529)
* @hamzaq453 made their first contribution in [PR #18480](https://github.com/BerriAI/litellm/pull/18480)
* @ktsaou made their first contribution in [PR #18627](https://github.com/BerriAI/litellm/pull/18627)
* @FlibbertyGibbitz made their first contribution in [PR #18624](https://github.com/BerriAI/litellm/pull/18624)
* @drorIvry made their first contribution in [PR #18594](https://github.com/BerriAI/litellm/pull/18594)
* @urainshah made their first contribution in [PR #18524](https://github.com/BerriAI/litellm/pull/18524)
* @mangabits made their first contribution in [PR #18279](https://github.com/BerriAI/litellm/pull/18279)
* @0717376 made their first contribution in [PR #18564](https://github.com/BerriAI/litellm/pull/18564)
* @nmgarza5 made their first contribution in [PR #17330](https://github.com/BerriAI/litellm/pull/17330)
* @wileykestner made their first contribution in [PR #18445](https://github.com/BerriAI/litellm/pull/18445)
* @minijeong-log made their first contribution in [PR #14440](https://github.com/BerriAI/litellm/pull/14440)
* @Isaac4real made their first contribution in [PR #18710](https://github.com/BerriAI/litellm/pull/18710)
* @marukaz made their first contribution in [PR #18711](https://github.com/BerriAI/litellm/pull/18711)
* @rohitravirane made their first contribution in [PR #18712](https://github.com/BerriAI/litellm/pull/18712)
* @lizzzcai made their first contribution in [PR #18714](https://github.com/BerriAI/litellm/pull/18714)
* @hkd987 made their first contribution in [PR #18673](https://github.com/BerriAI/litellm/pull/18673)
* @Mr-Pepe made their first contribution in [PR #18674](https://github.com/BerriAI/litellm/pull/18674)
* @gkarthi-signoz made their first contribution in [PR #18726](https://github.com/BerriAI/litellm/pull/18726)
* @Tianduo16 made their first contribution in [PR #18723](https://github.com/BerriAI/litellm/pull/18723)
* @wilsonjr made their first contribution in [PR #18721](https://github.com/BerriAI/litellm/pull/18721)
* @abliteration-ai made their first contribution in [PR #18678](https://github.com/BerriAI/litellm/pull/18678)
* @danialkhan02 made their first contribution in [PR #18770](https://github.com/BerriAI/litellm/pull/18770)
* @ihower made their first contribution in [PR #18409](https://github.com/BerriAI/litellm/pull/18409)
* @elkkhan made their first contribution in [PR #18391](https://github.com/BerriAI/litellm/pull/18391)
* @runixer made their first contribution in [PR #18435](https://github.com/BerriAI/litellm/pull/18435)
* @choby-shun made their first contribution in [PR #18776](https://github.com/BerriAI/litellm/pull/18776)
* @jutaz made their first contribution in [PR #18853](https://github.com/BerriAI/litellm/pull/18853)
* @sjmatta made their first contribution in [PR #18250](https://github.com/BerriAI/litellm/pull/18250)
* @andres-ortizl made their first contribution in [PR #18856](https://github.com/BerriAI/litellm/pull/18856)
* @gauthiermartin made their first contribution in [PR #18844](https://github.com/BerriAI/litellm/pull/18844)
* @mel2oo made their first contribution in [PR #18845](https://github.com/BerriAI/litellm/pull/18845)
* @DominikHallab made their first contribution in [PR #18846](https://github.com/BerriAI/litellm/pull/18846)
* @ji-chuan-che made their first contribution in [PR #18540](https://github.com/BerriAI/litellm/pull/18540)
* @raghav-stripe made their first contribution in [PR #18858](https://github.com/BerriAI/litellm/pull/18858)
* @akraines made their first contribution in [PR #18629](https://github.com/BerriAI/litellm/pull/18629)
* @otaviofbrito made their first contribution in [PR #18665](https://github.com/BerriAI/litellm/pull/18665)
* @chetanchoudhary-sumo made their first contribution in [PR #18587](https://github.com/BerriAI/litellm/pull/18587)
* @pascalwhoop made their first contribution in [PR #13328](https://github.com/BerriAI/litellm/pull/13328)
* @orgersh92 made their first contribution in [PR #18652](https://github.com/BerriAI/litellm/pull/18652)
* @DevajMody made their first contribution in [PR #18497](https://github.com/BerriAI/litellm/pull/18497)
* @matt-greathouse made their first contribution in [PR #18247](https://github.com/BerriAI/litellm/pull/18247)
* @emerzon made their first contribution in [PR #18290](https://github.com/BerriAI/litellm/pull/18290)
* @Eric84626 made their first contribution in [PR #18281](https://github.com/BerriAI/litellm/pull/18281)
* @LukasdeBoer made their first contribution in [PR #18055](https://github.com/BerriAI/litellm/pull/18055)
* @LingXuanYin made their first contribution in [PR #18513](https://github.com/BerriAI/litellm/pull/18513)
* @krisxia0506 made their first contribution in [PR #18698](https://github.com/BerriAI/litellm/pull/18698)
* @LouisShark made their first contribution in [PR #18414](https://github.com/BerriAI/litellm/pull/18414)
---
## Full Changelog
**[View complete changelog on GitHub](https://github.com/BerriAI/litellm/compare/v1.80.11.rc.1...v1.80.14.rc.1)**

View file

@ -8,6 +8,7 @@ https://platform.openai.com/docs/api-reference/files
import asyncio
import contextvars
import os
import time
from functools import partial
from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast
@ -60,7 +61,7 @@ async def acreate_file(
file: FileTypes,
purpose: Literal["assistants", "batch", "fine-tune"],
expires_after: Optional[FileExpiresAfter] = None,
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai",
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -105,7 +106,7 @@ def create_file(
file: FileTypes,
purpose: Literal["assistants", "batch", "fine-tune"],
expires_after: Optional[FileExpiresAfter] = None,
custom_llm_provider: Optional[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"]] = None,
custom_llm_provider: Optional[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "manus"]] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -274,7 +275,7 @@ def create_file(
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'create_file'. Only ['openai', 'azure', 'vertex_ai'] are supported.".format(
message="LiteLLM doesn't support {} for 'create_file'. Only ['openai', 'azure', 'vertex_ai', 'manus'] are supported.".format(
custom_llm_provider
),
model="n/a",
@ -293,7 +294,7 @@ def create_file(
@client
async def afile_retrieve(
file_id: str,
custom_llm_provider: Literal["openai", "azure", "hosted_vllm"] = "openai",
custom_llm_provider: Literal["openai", "azure", "hosted_vllm", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -334,7 +335,7 @@ async def afile_retrieve(
@client
def file_retrieve(
file_id: str,
custom_llm_provider: Literal["openai", "azure", "hosted_vllm"] = "openai",
custom_llm_provider: Literal["openai", "azure", "hosted_vllm", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -428,18 +429,60 @@ def file_retrieve(
file_id=file_id,
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'file_retrieve'. Only 'openai' and 'azure' are supported.".format(
custom_llm_provider
),
model="n/a",
llm_provider=custom_llm_provider,
response=httpx.Response(
status_code=400,
content="Unsupported provider",
request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
),
# Try using provider config pattern (for Manus, Bedrock, etc.)
provider_config = ProviderConfigManager.get_provider_files_config(
model="",
provider=LlmProviders(custom_llm_provider),
)
if provider_config is not None:
litellm_params_dict = get_litellm_params(**kwargs)
litellm_params_dict["api_key"] = optional_params.api_key
litellm_params_dict["api_base"] = optional_params.api_base
logging_obj = kwargs.get("litellm_logging_obj")
if logging_obj is None:
from litellm.litellm_core_utils.litellm_logging import (
Logging as LiteLLMLoggingObj,
)
logging_obj = LiteLLMLoggingObj(
model="",
messages=[],
stream=False,
call_type="afile_retrieve" if _is_async else "file_retrieve",
start_time=time.time(),
litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())),
function_id=str(kwargs.get("id") or ""),
)
client = kwargs.get("client")
response = base_llm_http_handler.retrieve_file(
file_id=file_id,
provider_config=provider_config,
litellm_params=litellm_params_dict,
headers=extra_headers or {},
logging_obj=logging_obj,
_is_async=_is_async,
client=(
client
if client is not None
and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
else None
),
timeout=timeout,
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'file_retrieve'. Only 'openai', 'azure', and 'manus' are supported.".format(
custom_llm_provider
),
model="n/a",
llm_provider=custom_llm_provider,
response=httpx.Response(
status_code=400,
content="Unsupported provider",
request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
),
)
return cast(FileObject, response)
except Exception as e:
@ -450,7 +493,7 @@ def file_retrieve(
@client
async def afile_delete(
file_id: str,
custom_llm_provider: Literal["openai", "azure"] = "openai",
custom_llm_provider: Literal["openai", "azure", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -494,7 +537,7 @@ async def afile_delete(
def file_delete(
file_id: str,
model: Optional[str] = None,
custom_llm_provider: Union[Literal["openai", "azure"], str] = "openai",
custom_llm_provider: Union[Literal["openai", "azure", "manus"], str] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -596,18 +639,58 @@ def file_delete(
litellm_params=litellm_params_dict,
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'delete_batch'. Only 'openai' is supported.".format(
custom_llm_provider
),
model="n/a",
llm_provider=custom_llm_provider,
response=httpx.Response(
status_code=400,
content="Unsupported provider",
request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
),
# Try using provider config pattern (for Manus, Bedrock, etc.)
provider_config = ProviderConfigManager.get_provider_files_config(
model="",
provider=LlmProviders(custom_llm_provider),
)
if provider_config is not None:
litellm_params_dict["api_key"] = optional_params.api_key
litellm_params_dict["api_base"] = optional_params.api_base
logging_obj = kwargs.get("litellm_logging_obj")
if logging_obj is None:
from litellm.litellm_core_utils.litellm_logging import (
Logging as LiteLLMLoggingObj,
)
logging_obj = LiteLLMLoggingObj(
model="",
messages=[],
stream=False,
call_type="afile_delete" if _is_async else "file_delete",
start_time=time.time(),
litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())),
function_id=str(kwargs.get("id") or ""),
)
response = base_llm_http_handler.delete_file(
file_id=file_id,
provider_config=provider_config,
litellm_params=litellm_params_dict,
headers=extra_headers or {},
logging_obj=logging_obj,
_is_async=_is_async,
client=(
client
if client is not None
and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
else None
),
timeout=timeout,
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'file_delete'. Only 'openai', 'azure', and 'manus' are supported.".format(
custom_llm_provider
),
model="n/a",
llm_provider=custom_llm_provider,
response=httpx.Response(
status_code=400,
content="Unsupported provider",
request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore
),
)
return cast(FileDeleted, response)
except Exception as e:
raise e
@ -616,7 +699,7 @@ def file_delete(
# List files
@client
async def afile_list(
custom_llm_provider: Literal["openai", "azure"] = "openai",
custom_llm_provider: Literal["openai", "azure", "manus"] = "openai",
purpose: Optional[str] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
@ -657,7 +740,7 @@ async def afile_list(
@client
def file_list(
custom_llm_provider: Literal["openai", "azure"] = "openai",
custom_llm_provider: Literal["openai", "azure", "manus"] = "openai",
purpose: Optional[str] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
@ -687,7 +770,50 @@ def file_list(
timeout = 600.0
_is_async = kwargs.pop("is_async", False) is True
if custom_llm_provider in OPENAI_COMPATIBLE_BATCH_AND_FILES_PROVIDERS:
# Check if provider has a custom files config (e.g., Manus, Bedrock, Vertex AI)
provider_config = ProviderConfigManager.get_provider_files_config(
model="",
provider=LlmProviders(custom_llm_provider),
)
if provider_config is not None:
litellm_params_dict = get_litellm_params(**kwargs)
litellm_params_dict["api_key"] = optional_params.api_key
litellm_params_dict["api_base"] = optional_params.api_base
logging_obj = kwargs.get("litellm_logging_obj")
if logging_obj is None:
from litellm.litellm_core_utils.litellm_logging import (
Logging as LiteLLMLoggingObj,
)
logging_obj = LiteLLMLoggingObj(
model="",
messages=[],
stream=False,
call_type="afile_list" if _is_async else "file_list",
start_time=time.time(),
litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())),
function_id=str(kwargs.get("id", "")),
)
client = kwargs.get("client")
response = base_llm_http_handler.list_files(
purpose=purpose,
provider_config=provider_config,
litellm_params=litellm_params_dict,
headers=extra_headers or {},
logging_obj=logging_obj,
_is_async=_is_async,
client=(
client
if client is not None
and isinstance(client, (HTTPHandler, AsyncHTTPHandler))
else None
),
timeout=timeout,
)
return response
elif custom_llm_provider in OPENAI_COMPATIBLE_BATCH_AND_FILES_PROVIDERS:
# for deepinfra/perplexity/anyscale/groq we check in get_llm_provider and pass in the api base from there
api_base = (
optional_params.api_base
@ -752,7 +878,7 @@ def file_list(
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'file_list'. Only 'openai' and 'azure' are supported.".format(
message="LiteLLM doesn't support {} for 'file_list'. Only 'openai', 'azure', and 'manus' are supported.".format(
custom_llm_provider
),
model="n/a",
@ -771,7 +897,7 @@ def file_list(
@client
async def afile_content(
file_id: str,
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic"] = "openai",
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic", "manus"] = "openai",
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
**kwargs,
@ -816,7 +942,7 @@ def file_content(
file_id: str,
model: Optional[str] = None,
custom_llm_provider: Optional[
Union[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic"], str]
Union[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic", "manus"], str]
] = None,
extra_headers: Optional[Dict[str, str]] = None,
extra_body: Optional[Dict[str, str]] = None,
@ -977,7 +1103,7 @@ def file_content(
)
else:
raise litellm.exceptions.BadRequestError(
message="LiteLLM doesn't support {} for 'custom_llm_provider'. Supported providers are 'openai', 'azure', 'vertex_ai', 'bedrock'.".format(
message="LiteLLM doesn't support {} for 'file_content'. Supported providers are 'openai', 'azure', 'vertex_ai', 'bedrock', 'manus'.".format(
custom_llm_provider
),
model="n/a",

View file

@ -875,16 +875,7 @@ class PrometheusLogger(CustomLogger):
# Include top-level metadata fields (excluding nested dictionaries)
# This allows accessing fields like requester_ip_address from top-level metadata
top_level_metadata = standard_logging_payload.get("metadata", {})
top_level_fields: Dict[str, Any] = {}
if isinstance(top_level_metadata, dict):
top_level_fields = {
k: v
for k, v in top_level_metadata.items()
if not isinstance(v, dict) # Exclude nested dicts to avoid conflicts
}
combined_metadata: Dict[str, Any] = {
**top_level_fields, # Include top-level fields first
**(_requester_metadata if _requester_metadata else {}),
**(user_api_key_auth_metadata if user_api_key_auth_metadata else {}),
}

View file

@ -2,11 +2,14 @@ from abc import ABC, abstractmethod
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union
import httpx
from openai.types.file_deleted import FileDeleted
from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.files import TwoStepFileUploadConfig
from litellm.types.llms.openai import (
AllMessageValues,
CreateFileRequest,
FileContentRequest,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
OpenAIFilesPurpose,
@ -75,7 +78,15 @@ class BaseFilesConfig(BaseConfig):
create_file_data: CreateFileRequest,
optional_params: dict,
litellm_params: dict,
) -> Union[dict, str, bytes]:
) -> Union[dict, str, bytes, "TwoStepFileUploadConfig"]:
"""
Transform OpenAI-style file creation request into provider-specific format.
Returns:
- dict: For pre-signed single-step uploads (e.g., Bedrock S3)
- str/bytes: For traditional file uploads
- TwoStepFileUploadConfig: For two-step upload process (e.g., Manus, GCS)
"""
pass
@abstractmethod
@ -88,6 +99,86 @@ class BaseFilesConfig(BaseConfig):
) -> OpenAIFileObject:
pass
@abstractmethod
def transform_retrieve_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Transform file retrieve request into provider-specific format."""
pass
@abstractmethod
def transform_retrieve_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> OpenAIFileObject:
"""Transform file retrieve response into OpenAI format."""
pass
@abstractmethod
def transform_delete_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Transform file delete request into provider-specific format."""
pass
@abstractmethod
def transform_delete_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> "FileDeleted":
"""Transform file delete response into OpenAI format."""
pass
@abstractmethod
def transform_list_files_request(
self,
purpose: Optional[str],
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Transform file list request into provider-specific format."""
pass
@abstractmethod
def transform_list_files_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> List[OpenAIFileObject]:
"""Transform file list response into OpenAI format."""
pass
@abstractmethod
def transform_file_content_request(
self,
file_content_request: "FileContentRequest",
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Transform file content request into provider-specific format."""
pass
@abstractmethod
def transform_file_content_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> "HttpxBinaryResponseContent":
"""Transform file content response into OpenAI format."""
pass
def transform_request(
self,
model: str,

View file

@ -74,41 +74,6 @@ class BaseAWSLLM:
"aws_external_id",
]
def _get_ssl_verify(self):
"""
Get SSL verification setting for boto3 clients.
This ensures that custom CA certificates are properly used for all AWS API calls,
including STS and Bedrock services.
Returns:
Union[bool, str]: SSL verification setting - False to disable, True to enable,
or a string path to a CA bundle file
"""
import litellm
from litellm.secret_managers.main import str_to_bool
# Check environment variable first (highest priority)
ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify)
# Convert string "False"/"True" to boolean
if isinstance(ssl_verify, str):
# Check if it's a file path
if os.path.exists(ssl_verify):
return ssl_verify
# Otherwise try to convert to boolean
ssl_verify_bool = str_to_bool(ssl_verify)
if ssl_verify_bool is not None:
ssl_verify = ssl_verify_bool
# Check SSL_CERT_FILE environment variable for custom CA bundle
if ssl_verify is True or ssl_verify == "True":
ssl_cert_file = os.getenv("SSL_CERT_FILE")
if ssl_cert_file and os.path.exists(ssl_cert_file):
return ssl_cert_file
return ssl_verify
def get_cache_key(self, credential_args: Dict[str, Optional[str]]) -> str:
"""
Generate a unique cache key based on the credential arguments.
@ -604,7 +569,6 @@ class BaseAWSLLM:
"sts",
region_name=aws_region_name,
endpoint_url=sts_endpoint,
verify=self._get_ssl_verify(),
)
# https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html
@ -661,7 +625,7 @@ class BaseAWSLLM:
# Create an STS client without credentials
with tracer.trace("boto3.client(sts) for manual IRSA"):
sts_client = boto3.client("sts", region_name=region, verify=self._get_ssl_verify())
sts_client = boto3.client("sts", region_name=region)
# Manually assume the IRSA role with the session name
verbose_logger.debug(
@ -684,7 +648,6 @@ class BaseAWSLLM:
aws_access_key_id=irsa_creds["AccessKeyId"],
aws_secret_access_key=irsa_creds["SecretAccessKey"],
aws_session_token=irsa_creds["SessionToken"],
verify=self._get_ssl_verify(),
)
# Get current caller identity for debugging
@ -723,7 +686,7 @@ class BaseAWSLLM:
verbose_logger.debug("Same account role assumption, using automatic IRSA")
with tracer.trace("boto3.client(sts) with automatic IRSA"):
sts_client = boto3.client("sts", region_name=region, verify=self._get_ssl_verify())
sts_client = boto3.client("sts", region_name=region)
# Get current caller identity for debugging
try:
@ -846,7 +809,7 @@ class BaseAWSLLM:
# This allows the web identity token to work automatically
if aws_access_key_id is None and aws_secret_access_key is None:
with tracer.trace("boto3.client(sts)"):
sts_client = boto3.client("sts", verify=self._get_ssl_verify())
sts_client = boto3.client("sts")
else:
with tracer.trace("boto3.client(sts)"):
sts_client = boto3.client(
@ -854,7 +817,6 @@ class BaseAWSLLM:
aws_access_key_id=aws_access_key_id,
aws_secret_access_key=aws_secret_access_key,
aws_session_token=aws_session_token,
verify=self._get_ssl_verify(),
)
assume_role_params = {

View file

@ -260,7 +260,7 @@ def init_bedrock_client(
status_code=401,
)
sts_client = boto3.client("sts", verify=ssl_verify)
sts_client = boto3.client("sts")
# https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html
# https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts/client/assume_role_with_web_identity.html

View file

@ -142,7 +142,6 @@ class BedrockFilesHandler(BaseAWSLLM):
aws_secret_access_key=credentials.secret_key,
aws_session_token=credentials.token,
region_name=aws_region_name,
verify=self._get_ssl_verify(),
)
# Download file from S3

View file

@ -1,12 +1,14 @@
import json
import os
import time
from litellm._uuid import uuid
from typing import Any, Dict, List, Optional, Tuple, Union
import httpx
from httpx import Headers, Response
from openai.types.file_deleted import FileDeleted
from litellm._logging import verbose_logger
from litellm._uuid import uuid
from litellm.files.utils import FilesAPIUtils
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@ -18,6 +20,7 @@ from litellm.types.llms.openai import (
AllMessageValues,
CreateFileRequest,
FileTypes,
HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
PathLike,
@ -539,6 +542,70 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
status_code=status_code, message=error_message, headers=headers
)
def transform_retrieve_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("BedrockFilesConfig does not support file retrieval")
def transform_retrieve_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> OpenAIFileObject:
raise NotImplementedError("BedrockFilesConfig does not support file retrieval")
def transform_delete_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("BedrockFilesConfig does not support file deletion")
def transform_delete_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> FileDeleted:
raise NotImplementedError("BedrockFilesConfig does not support file deletion")
def transform_list_files_request(
self,
purpose: Optional[str],
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("BedrockFilesConfig does not support file listing")
def transform_list_files_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> List[OpenAIFileObject]:
raise NotImplementedError("BedrockFilesConfig does not support file listing")
def transform_file_content_request(
self,
file_content_request,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("BedrockFilesConfig does not support file content retrieval")
def transform_file_content_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> HttpxBinaryResponseContent:
raise NotImplementedError("BedrockFilesConfig does not support file content retrieval")
class BedrockJsonlFilesTransformation:
"""

View file

@ -14,6 +14,7 @@ from typing import (
)
import httpx # type: ignore
from openai.types.file_deleted import FileDeleted
import litellm
import litellm.litellm_core_utils
@ -71,6 +72,7 @@ from litellm.types.containers.main import (
ContainerObject,
DeleteContainerResult,
)
from litellm.types.files import TwoStepFileUploadConfig
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
@ -82,6 +84,7 @@ from litellm.types.llms.anthropic_skills import (
from litellm.types.llms.openai import (
CreateBatchRequest,
CreateFileRequest,
FileContentRequest,
HttpxBinaryResponseContent,
OpenAIFileObject,
ResponseInputParam,
@ -2782,6 +2785,38 @@ class BaseLLMHTTPHandler:
logging_obj=logging_obj,
)
def _extract_upload_url_from_response(
self,
response: httpx.Response,
upload_url_location: str,
upload_url_key: str = "upload_url",
) -> tuple[Optional[str], Optional[dict]]:
"""
Extract upload URL from initial file creation response.
Args:
response: HTTP response from initial file creation request
upload_url_location: Where to find URL ('headers' or 'body')
upload_url_key: Key name for URL in response body (default: 'upload_url')
Returns:
Tuple of (upload_url, response_data)
- upload_url: The extracted upload URL, or None if not found
- response_data: Parsed response body (for 'body' location), or None
"""
if upload_url_location == "headers":
# Google Cloud Storage style - URL in X-Goog-Upload-URL header
upload_url = response.headers.get("X-Goog-Upload-URL")
return upload_url, None
else:
# Response body style (e.g., Manus, S3 presigned URLs)
try:
response_data = response.json()
upload_url = response_data.get(upload_url_key)
return upload_url, response_data if upload_url else None
except Exception:
return None, None
def create_file(
self,
create_file_data: CreateFileRequest,
@ -2844,14 +2879,58 @@ class BaseLLMHTTPHandler:
else:
sync_httpx_client = client
if isinstance(transformed_request, dict) and "method" in transformed_request:
if isinstance(transformed_request, dict) and "initial_request" in transformed_request:
# Handle two-step uploads (TwoStepFileUploadConfig)
# Used by providers like Manus, Google Cloud Storage
try:
# Step 1: Initial request to get upload URL
initial_response = sync_httpx_client.post(
url=api_base,
headers={
**headers,
**transformed_request["initial_request"]["headers"],
},
data=json.dumps(transformed_request["initial_request"]["data"]),
timeout=timeout,
)
# Extract upload URL from response
upload_url, initial_response_data = self._extract_upload_url_from_response(
response=initial_response,
upload_url_location=transformed_request.get("upload_url_location", "headers"),
upload_url_key=transformed_request.get("upload_url_key", "upload_url"),
)
if not upload_url:
raise ValueError("Failed to get upload URL from initial request")
# Step 2: Upload the actual file
upload_method = transformed_request["upload_request"].get("method", "POST").lower()
upload_response = getattr(sync_httpx_client, upload_method)(
url=upload_url,
headers=transformed_request["upload_request"]["headers"],
data=transformed_request["upload_request"]["data"],
timeout=timeout,
)
# Store initial response for transformation
if initial_response_data:
litellm_params["initial_file_response"] = initial_response_data
except Exception as e:
raise self._handle_error(
e=e,
provider_config=provider_config,
)
elif isinstance(transformed_request, dict) and "method" in transformed_request and "initial_request" not in transformed_request:
# Handle pre-signed requests (e.g., from Bedrock S3 uploads)
# Type narrowing: this is a plain dict, not TwoStepFileUploadConfig
presigned_request = cast(Dict[str, Any], transformed_request)
upload_response = getattr(
sync_httpx_client, transformed_request["method"].lower()
sync_httpx_client, presigned_request["method"].lower()
)(
url=transformed_request["url"],
headers=transformed_request["headers"],
data=transformed_request["data"],
url=presigned_request["url"],
headers=presigned_request["headers"],
data=presigned_request["data"],
timeout=timeout,
)
elif isinstance(transformed_request, str) or isinstance(
@ -2879,36 +2958,7 @@ class BaseLLMHTTPHandler:
timeout=timeout,
)
else:
try:
# Step 1: Initial request to get upload URL
initial_response = sync_httpx_client.post(
url=api_base,
headers={
**headers,
**transformed_request["initial_request"]["headers"],
},
data=json.dumps(transformed_request["initial_request"]["data"]),
timeout=timeout,
)
# Extract upload URL from response headers
upload_url = initial_response.headers.get("X-Goog-Upload-URL")
if not upload_url:
raise ValueError("Failed to get upload URL from initial request")
# Step 2: Upload the actual file
upload_response = sync_httpx_client.post(
url=upload_url,
headers=transformed_request["upload_request"]["headers"],
data=transformed_request["upload_request"]["data"],
timeout=timeout,
)
except Exception as e:
raise self._handle_error(
e=e,
provider_config=provider_config,
)
raise ValueError(f"Unsupported transformed_request type: {type(transformed_request)}")
# Store the upload URL in litellm_params for the transformation method
litellm_params_with_url = dict(litellm_params)
@ -2923,7 +2973,7 @@ class BaseLLMHTTPHandler:
async def async_create_file(
self,
transformed_request: Union[bytes, str, dict],
transformed_request: Union[bytes, str, dict, "TwoStepFileUploadConfig"],
litellm_params: dict,
provider_config: BaseFilesConfig,
headers: dict,
@ -2955,14 +3005,59 @@ class BaseLLMHTTPHandler:
},
)
if isinstance(transformed_request, dict) and "method" in transformed_request:
if isinstance(transformed_request, dict) and "initial_request" in transformed_request:
# Handle two-step uploads (TwoStepFileUploadConfig)
# Used by providers like Manus, Google Cloud Storage
try:
# Step 1: Initial request to get upload URL
initial_response = await async_httpx_client.post(
url=api_base,
headers={
**headers,
**transformed_request["initial_request"]["headers"],
},
data=json.dumps(transformed_request["initial_request"]["data"]),
timeout=timeout,
)
# Extract upload URL from response
upload_url, initial_response_data = self._extract_upload_url_from_response(
response=initial_response,
upload_url_location=transformed_request.get("upload_url_location", "headers"),
upload_url_key=transformed_request.get("upload_url_key", "upload_url"),
)
if not upload_url:
raise ValueError("Failed to get upload URL from initial request")
# Step 2: Upload the actual file
upload_method = transformed_request["upload_request"].get("method", "POST").lower()
upload_response = await getattr(async_httpx_client, upload_method)(
url=upload_url,
headers=transformed_request["upload_request"]["headers"],
data=transformed_request["upload_request"]["data"],
timeout=timeout,
)
# Store initial response for transformation
if initial_response_data:
litellm_params["initial_file_response"] = initial_response_data
except Exception as e:
verbose_logger.exception(f"Error creating file: {e}")
raise self._handle_error(
e=e,
provider_config=provider_config,
)
elif isinstance(transformed_request, dict) and "method" in transformed_request and "initial_request" not in transformed_request:
# Handle pre-signed requests (e.g., from Bedrock S3 uploads)
# Type narrowing: this is a plain dict, not TwoStepFileUploadConfig
presigned_request = cast(Dict[str, Any], transformed_request)
upload_response = await getattr(
async_httpx_client, transformed_request["method"].lower()
async_httpx_client, presigned_request["method"].lower()
)(
url=transformed_request["url"],
headers=transformed_request["headers"],
data=transformed_request["data"],
url=presigned_request["url"],
headers=presigned_request["headers"],
data=presigned_request["data"],
timeout=timeout,
)
elif isinstance(transformed_request, str) or isinstance(
@ -2990,37 +3085,7 @@ class BaseLLMHTTPHandler:
timeout=timeout,
)
else:
try:
# Step 1: Initial request to get upload URL
initial_response = await async_httpx_client.post(
url=api_base,
headers={
**headers,
**transformed_request["initial_request"]["headers"],
},
data=json.dumps(transformed_request["initial_request"]["data"]),
timeout=timeout,
)
# Extract upload URL from response headers
upload_url = initial_response.headers.get("X-Goog-Upload-URL")
if not upload_url:
raise ValueError("Failed to get upload URL from initial request")
# Step 2: Upload the actual file
upload_response = await async_httpx_client.post(
url=upload_url,
headers=transformed_request["upload_request"]["headers"],
data=transformed_request["upload_request"]["data"],
timeout=timeout,
)
except Exception as e:
verbose_logger.exception(f"Error creating file: {e}")
raise self._handle_error(
e=e,
provider_config=provider_config,
)
raise ValueError(f"Unsupported transformed_request type: {type(transformed_request)}")
return provider_config.transform_create_file_response(
model=None,
@ -3734,29 +3799,525 @@ class BaseLLMHTTPHandler:
logging_obj=logging_obj,
)
def list_files(self):
def retrieve_file(
self,
file_id: str,
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
_is_async: bool = False,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Union[OpenAIFileObject, Coroutine[Any, Any, OpenAIFileObject]]:
"""
Lists all files
Retrieve file metadata by ID
"""
pass
if _is_async:
return self.async_retrieve_file(
file_id=file_id,
provider_config=provider_config,
litellm_params=litellm_params,
headers=headers,
logging_obj=logging_obj,
client=client,
timeout=timeout,
)
def delete_file(self):
"""
Deletes a file
"""
pass
if client is None or not isinstance(client, HTTPHandler):
sync_httpx_client = _get_httpx_client()
else:
sync_httpx_client = client
def retrieve_file(self):
"""
Returns the metadata of the file
"""
pass
# Get URL and params from provider config
url, params = provider_config.transform_retrieve_file_request(
file_id=file_id,
optional_params={},
litellm_params=litellm_params,
)
def retrieve_file_content(self):
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"file_id": file_id,
},
)
try:
response = sync_httpx_client.get(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_retrieve_file_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
async def async_retrieve_file(
self,
file_id: str,
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> OpenAIFileObject:
"""
Returns the content of the file
Async retrieve file metadata by ID
"""
pass
if client is None or not isinstance(client, AsyncHTTPHandler):
async_httpx_client = get_async_httpx_client(
llm_provider=provider_config.custom_llm_provider
)
else:
async_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_retrieve_file_request(
file_id=file_id,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"file_id": file_id,
},
)
try:
response = await async_httpx_client.get(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_retrieve_file_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
def delete_file(
self,
file_id: str,
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
_is_async: bool = False,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Union["FileDeleted", Coroutine[Any, Any, "FileDeleted"]]:
"""
Delete a file by ID
"""
if _is_async:
return self.async_delete_file(
file_id=file_id,
provider_config=provider_config,
litellm_params=litellm_params,
headers=headers,
logging_obj=logging_obj,
client=client,
timeout=timeout,
)
if client is None or not isinstance(client, HTTPHandler):
sync_httpx_client = _get_httpx_client()
else:
sync_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_delete_file_request(
file_id=file_id,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"file_id": file_id,
},
)
try:
response = sync_httpx_client.delete(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_delete_file_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
async def async_delete_file(
self,
file_id: str,
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> "FileDeleted":
"""
Async delete a file by ID
"""
if client is None or not isinstance(client, AsyncHTTPHandler):
async_httpx_client = get_async_httpx_client(
llm_provider=provider_config.custom_llm_provider
)
else:
async_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_delete_file_request(
file_id=file_id,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"file_id": file_id,
},
)
try:
response = await async_httpx_client.delete(
url=url, headers=headers, params=params, timeout=timeout
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_delete_file_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
def list_files(
self,
purpose: Optional[str],
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
_is_async: bool = False,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Union[List[OpenAIFileObject], Coroutine[Any, Any, List[OpenAIFileObject]]]:
"""
List all files
"""
if _is_async:
return self.async_list_files(
purpose=purpose,
provider_config=provider_config,
litellm_params=litellm_params,
headers=headers,
logging_obj=logging_obj,
client=client,
timeout=timeout,
)
if client is None or not isinstance(client, HTTPHandler):
sync_httpx_client = _get_httpx_client()
else:
sync_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_list_files_request(
purpose=purpose,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"purpose": purpose,
},
)
try:
response = sync_httpx_client.get(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_list_files_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
async def async_list_files(
self,
purpose: Optional[str],
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> List[OpenAIFileObject]:
"""
Async list all files
"""
if client is None or not isinstance(client, AsyncHTTPHandler):
async_httpx_client = get_async_httpx_client(
llm_provider=provider_config.custom_llm_provider
)
else:
async_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_list_files_request(
purpose=purpose,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"purpose": purpose,
},
)
try:
response = await async_httpx_client.get(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_list_files_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
def retrieve_file_content(
self,
file_content_request: "FileContentRequest",
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
_is_async: bool = False,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> Union["HttpxBinaryResponseContent", Coroutine[Any, Any, "HttpxBinaryResponseContent"]]:
"""
Retrieve file content by ID
"""
if _is_async:
return self.async_retrieve_file_content(
file_content_request=file_content_request,
provider_config=provider_config,
litellm_params=litellm_params,
headers=headers,
logging_obj=logging_obj,
client=client,
timeout=timeout,
)
if client is None or not isinstance(client, HTTPHandler):
sync_httpx_client = _get_httpx_client()
else:
sync_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_file_content_request(
file_content_request=file_content_request,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"file_id": file_content_request.get("file_id"),
},
)
try:
response = sync_httpx_client.get(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_file_content_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
async def async_retrieve_file_content(
self,
file_content_request: "FileContentRequest",
provider_config: BaseFilesConfig,
litellm_params: dict,
headers: dict,
logging_obj: LiteLLMLoggingObj,
client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None,
timeout: Optional[Union[float, httpx.Timeout]] = None,
) -> "HttpxBinaryResponseContent":
"""
Async retrieve file content by ID
"""
if client is None or not isinstance(client, AsyncHTTPHandler):
async_httpx_client = get_async_httpx_client(
llm_provider=provider_config.custom_llm_provider
)
else:
async_httpx_client = client
# Get URL and params from provider config
url, params = provider_config.transform_file_content_request(
file_content_request=file_content_request,
optional_params={},
litellm_params=litellm_params,
)
# Validate environment and get headers
headers = provider_config.validate_environment(
api_key=litellm_params.get("api_key"),
headers=headers,
model="",
messages=[],
optional_params={},
litellm_params=litellm_params,
)
logging_obj.pre_call(
input="",
api_key="",
additional_args={
"api_base": url,
"headers": headers,
"file_id": file_content_request.get("file_id"),
},
)
try:
response = await async_httpx_client.get(
url=url, headers=headers, params=params
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
return provider_config.transform_file_content_response(
raw_response=response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
def _prepare_fake_stream_request(
self,

View file

@ -150,15 +150,6 @@ def get_api_key_from_env() -> Optional[str]:
return get_secret_str("GOOGLE_API_KEY") or get_secret_str("GEMINI_API_KEY")
def get_vertex_api_key_from_env() -> Optional[str]:
"""
Get API key from environment for Vertex AI.
Checks VERTEXAI_API_KEY and VERTEX_API_KEY environment variables.
This allows using Vertex AI with API keys instead of service account credentials.
"""
return get_secret_str("VERTEXAI_API_KEY") or get_secret_str("VERTEX_API_KEY")
class GoogleAIStudioTokenCounter(BaseTokenCounter):
"""Token counter implementation for Google AI Studio provider."""
def should_use_token_counting_api(

View file

@ -7,6 +7,7 @@ import time
from typing import List, Optional
import httpx
from openai.types.file_deleted import FileDeleted
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
@ -17,6 +18,7 @@ from litellm.llms.base_llm.files.transformation import (
from litellm.types.llms.gemini import GeminiCreateFilesResponseObject
from litellm.types.llms.openai import (
CreateFileRequest,
HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
)
@ -171,3 +173,67 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig):
except Exception as e:
verbose_logger.exception(f"Error parsing file upload response: {str(e)}")
raise ValueError(f"Error parsing file upload response: {str(e)}")
def transform_retrieve_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file retrieval")
def transform_retrieve_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> OpenAIFileObject:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file retrieval")
def transform_delete_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file deletion")
def transform_delete_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> FileDeleted:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file deletion")
def transform_list_files_request(
self,
purpose: Optional[str],
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file listing")
def transform_list_files_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> List[OpenAIFileObject]:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file listing")
def transform_file_content_request(
self,
file_content_request,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file content retrieval")
def transform_file_content_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> HttpxBinaryResponseContent:
raise NotImplementedError("GoogleAIStudioFilesHandler does not support file content retrieval")

View file

@ -0,0 +1,2 @@
# Manus Files API implementation

View file

@ -0,0 +1,439 @@
"""
Manus Files API implementation.
Manus has an OpenAI-compatible Files API with some differences:
- Uses API_KEY header instead of Authorization: Bearer
- File upload is a two-step process:
1. Create file record to get upload URL
2. Upload file content to the upload URL
Reference: https://open.manus.im/docs/openai-compatibility#file-management
"""
import time
from typing import Any, Dict, List, Optional, Union
import httpx
from openai.types.file_deleted import FileDeleted
import litellm
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.files.transformation import (
BaseFilesConfig,
LiteLLMLoggingObj,
)
from litellm.llms.openai.common_utils import OpenAIError
from litellm.secret_managers.main import get_secret_str
from litellm.types.files import TwoStepFileUploadConfig, TwoStepFileUploadRequest
from litellm.types.llms.openai import (
CreateFileRequest,
FileContentRequest,
HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
)
from litellm.types.utils import LlmProviders
MANUS_API_BASE = "https://api.manus.im"
class ManusFilesConfig(BaseFilesConfig):
"""
Configuration for Manus Files API.
Manus uses:
- API_KEY header for authentication (not Authorization: Bearer)
- Two-step file upload process
- Content-Type: application/json for all requests
Reference: https://open.manus.im/docs/openai-compatibility#file-management
"""
def __init__(self):
pass
@property
def custom_llm_provider(self) -> LlmProviders:
return LlmProviders.MANUS
def validate_environment(
self,
headers: dict,
model: str,
messages: list,
optional_params: dict,
litellm_params: dict,
api_key: Optional[str] = None,
api_base: Optional[str] = None,
) -> dict:
"""
Validate environment and set up headers for Manus API.
Manus uses API_KEY header instead of Authorization: Bearer.
For file uploads, don't set Content-Type - httpx will set it for multipart.
"""
api_key = (
api_key
or litellm.api_key
or get_secret_str("MANUS_API_KEY")
)
if not api_key:
raise ValueError(
"Manus API key is required. Set MANUS_API_KEY environment variable or pass api_key parameter."
)
# Manus uses API_KEY header, not Authorization: Bearer
# Manus requires Content-Type: application/json for all requests (even GET)
headers.update(
{
"API_KEY": api_key,
"Content-Type": "application/json",
}
)
return headers
def get_supported_openai_params(
self, model: str
) -> List[OpenAICreateFileRequestOptionalParams]:
"""
Return supported OpenAI file creation parameters for Manus.
Manus supports the standard 'purpose' parameter.
"""
return ["purpose"]
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
"""
Map OpenAI parameters to Manus-specific parameters.
Manus is OpenAI-compatible, so no special mapping needed.
"""
return optional_params
def get_complete_url(
self,
api_base: Optional[str],
api_key: Optional[str],
model: str,
optional_params: dict,
litellm_params: dict,
stream: Optional[bool] = None,
) -> str:
"""
Get the complete URL for Manus Files API endpoint.
Returns:
str: The full URL for the Manus /v1/files endpoint
"""
api_base = (
api_base
or litellm.api_base
or get_secret_str("MANUS_API_BASE")
or MANUS_API_BASE
)
# Remove trailing slashes
api_base = api_base.rstrip("/")
# Manus API uses /v1/files endpoint
if api_base.endswith("/v1"):
return f"{api_base}/files"
return f"{api_base}/v1/files"
def get_error_class(
self,
error_message: str,
status_code: int,
headers: Union[dict, httpx.Headers],
) -> BaseLLMException:
"""
Return the appropriate error class for Manus API errors.
Uses OpenAIError since Manus is OpenAI-compatible.
"""
return OpenAIError(
status_code=status_code,
message=error_message,
headers=headers,
)
def transform_create_file_request(
self,
model: str,
create_file_data: CreateFileRequest,
optional_params: dict,
litellm_params: dict,
) -> TwoStepFileUploadConfig:
"""
Transform OpenAI-style file creation request into Manus's two-step format.
Manus API spec (https://open.manus.im/docs/openai-compatibility#file-management):
1. POST /v1/files with JSON {"filename": "..."} → returns {"id": "...", "upload_url": "..."}
2. PUT to upload_url with raw file content
"""
# Extract file data
file_data = create_file_data.get("file")
if file_data is None:
raise ValueError("File data is required")
extracted_data = extract_file_data(file_data)
filename = extracted_data["filename"] or f"file_{int(time.time())}"
content = extracted_data["content"]
# Get API base URL
api_base = self.get_complete_url(
api_base=litellm_params.get("api_base"),
api_key=litellm_params.get("api_key"),
model=model,
optional_params=optional_params,
litellm_params=litellm_params,
)
# Get API key
api_key = (
litellm_params.get("api_key")
or litellm.api_key
or get_secret_str("MANUS_API_KEY")
)
if not api_key:
raise ValueError(
"Manus API key is required. Set MANUS_API_KEY environment variable or pass api_key parameter."
)
# Build typed two-step upload config
return TwoStepFileUploadConfig(
initial_request=TwoStepFileUploadRequest(
method="POST",
url=api_base,
headers={
"API_KEY": api_key,
"Content-Type": "application/json",
},
data={"filename": filename},
),
upload_request=TwoStepFileUploadRequest(
method="PUT",
url="", # Will be populated from initial_request response
headers={},
data=content,
),
upload_url_location="body",
upload_url_key="upload_url",
)
def transform_create_file_response(
self,
model: Optional[str],
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> OpenAIFileObject:
"""
Transform Manus's file upload response into OpenAI-style FileObject.
For two-step uploads, the handler stores the initial response in litellm_params.
We need to return the file object from the initial POST, not the final PUT.
Manus initial response format:
{
"id": "file-abc123xyz",
"object": "file",
"filename": "document.pdf",
"status": "pending",
"upload_url": "https://...",
"upload_expires_at": "...",
"created_at": "..."
}
"""
try:
# For two-step uploads, get the initial response from litellm_params
initial_response_data = litellm_params.get("initial_file_response")
if initial_response_data:
response_json = initial_response_data
else:
# Log raw response for debugging
verbose_logger.debug(f"Manus raw response text: {raw_response.text}")
response_json = raw_response.json()
verbose_logger.debug(f"Manus file response: {response_json}")
# Parse created_at timestamp
created_at_str = response_json.get("created_at", "")
if created_at_str:
try:
# Try parsing ISO format
created_at = int(
time.mktime(
time.strptime(
created_at_str.replace("Z", "+00:00")[:19],
"%Y-%m-%dT%H:%M:%S",
)
)
)
except (ValueError, TypeError):
created_at = int(time.time())
else:
created_at = int(time.time())
return OpenAIFileObject(
id=response_json.get("id", ""),
bytes=response_json.get("bytes", 0),
created_at=created_at,
filename=response_json.get("filename", ""),
object="file",
purpose=response_json.get("purpose", "assistants"),
status="uploaded", # After successful upload, status is uploaded
status_details=response_json.get("status_details"),
)
except Exception as e:
verbose_logger.exception(f"Error parsing Manus file response: {str(e)}")
raise ValueError(f"Error parsing Manus file response: {str(e)}")
def transform_retrieve_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Get URL and params for retrieving a file."""
api_base = self.get_complete_url(
api_base=litellm_params.get("api_base"),
api_key=litellm_params.get("api_key"),
model="",
optional_params=optional_params,
litellm_params=litellm_params,
)
return f"{api_base}/{file_id}", {}
def transform_retrieve_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> OpenAIFileObject:
"""Transform retrieve file response."""
return self.transform_create_file_response(
model=None,
raw_response=raw_response,
logging_obj=logging_obj,
litellm_params=litellm_params,
)
def transform_delete_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Get URL and params for deleting a file."""
api_base = self.get_complete_url(
api_base=litellm_params.get("api_base"),
api_key=litellm_params.get("api_key"),
model="",
optional_params=optional_params,
litellm_params=litellm_params,
)
return f"{api_base}/{file_id}", {}
def transform_delete_file_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> FileDeleted:
"""Transform delete file response."""
response_json = raw_response.json()
return FileDeleted(**response_json)
def transform_list_files_request(
self,
purpose: Optional[str],
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Get URL and params for listing files."""
api_base = self.get_complete_url(
api_base=litellm_params.get("api_base"),
api_key=litellm_params.get("api_key"),
model="",
optional_params=optional_params,
litellm_params=litellm_params,
)
params = {}
if purpose:
params["purpose"] = purpose
return api_base, params
def transform_list_files_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> List[OpenAIFileObject]:
"""Transform list files response."""
response_json = raw_response.json()
files_data = response_json.get("data", [])
return [self._parse_file_dict(f) for f in files_data]
def _parse_file_dict(self, file_dict: Dict[str, Any]) -> OpenAIFileObject:
"""Parse a file dict into OpenAIFileObject."""
created_at_str = file_dict.get("created_at", "")
if created_at_str:
try:
created_at = int(
time.mktime(
time.strptime(
created_at_str.replace("Z", "+00:00")[:19],
"%Y-%m-%dT%H:%M:%S",
)
)
)
except (ValueError, TypeError):
created_at = int(time.time())
else:
created_at = int(time.time())
return OpenAIFileObject(
id=file_dict.get("id", ""),
bytes=file_dict.get("bytes", 0),
created_at=created_at,
filename=file_dict.get("filename", ""),
object="file",
purpose=file_dict.get("purpose", "assistants"),
status=file_dict.get("status", "uploaded"),
status_details=file_dict.get("status_details"),
)
def transform_file_content_request(
self,
file_content_request: FileContentRequest,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
"""Get URL and params for retrieving file content."""
file_id = file_content_request.get("file_id")
api_base = self.get_complete_url(
api_base=litellm_params.get("api_base"),
api_key=litellm_params.get("api_key"),
model="",
optional_params=optional_params,
litellm_params=litellm_params,
)
return f"{api_base}/{file_id}/content", {}
def transform_file_content_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> HttpxBinaryResponseContent:
"""Transform file content response."""
return HttpxBinaryResponseContent(response=raw_response)

View file

@ -1,3 +1,4 @@
import uuid
from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple, Union
import httpx
@ -227,6 +228,12 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig):
total_tokens=0,
)
# Ensure id is present - failed responses may not include it
if "id" not in raw_response_json or raw_response_json.get("id") is None:
# Generate a placeholder id for failed responses
# This allows the response object to be created even when the API doesn't return an id
raw_response_json["id"] = f"unknown-{uuid.uuid4().hex[:8]}"
try:
response = ResponsesAPIResponse(**raw_response_json)
except Exception:
@ -296,6 +303,28 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig):
raw_response_headers = dict(raw_response.headers)
processed_headers = process_response_headers(raw_response_headers)
# Ensure reasoning, text, output, and usage are present with defaults
if "reasoning" not in raw_response_json or raw_response_json.get("reasoning") is None:
raw_response_json["reasoning"] = {}
if "text" not in raw_response_json or raw_response_json.get("text") is None:
raw_response_json["text"] = {}
if "output" not in raw_response_json or raw_response_json.get("output") is None:
raw_response_json["output"] = []
if "usage" not in raw_response_json or raw_response_json.get("usage") is None:
raw_response_json["usage"] = ResponseAPIUsage(
input_tokens=0,
output_tokens=0,
total_tokens=0,
)
# Ensure id is present - failed responses may not include it
if "id" not in raw_response_json or raw_response_json.get("id") is None:
# Generate a placeholder id for failed responses
raw_response_json["id"] = f"unknown-{uuid.uuid4().hex[:8]}"
try:
response = ResponsesAPIResponse(**raw_response_json)
except Exception:

View file

@ -1,11 +1,12 @@
import json
import os
import time
from litellm._uuid import uuid
from typing import Any, Dict, List, Optional, Tuple, Union
from httpx import Headers, Response
from openai.types.file_deleted import FileDeleted
from litellm._uuid import uuid
from litellm.files.utils import FilesAPIUtils
from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@ -24,6 +25,7 @@ from litellm.types.llms.openai import (
AllMessageValues,
CreateFileRequest,
FileTypes,
HttpxBinaryResponseContent,
OpenAICreateFileRequestOptionalParams,
OpenAIFileObject,
PathLike,
@ -333,6 +335,70 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
status_code=status_code, message=error_message, headers=headers
)
def transform_retrieve_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("VertexAIFilesConfig does not support file retrieval")
def transform_retrieve_file_response(
self,
raw_response: Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> OpenAIFileObject:
raise NotImplementedError("VertexAIFilesConfig does not support file retrieval")
def transform_delete_file_request(
self,
file_id: str,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("VertexAIFilesConfig does not support file deletion")
def transform_delete_file_response(
self,
raw_response: Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> FileDeleted:
raise NotImplementedError("VertexAIFilesConfig does not support file deletion")
def transform_list_files_request(
self,
purpose: Optional[str],
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("VertexAIFilesConfig does not support file listing")
def transform_list_files_response(
self,
raw_response: Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> List[OpenAIFileObject]:
raise NotImplementedError("VertexAIFilesConfig does not support file listing")
def transform_file_content_request(
self,
file_content_request,
optional_params: dict,
litellm_params: dict,
) -> tuple[str, dict]:
raise NotImplementedError("VertexAIFilesConfig does not support file content retrieval")
def transform_file_content_response(
self,
raw_response: Response,
logging_obj: LiteLLMLoggingObj,
litellm_params: dict,
) -> HttpxBinaryResponseContent:
raise NotImplementedError("VertexAIFilesConfig does not support file content retrieval")
class VertexAIJsonlFilesTransformation(VertexGeminiConfig):
"""

View file

@ -388,10 +388,6 @@ class VertexBase:
Internal function. Returns the token and url for the call.
Handles logic if it's google ai studio vs. vertex ai.
For Vertex AI:
- If gemini_api_key is provided, use API key authentication (x-goog-api-key header)
- Otherwise, use service account credentials (OAuth2 Bearer token)
Returns
token, url
@ -404,7 +400,7 @@ class VertexBase:
stream=stream,
gemini_api_key=gemini_api_key,
)
auth_header = None # this field is not used for gemini
auth_header = None # this field is not used for gemin
else:
vertex_location = self.get_vertex_region(
vertex_region=vertex_location,
@ -413,32 +409,14 @@ class VertexBase:
### SET RUNTIME ENDPOINT ###
version = "v1beta1" if should_use_v1beta1_features is True else "v1"
# Check if using API key authentication for Vertex AI
if gemini_api_key and not vertex_credentials:
# When using API key with Vertex AI, use the Google AI Studio endpoint
# This is because Vertex AI API keys work with generativelanguage.googleapis.com
verbose_logger.debug(
f"Using Vertex AI API key authentication for model: {model} - routing to Google AI Studio endpoint"
)
url, endpoint = _get_gemini_url(
mode=mode,
model=model,
stream=stream,
gemini_api_key=gemini_api_key,
)
# API key is already included in the URL by _get_gemini_url
auth_header = None
else:
# Use OAuth2 Bearer token authentication (traditional Vertex AI)
url, endpoint = _get_vertex_url(
mode=mode,
model=model,
stream=stream,
vertex_project=vertex_project,
vertex_location=vertex_location,
vertex_api_version=version,
)
url, endpoint = _get_vertex_url(
mode=mode,
model=model,
stream=stream,
vertex_project=vertex_project,
vertex_location=vertex_location,
vertex_api_version=version,
)
return self._check_custom_proxy(
api_base=api_base,

View file

@ -189,7 +189,7 @@ from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from .llms.custom_llm import CustomLLM, custom_chat_llm_router
from .llms.databricks.embed.handler import DatabricksEmbeddingHandler
from .llms.deprecated_providers import aleph_alpha, palm
from .llms.gemini.common_utils import get_api_key_from_env, get_vertex_api_key_from_env
from .llms.gemini.common_utils import get_api_key_from_env
from .llms.groq.chat.handler import GroqChatCompletion
from .llms.heroku.chat.transformation import HerokuChatConfig
from .llms.huggingface.embedding.handler import HuggingFaceEmbedding
@ -3230,12 +3230,6 @@ def completion( # type: ignore # noqa: PLR0915
or get_secret("VERTEXAI_CREDENTIALS")
)
vertex_api_key = (
api_key
or get_vertex_api_key_from_env()
or litellm.api_key
)
api_base = api_base or litellm.api_base or get_secret("VERTEXAI_API_BASE")
new_params = safe_deep_copy(optional_params or {})
@ -3277,7 +3271,7 @@ def completion( # type: ignore # noqa: PLR0915
vertex_location=vertex_ai_location,
vertex_project=vertex_ai_project,
vertex_credentials=vertex_credentials,
gemini_api_key=vertex_api_key, # Support for Vertex AI API Key
gemini_api_key=None,
logging_obj=logging,
acompletion=acompletion,
timeout=timeout,

View file

@ -1794,11 +1794,20 @@ async def get_org_object(
user_api_key_cache: DualCache,
parent_otel_span: Optional[Span] = None,
proxy_logging_obj: Optional[ProxyLogging] = None,
include_budget_table: bool = False,
) -> Optional[LiteLLM_OrganizationTable]:
"""
- Check if org id in proxy Org Table
- if valid, return LiteLLM_OrganizationTable object
- if not, then raise an error
Args:
org_id: Organization ID to look up
prisma_client: Database client
user_api_key_cache: Cache for storing results
parent_otel_span: Optional OpenTelemetry span
proxy_logging_obj: Optional proxy logging object
include_budget_table: If True, includes litellm_budget_table in the query
"""
if prisma_client is None:
raise Exception(
@ -1807,8 +1816,13 @@ async def get_org_object(
if not isinstance(org_id, str):
return None
# Use different cache key if budget table is included
cache_key = "org_id:{}".format(org_id)
if include_budget_table:
cache_key = "org_id:{}:with_budget".format(org_id)
# check if in cache
cached_org_obj = user_api_key_cache.async_get_cache(key="org_id:{}".format(org_id))
cached_org_obj = user_api_key_cache.async_get_cache(key=cache_key)
if cached_org_obj is not None:
if isinstance(cached_org_obj, dict):
return LiteLLM_OrganizationTable(**cached_org_obj)
@ -1816,13 +1830,24 @@ async def get_org_object(
return cached_org_obj
# else, check db
try:
query_kwargs = {"where": {"organization_id": org_id}}
if include_budget_table:
query_kwargs["include"] = {"litellm_budget_table": True}
response = await prisma_client.db.litellm_organizationtable.find_unique(
where={"organization_id": org_id}
**query_kwargs
)
if response is None:
raise Exception
# Cache the result
await user_api_key_cache.async_set_cache(
key=cache_key,
value=response.model_dump() if hasattr(response, "model_dump") else response,
ttl=DEFAULT_IN_MEMORY_TTL,
)
return response
except Exception:
raise Exception(
@ -2344,11 +2369,14 @@ async def _organization_max_budget_check(
if org_id is None:
return
# Get organization object with budget table to check current spend and max budget
# Get organization object with budget table - use get_org_object so it can be mocked in tests
try:
org_table = await prisma_client.db.litellm_organizationtable.find_unique(
where={"organization_id": org_id},
include={"litellm_budget_table": True},
org_table = await get_org_object(
org_id=org_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
proxy_logging_obj=proxy_logging_obj,
include_budget_table=True,
)
except Exception:
# If organization lookup fails, skip the check

View file

@ -4479,21 +4479,35 @@ def validate_model_access(
) -> None:
"""
Validate that a model is accessible to the user.
Supports batch requests with comma-separated model IDs.
Args:
model_id: The model ID to validate
model_id: The model ID to validate (can be comma-separated for batch requests)
available_models: List of models available to the user
Raises:
HTTPException: If the model is not accessible
"""
if model_id not in available_models:
raise HTTPException(
status_code=404,
detail="The model `{}` does not exist or is not accessible".format(
model_id
),
)
# Handle batch requests with comma-separated models
if "," in model_id:
models = [m.strip() for m in model_id.split(",")]
inaccessible_models = [m for m in models if m not in available_models]
if inaccessible_models:
raise HTTPException(
status_code=404,
detail="The following model(s) do not exist or are not accessible: {}".format(
", ".join(inaccessible_models)
),
)
else:
# Single model validation
if model_id not in available_models:
raise HTTPException(
status_code=404,
detail="The model `{}` does not exist or is not accessible".format(
model_id
),
)
def _path_matches_pattern(path: str, pattern: str) -> bool:

View file

@ -198,11 +198,14 @@ class ResponsesAPIRequestUtils:
model_id = model_info.get("id")
# access the response id based on the object type
response_id = (
responses_api_response["id"]
if isinstance(responses_api_response, dict)
else responses_api_response.id
)
if isinstance(responses_api_response, dict):
response_id = responses_api_response.get("id")
else:
response_id = getattr(responses_api_response, "id", None)
# If no response_id, return the response as-is (likely an error response)
if response_id is None:
return responses_api_response
updated_id = ResponsesAPIRequestUtils._build_responses_api_response_id(
model_id=model_id,

View file

@ -4486,21 +4486,9 @@ class Router:
if hasattr(original_exception, "message"):
# add the available fallbacks to the exception
deployment_info = ""
if kwargs is not None:
metadata = kwargs.get('metadata', {})
if metadata and 'deployment' in metadata:
deployment_info = f"\nUsed Deployment: {metadata['deployment']}"
if 'model_info' in metadata:
model_info = metadata['model_info']
if isinstance(model_info, dict):
deployment_info += f"\nDeployment ID: {model_info.get('id', 'unknown')}"
original_exception.message += ( # type: ignore
f". Received Model Group={model_group}"
f"\nAvailable Model Group Fallbacks={fallback_model_group}"
f"{deployment_info}"
f"\n\n💡 Tip: If using wildcard patterns (e.g., 'openai/*'), ensure all matching deployments have credentials with access to this model."
original_exception.message += ". Received Model Group={}\nAvailable Model Group Fallbacks={}".format( # type: ignore
model_group,
fallback_model_group,
)
if len(fallback_failure_exception_str) > 0:
original_exception.message += ( # type: ignore
@ -7679,10 +7667,6 @@ class Router:
)
if pattern_deployments:
verbose_router_logger.debug(
f"Pattern match for model='{model}': Found {len(pattern_deployments)} deployments. "
f"Deployment IDs: {[d.get('model_info', {}).get('id', 'unknown') for d in pattern_deployments]}"
)
return model, pattern_deployments
if (

View file

@ -1,6 +1,8 @@
from enum import Enum
from types import MappingProxyType
from typing import List, Set, Mapping
from typing import Any, Dict, List, Literal, Mapping, Set, Union
from typing_extensions import Required, TypedDict
"""
Base Enums/Consts
@ -281,3 +283,41 @@ GEMINI_1_5_ACCEPTED_FILE_TYPES: Set[FileType] = {
def is_gemini_1_5_accepted_file_type(file_type: FileType) -> bool:
return file_type in GEMINI_1_5_ACCEPTED_FILE_TYPES
"""
Two-Step File Upload Types
"""
class TwoStepFileUploadRequest(TypedDict):
"""
Request structure for two-step file upload process.
Step 1: Initial request to get upload URL
Step 2: Upload file content to the upload URL
Used by providers like Manus and Google Cloud Storage.
"""
method: Required[str]
url: Required[str]
headers: Required[Dict[str, str]]
data: Required[Union[str, bytes, Dict[str, Any]]]
class TwoStepFileUploadConfig(TypedDict, total=False):
"""
Configuration for two-step file upload process.
Properties:
initial_request: Request to create file record and get upload URL
upload_request: Request to upload actual file content
upload_url_location: Where to find upload URL ('headers' or 'body')
upload_url_key: Key name for upload URL in response (default: 'upload_url')
"""
initial_request: Required[TwoStepFileUploadRequest]
upload_request: Required[TwoStepFileUploadRequest]
upload_url_location: Required[Literal["headers", "body"]]
upload_url_key: str

View file

@ -48,6 +48,7 @@ from tokenizers import Tokenizer
import litellm
import litellm.litellm_core_utils
# audio_utils.utils is lazy-loaded - only imported when needed for transcription calls
import litellm.litellm_core_utils.json_validation_rule
from litellm._lazy_imports import (
@ -71,8 +72,6 @@ from litellm.constants import (
TOOL_CHOICE_OBJECT_TOKEN_COUNT,
)
_CachingHandlerResponse = None
_LLMCachingHandler = None
_CustomGuardrail = None
@ -7959,6 +7958,10 @@ class ProviderConfigManager:
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
return BedrockFilesConfig()
elif LlmProviders.MANUS == provider:
from litellm.llms.manus.files.transformation import ManusFilesConfig
return ManusFilesConfig()
return None
@staticmethod

8
poetry.lock generated
View file

@ -3081,15 +3081,15 @@ files = [
[[package]]
name = "litellm-proxy-extras"
version = "0.4.18"
version = "0.4.21"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
optional = true
python-versions = "!=2.7.*,!=3.0.*,!=3.1.*,!=3.2.*,!=3.3.*,!=3.4.*,!=3.5.*,!=3.6.*,!=3.7.*,>=3.8"
groups = ["main"]
markers = "extra == \"proxy\""
files = [
{file = "litellm_proxy_extras-0.4.18-py3-none-any.whl", hash = "sha256:c3edee68bf8eb073c6158dcf7df05727dfc829e63c03a617fcb48853d11490df"},
{file = "litellm_proxy_extras-0.4.18.tar.gz", hash = "sha256:898b28e3e74acdc29142906b84787ab05a90e30aa3c0c8aee849915e3a16adb3"},
{file = "litellm_proxy_extras-0.4.21-py3-none-any.whl", hash = "sha256:83a1734e9773610945230606012e602bbcbfba1c60fde836d51102c1a296f166"},
{file = "litellm_proxy_extras-0.4.21.tar.gz", hash = "sha256:fa0e012984aa8e5114f88f4bad53d6abb589e5ca3eab445f74f8ddeceb62d848"},
]
[[package]]
@ -7981,4 +7981,4 @@ utils = ["numpydoc"]
[metadata]
lock-version = "2.1"
python-versions = ">=3.9,<4.0"
content-hash = "e9fd12b5ccc703ec156d98877452417083e3ac18b5970cb3a58c3bde09d267bb"
content-hash = "ea62b77c662ab9fc486e421c576f0868bcde16d62a24703ee1f4916a0465ffb2"

View file

@ -2335,6 +2335,7 @@
"audio_speech": false,
"moderations": false,
"batches": false,
"files": true,
"rerank": false,
"a2a": true,
"interactions": true

View file

@ -167,7 +167,7 @@ requires = ["poetry-core", "wheel"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "1.80.13"
version = "1.80.14"
version_files = [
"pyproject.toml:^version"
]

View file

@ -0,0 +1,72 @@
"""
E2E test for all Manus Files API methods.
"""
import os
import pytest
import litellm
@pytest.mark.asyncio
async def test_manus_files_api_e2e_all_methods():
"""
E2E test for Manus Files API: create, retrieve, list, delete.
"""
litellm._turn_on_debug()
api_key = os.getenv("MANUS_API_KEY")
if api_key is None:
pytest.skip("MANUS_API_KEY not set")
# Create a simple test file content
test_content = b"This is a test file for Manus Files API - all methods test."
test_filename = "test_file_all_methods.txt"
# Step 1: Create file
print("Step 1: Creating file...")
created_file = await litellm.acreate_file(
file=(test_filename, test_content),
purpose="assistants",
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Created file: {created_file}")
assert created_file.filename == test_filename
assert created_file.status == "uploaded"
# Note: Manus doesn't return bytes in initial response
file_id = created_file.id
# Step 2: Retrieve file
print(f"\nStep 2: Retrieving file {file_id}...")
retrieved_file = await litellm.afile_retrieve(
file_id=file_id,
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Retrieved file: {retrieved_file}")
assert retrieved_file.id == file_id
assert retrieved_file.filename == test_filename
# Step 3: List files
print("\nStep 3: Listing files...")
files_list = await litellm.afile_list(
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Files list: {files_list}")
assert isinstance(files_list, list)
assert any(f.id == file_id for f in files_list)
# Step 4: Delete file
print(f"\nStep 4: Deleting file {file_id}...")
deleted_file = await litellm.afile_delete(
file_id=file_id,
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Deleted file: {deleted_file}")
assert deleted_file.id == file_id
assert deleted_file.deleted is True
print("\n✅ All Manus Files API methods working!")

View file

@ -35,6 +35,7 @@ IGNORE_FUNCTIONS = [
"_fix_enum_types", # max depth set.
"_collect_argument_paths", # max depth set.
"_split_text", # max depth set.
"_mask_sequence", # max depth set.
"_delete_nested_value_custom", # max depth set (bounded by number of path segments).
"filter_exceptions_from_params", # max depth set (default 20) to prevent infinite recursion.
"__getattr__", # lazy loading pattern in litellm/__init__.py with proper caching to prevent infinite recursion.

View file

@ -1124,124 +1124,6 @@ def test_get_custom_labels_from_metadata_tags(monkeypatch):
assert get_custom_labels_from_metadata(metadata) == {}
def test_get_custom_labels_from_top_level_metadata(monkeypatch):
"""
Test that get_custom_labels_from_metadata can extract fields from top-level metadata,
such as requester_ip_address, not just from nested dictionaries like requester_metadata.
"""
monkeypatch.setattr(
"litellm.custom_prometheus_metadata_labels",
["requester_ip_address", "user_api_key_alias"],
)
# Simulate metadata structure with top-level fields
metadata = {
"requester_ip_address": "10.48.203.20", # Top-level field
"user_api_key_alias": "TestAlias", # Top-level field
"requester_metadata": {"nested_field": "nested_value"}, # Nested dict (excluded)
"user_api_key_auth_metadata": {"another_nested": "value"}, # Nested dict (excluded)
}
result = get_custom_labels_from_metadata(metadata)
assert result == {
"requester_ip_address": "10.48.203.20",
"user_api_key_alias": "TestAlias",
}
def test_get_custom_labels_from_top_level_and_nested_metadata(monkeypatch):
"""
Test that get_custom_labels_from_metadata can extract fields from both top-level
and nested metadata (requester_metadata, user_api_key_auth_metadata).
"""
monkeypatch.setattr(
"litellm.custom_prometheus_metadata_labels",
[
"requester_ip_address", # Top-level
"metadata.foo", # From requester_metadata
"metadata.bar", # From user_api_key_auth_metadata
],
)
# Simulate combined_metadata structure as it would appear after merging
# This is what gets passed to get_custom_labels_from_metadata
combined_metadata = {
"requester_ip_address": "10.48.203.20", # Top-level field
"foo": "bar_value", # From requester_metadata (spread)
"bar": "baz_value", # From user_api_key_auth_metadata (spread)
}
result = get_custom_labels_from_metadata(combined_metadata)
assert result == {
"requester_ip_address": "10.48.203.20",
"metadata_foo": "bar_value",
"metadata_bar": "baz_value",
}
async def test_async_log_success_event_with_top_level_metadata(prometheus_logger, monkeypatch):
"""
Test that async_log_success_event correctly extracts custom labels from top-level metadata
fields like requester_ip_address, not just from nested dictionaries.
"""
# Configure custom metadata labels to extract requester_ip_address
monkeypatch.setattr(
"litellm.custom_prometheus_metadata_labels", ["requester_ip_address"]
)
# Create standard logging payload with requester_ip_address at top-level metadata
standard_logging_object = create_standard_logging_payload()
standard_logging_object["metadata"]["requester_ip_address"] = "10.48.203.20"
standard_logging_object["metadata"]["requester_metadata"] = {} # Empty nested dict
standard_logging_object["metadata"]["user_api_key_auth_metadata"] = {} # Empty nested dict
kwargs = {
"model": "gpt-3.5-turbo",
"stream": True,
"litellm_params": {
"metadata": {
"user_api_key": "test_key",
"user_api_key_user_id": "test_user",
"user_api_key_team_id": "test_team",
"user_api_key_end_user_id": "test_end_user",
}
},
"start_time": datetime.now(),
"completion_start_time": datetime.now(),
"api_call_start_time": datetime.now(),
"end_time": datetime.now() + timedelta(seconds=1),
"standard_logging_object": standard_logging_object,
}
response_obj = MagicMock()
# Mock the prometheus client methods
prometheus_logger.litellm_requests_metric = MagicMock()
prometheus_logger.litellm_spend_metric = MagicMock()
prometheus_logger.litellm_tokens_metric = MagicMock()
prometheus_logger.litellm_input_tokens_metric = MagicMock()
prometheus_logger.litellm_output_tokens_metric = MagicMock()
prometheus_logger.litellm_remaining_team_budget_metric = MagicMock()
prometheus_logger.litellm_remaining_api_key_budget_metric = MagicMock()
prometheus_logger.litellm_remaining_api_key_requests_for_model = MagicMock()
prometheus_logger.litellm_remaining_api_key_tokens_for_model = MagicMock()
prometheus_logger.litellm_llm_api_time_to_first_token_metric = MagicMock()
prometheus_logger.litellm_llm_api_latency_metric = MagicMock()
prometheus_logger.litellm_request_total_latency_metric = MagicMock()
await prometheus_logger.async_log_success_event(
kwargs, response_obj, kwargs["start_time"], kwargs["end_time"]
)
# Verify that the metrics were called with labels including requester_ip_address
# Check that labels() was called - the actual labels dict should include requester_ip_address
assert prometheus_logger.litellm_requests_metric.labels.called
assert prometheus_logger.litellm_spend_metric.labels.called
# Get the actual call arguments to verify requester_ip_address is included
# The custom labels should be extracted and included in the label factory
call_args = prometheus_logger.litellm_requests_metric.labels.call_args
assert call_args is not None
# The labels() method receives a dict with label names and values
# We can't easily assert the exact values without checking the internal implementation,
# but we've verified the function is called, which means the extraction happened
def test_get_custom_labels_from_tags(monkeypatch):
from litellm.integrations.prometheus import get_custom_labels_from_tags

View file

@ -0,0 +1,71 @@
"""
E2E test for all Manus Files API methods.
"""
import os
import pytest
import litellm
@pytest.mark.asyncio
async def test_manus_files_api_e2e_all_methods():
"""
E2E test for Manus Files API: create, retrieve, list, delete.
"""
litellm._turn_on_debug()
api_key = os.getenv("MANUS_API_KEY")
if api_key is None:
pytest.skip("MANUS_API_KEY not set")
# Create a simple test file content
test_content = b"This is a test file for Manus Files API - all methods test."
test_filename = "test_file_all_methods.txt"
# Step 1: Create file
print("Step 1: Creating file...")
created_file = await litellm.acreate_file(
file=(test_filename, test_content),
purpose="assistants",
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Created file: {created_file}")
assert created_file.filename == test_filename
assert created_file.status == "uploaded"
# Note: Manus doesn't return bytes in initial response
file_id = created_file.id
# Step 2: Retrieve file
print(f"\nStep 2: Retrieving file {file_id}...")
retrieved_file = await litellm.afile_retrieve(
file_id=file_id,
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Retrieved file: {retrieved_file}")
assert retrieved_file.id == file_id
assert retrieved_file.filename == test_filename
# Step 3: List files
print("\nStep 3: Listing files...")
files_list = await litellm.afile_list(
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Files list: {files_list}")
assert isinstance(files_list, list)
assert any(f.id == file_id for f in files_list)
# Step 4: Delete file
print(f"\nStep 4: Deleting file {file_id}...")
deleted_file = await litellm.afile_delete(
file_id=file_id,
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Deleted file: {deleted_file}")
assert deleted_file.id == file_id
assert deleted_file.deleted is True
print("\n✅ All Manus Files API methods working!")

View file

@ -38,7 +38,11 @@ async def test_manus_responses_api_with_agent_profile():
print("Manus response=", json.dumps(response, indent=4, default=str))
## Get the status of the response
got_response = await litellm.aget_responses(response_id=response.id)
got_response = await litellm.aget_responses(
response_id=response.id,
custom_llm_provider="manus",
api_key=os.getenv("MANUS_API_KEY"),
)
print("GET API MANUS RESPONSE=", json.dumps(got_response, indent=4, default=str))
if got_response.status == "completed":
assert got_response.output is not None
@ -47,6 +51,95 @@ async def test_manus_responses_api_with_agent_profile():
# Manus can return "running" or "pending" status
assert got_response.status in ["running", "pending"]
assert got_response.id is not None
@pytest.mark.asyncio
async def test_manus_responses_api_with_file_upload():
"""
Test that uploads a file via Files API and then passes it to Responses API.
"""
litellm._turn_on_debug()
api_key = os.getenv("MANUS_API_KEY")
if api_key is None:
pytest.skip("MANUS_API_KEY not set")
# Step 1: Upload a file
test_content = b"Warren Buffett's 2023 Letter to Shareholders\n\nKey Points:\n1. Long-term value creation\n2. Capital allocation strategy\n3. Market volatility perspective"
test_filename = "buffett_letter_summary.txt"
print("Step 1: Uploading file...")
uploaded_file = await litellm.acreate_file(
file=(test_filename, test_content),
purpose="assistants",
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Uploaded file: {uploaded_file}")
assert uploaded_file.id is not None
file_id = uploaded_file.id
# Step 2: Create a response with the uploaded file
print(f"\nStep 2: Creating response with file {file_id}...")
response = await litellm.aresponses(
model="manus/manus-1.6-lite",
input=[
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "Summarize the key points from this letter.",
},
{
"type": "input_file",
"file_id": file_id,
},
],
},
],
api_key=api_key,
max_output_tokens=100,
)
print(f"Response created: {response}")
print(f"Response type: {type(response)}")
print(f"Response has id: {hasattr(response, 'id')}")
# Handle both dict and ResponsesAPIResponse object
if isinstance(response, dict):
response_id = response.get("id")
else:
response_id = getattr(response, "id", None)
assert response_id is not None, f"Response ID is None. Response: {response}"
# Step 3: Get the response status
print(f"\nStep 3: Getting response status...")
got_response = await litellm.aget_responses(
response_id=response_id,
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Response status: {got_response}")
got_response_id = getattr(got_response, "id", None)
got_response_status = getattr(got_response, "status", None)
assert got_response_id == response_id
assert got_response_status in ["completed", "running", "pending"]
# Step 4: Clean up - delete the file
print(f"\nStep 4: Cleaning up - deleting file {file_id}...")
deleted_file = await litellm.afile_delete(
file_id=file_id,
custom_llm_provider="manus",
api_key=api_key,
)
print(f"Deleted file: {deleted_file}")
assert deleted_file.deleted is True
print("\n✅ File upload and responses API integration test passed!")

View file

@ -642,8 +642,8 @@ def test_embedding(mock_aembedding, client_no_auth):
during_call_kwargs = mock_during_hook.await_args_list[0].kwargs
assert (
during_call_kwargs.get("call_type") == "embeddings"
), f"expected during_call_hook to receive call_type='embeddings', got {during_call_kwargs.get('call_type')}"
during_call_kwargs.get("call_type") == "embedding"
), f"expected during_call_hook to receive call_type='embedding', got {during_call_kwargs.get('call_type')}"
except Exception as e:
pytest.fail(f"LiteLLM Proxy test failed. Exception - {str(e)}")
@ -2185,6 +2185,10 @@ async def test_proxy_server_prisma_setup():
mock_client._set_spend_logs_row_count_in_proxy_state = (
AsyncMock()
) # Mock the _set_spend_logs_row_count_in_proxy_state method
# Mock the db attribute with start_token_refresh_task for RDS IAM token refresh
mock_db = MagicMock()
mock_db.start_token_refresh_task = AsyncMock()
mock_client.db = mock_db
await ProxyStartupEvent._setup_prisma_client(
database_url=os.getenv("DATABASE_URL"),

View file

@ -1574,6 +1574,10 @@ async def test_health_check_not_called_when_disabled(monkeypatch):
mock_prisma.health_check = AsyncMock()
mock_prisma.check_view_exists = AsyncMock()
mock_prisma._set_spend_logs_row_count_in_proxy_state = AsyncMock()
# Mock the db attribute with start_token_refresh_task for RDS IAM token refresh
mock_db = MagicMock()
mock_db.start_token_refresh_task = AsyncMock()
mock_prisma.db = mock_db
# Mock PrismaClient constructor
monkeypatch.setattr(
"litellm.proxy.proxy_server.PrismaClient", lambda **kwargs: mock_prisma

View file

@ -108,14 +108,16 @@ def test_lists_with_sensitive_keys_are_masked():
"""
masker = SensitiveDataMasker()
data = {
"api_key": ["sk-123", "sk-456"],
"api_key": ["sk-1234567890abcdef", "sk-9876543210fedcba"],
"tags": ["prod", "test"],
}
masked = masker.mask_dict(data)
# sensitive key list entries should be masked
assert masked["api_key"][0] != "sk-123"
assert masked["api_key"][0] != "sk-1234567890abcdef"
assert "*" in masked["api_key"][0]
assert masked["api_key"][1] != "sk-9876543210fedcba"
assert "*" in masked["api_key"][1]
# non-sensitive list should remain unchanged
assert masked["tags"] == ["prod", "test"]

View file

@ -1,349 +0,0 @@
"""
Test SSL verification for AWS Bedrock boto3 clients.
This test ensures that custom CA certificates are properly passed to all boto3 clients
(STS and Bedrock services) to support internal certificate authorities.
Issue: https://github.com/BerriAI/litellm/issues/XXXX
User reported that SSL_CERT_FILE environment variable and ssl_verify config were not
being applied to boto3 clients, causing "certificate verify failed" errors.
"""
import os
import sys
import tempfile
from unittest.mock import MagicMock, Mock, patch
import pytest
sys.path.insert(0, os.path.abspath("../.."))
import litellm
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import init_bedrock_client
class TestBedrockSSLVerify:
"""Test suite for SSL verification in Bedrock boto3 clients."""
def test_base_aws_llm_get_ssl_verify_default(self):
"""Test that _get_ssl_verify returns default value when no custom config is set."""
base_aws = BaseAWSLLM()
# Clear any environment variables
os.environ.pop("SSL_VERIFY", None)
os.environ.pop("SSL_CERT_FILE", None)
# Reset litellm.ssl_verify to default
litellm.ssl_verify = True
ssl_verify = base_aws._get_ssl_verify()
assert ssl_verify is True
def test_base_aws_llm_get_ssl_verify_false(self):
"""Test that _get_ssl_verify returns False when SSL verification is disabled."""
base_aws = BaseAWSLLM()
# Set SSL_VERIFY to False via environment
os.environ["SSL_VERIFY"] = "False"
ssl_verify = base_aws._get_ssl_verify()
assert ssl_verify is False
# Clean up
os.environ.pop("SSL_VERIFY", None)
def test_base_aws_llm_get_ssl_verify_custom_ca_bundle(self):
"""Test that _get_ssl_verify returns custom CA bundle path when SSL_CERT_FILE is set."""
base_aws = BaseAWSLLM()
# Create a temporary CA bundle file
with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
f.write("-----BEGIN CERTIFICATE-----\n")
f.write("FAKE CERTIFICATE FOR TESTING\n")
f.write("-----END CERTIFICATE-----\n")
ca_bundle_path = f.name
try:
# Set SSL_CERT_FILE environment variable
os.environ["SSL_CERT_FILE"] = ca_bundle_path
os.environ.pop("SSL_VERIFY", None)
litellm.ssl_verify = True
ssl_verify = base_aws._get_ssl_verify()
assert ssl_verify == ca_bundle_path
finally:
# Clean up
os.environ.pop("SSL_CERT_FILE", None)
os.unlink(ca_bundle_path)
def test_base_aws_llm_get_ssl_verify_litellm_config(self):
"""Test that _get_ssl_verify uses litellm.ssl_verify when set."""
base_aws = BaseAWSLLM()
# Clear environment variables
os.environ.pop("SSL_VERIFY", None)
os.environ.pop("SSL_CERT_FILE", None)
# Create a temporary CA bundle file
with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
f.write("-----BEGIN CERTIFICATE-----\n")
f.write("FAKE CERTIFICATE FOR TESTING\n")
f.write("-----END CERTIFICATE-----\n")
ca_bundle_path = f.name
try:
# Set litellm.ssl_verify to custom CA bundle
litellm.ssl_verify = ca_bundle_path
ssl_verify = base_aws._get_ssl_verify()
# When ssl_verify is a path, it should be returned directly
assert ssl_verify == ca_bundle_path
finally:
# Clean up
litellm.ssl_verify = True
os.unlink(ca_bundle_path)
@patch("boto3.client")
def test_init_bedrock_client_passes_ssl_verify_to_sts(self, mock_boto3_client):
"""Test that init_bedrock_client passes ssl_verify to STS client."""
# Create a temporary CA bundle file
with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
f.write("-----BEGIN CERTIFICATE-----\n")
f.write("FAKE CERTIFICATE FOR TESTING\n")
f.write("-----END CERTIFICATE-----\n")
ca_bundle_path = f.name
try:
# Set SSL_CERT_FILE environment variable
os.environ["SSL_CERT_FILE"] = ca_bundle_path
litellm.ssl_verify = True
# Mock the STS client and Bedrock client
mock_sts_client = MagicMock()
mock_sts_response = {
"Credentials": {
"AccessKeyId": "test_access_key",
"SecretAccessKey": "test_secret_key",
"SessionToken": "test_session_token",
}
}
mock_sts_client.assume_role.return_value = mock_sts_response
mock_bedrock_client = MagicMock()
# Configure mock to return different clients based on service name
def side_effect(service_name=None, **kwargs):
if service_name == "sts":
return mock_sts_client
elif service_name == "bedrock-runtime":
return mock_bedrock_client
return MagicMock()
mock_boto3_client.side_effect = side_effect
# Call init_bedrock_client with role assumption
client = init_bedrock_client(
aws_region_name="us-west-2",
aws_access_key_id="test_key",
aws_secret_access_key="test_secret",
aws_role_name="arn:aws:iam::123456789012:role/test-role",
aws_session_name="test-session",
)
# Verify that boto3.client was called with verify parameter for STS
sts_calls = [
call for call in mock_boto3_client.call_args_list
if (len(call[0]) > 0 and call[0][0] == "sts") or
("service_name" not in call[1]) # STS calls don't use service_name kwarg
]
assert len(sts_calls) > 0, "STS client should have been created"
# Check that verify parameter was passed to STS client
sts_call = sts_calls[0]
assert "verify" in sts_call[1], "verify parameter should be passed to STS client"
assert sts_call[1]["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {sts_call[1]['verify']}"
# Verify that boto3.client was called with verify parameter for Bedrock
bedrock_calls = [
call for call in mock_boto3_client.call_args_list
if "service_name" in call[1] and call[1]["service_name"] == "bedrock-runtime"
]
assert len(bedrock_calls) > 0, "Bedrock client should have been created"
bedrock_call = bedrock_calls[0]
assert "verify" in bedrock_call[1], "verify parameter should be passed to Bedrock client"
assert bedrock_call[1]["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {bedrock_call[1]['verify']}"
finally:
# Clean up
os.environ.pop("SSL_CERT_FILE", None)
os.unlink(ca_bundle_path)
@patch("boto3.client")
def test_base_aws_llm_auth_with_role_passes_ssl_verify(self, mock_boto3_client):
"""Test that _auth_with_aws_role passes ssl_verify to STS client."""
base_aws = BaseAWSLLM()
# Create a temporary CA bundle file
with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
f.write("-----BEGIN CERTIFICATE-----\n")
f.write("FAKE CERTIFICATE FOR TESTING\n")
f.write("-----END CERTIFICATE-----\n")
ca_bundle_path = f.name
try:
# Set SSL_CERT_FILE environment variable
os.environ["SSL_CERT_FILE"] = ca_bundle_path
litellm.ssl_verify = True
# Mock the STS client
mock_sts_client = MagicMock()
mock_sts_response = {
"Credentials": {
"AccessKeyId": "test_access_key",
"SecretAccessKey": "test_secret_key",
"SessionToken": "test_session_token",
"Expiration": "2025-01-10T00:00:00Z",
}
}
# Convert Expiration to datetime
from datetime import datetime, timezone
mock_sts_response["Credentials"]["Expiration"] = datetime.now(timezone.utc)
mock_sts_client.assume_role.return_value = mock_sts_response
mock_boto3_client.return_value = mock_sts_client
# Call _auth_with_aws_role
credentials, ttl = base_aws._auth_with_aws_role(
aws_access_key_id="test_key",
aws_secret_access_key="test_secret",
aws_session_token=None,
aws_role_name="arn:aws:iam::123456789012:role/test-role",
aws_session_name="test-session",
)
# Verify that boto3.client was called with verify parameter
assert mock_boto3_client.called, "boto3.client should have been called"
call_kwargs = mock_boto3_client.call_args[1]
assert "verify" in call_kwargs, "verify parameter should be passed to STS client"
assert call_kwargs["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {call_kwargs['verify']}"
finally:
# Clean up
os.environ.pop("SSL_CERT_FILE", None)
os.unlink(ca_bundle_path)
@patch("litellm.llms.bedrock.base_aws_llm.get_secret")
@patch("boto3.client")
def test_base_aws_llm_auth_with_web_identity_passes_ssl_verify(self, mock_boto3_client, mock_get_secret):
"""Test that _auth_with_web_identity_token passes ssl_verify to STS client."""
base_aws = BaseAWSLLM()
# Create a temporary CA bundle file
with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
f.write("-----BEGIN CERTIFICATE-----\n")
f.write("FAKE CERTIFICATE FOR TESTING\n")
f.write("-----END CERTIFICATE-----\n")
ca_bundle_path = f.name
try:
# Set SSL_CERT_FILE environment variable
os.environ["SSL_CERT_FILE"] = ca_bundle_path
litellm.ssl_verify = True
# Mock get_secret to return the token
mock_get_secret.return_value = "mocked_oidc_token"
# Mock the STS client
mock_sts_client = MagicMock()
mock_sts_response = {
"Credentials": {
"AccessKeyId": "test_access_key",
"SecretAccessKey": "test_secret_key",
"SessionToken": "test_session_token",
},
"PackedPolicySize": 100,
}
mock_sts_client.assume_role_with_web_identity.return_value = mock_sts_response
# Mock boto3.Session
mock_session = MagicMock()
mock_credentials = MagicMock()
mock_session.get_credentials.return_value = mock_credentials
mock_boto3_client.return_value = mock_sts_client
with patch("boto3.Session", return_value=mock_session):
# Call _auth_with_web_identity_token
credentials, ttl = base_aws._auth_with_web_identity_token(
aws_web_identity_token="test_token",
aws_role_name="arn:aws:iam::123456789012:role/test-role",
aws_session_name="test-session",
aws_region_name="us-west-2",
aws_sts_endpoint=None,
)
# Verify that boto3.client was called with verify parameter
assert mock_boto3_client.called, "boto3.client should have been called"
call_kwargs = mock_boto3_client.call_args[1]
assert "verify" in call_kwargs, "verify parameter should be passed to STS client"
assert call_kwargs["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {call_kwargs['verify']}"
finally:
# Clean up
os.environ.pop("SSL_CERT_FILE", None)
os.unlink(ca_bundle_path)
def test_ssl_verify_priority_env_over_litellm_config(self):
"""Test that SSL_VERIFY environment variable takes priority over litellm.ssl_verify."""
base_aws = BaseAWSLLM()
# Set litellm.ssl_verify to True
litellm.ssl_verify = True
# Set SSL_VERIFY environment variable to False
os.environ["SSL_VERIFY"] = "False"
try:
ssl_verify = base_aws._get_ssl_verify()
assert ssl_verify is False, "Environment variable should take priority"
finally:
# Clean up
os.environ.pop("SSL_VERIFY", None)
litellm.ssl_verify = True
def test_ssl_cert_file_priority_over_default(self):
"""Test that SSL_CERT_FILE takes priority when ssl_verify is True."""
base_aws = BaseAWSLLM()
# Create a temporary CA bundle file
with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f:
f.write("-----BEGIN CERTIFICATE-----\n")
f.write("FAKE CERTIFICATE FOR TESTING\n")
f.write("-----END CERTIFICATE-----\n")
ca_bundle_path = f.name
try:
# Set SSL_CERT_FILE environment variable
os.environ["SSL_CERT_FILE"] = ca_bundle_path
os.environ.pop("SSL_VERIFY", None)
litellm.ssl_verify = True
ssl_verify = base_aws._get_ssl_verify()
assert ssl_verify == ca_bundle_path, "SSL_CERT_FILE should be used when ssl_verify is True"
finally:
# Clean up
os.environ.pop("SSL_CERT_FILE", None)
os.unlink(ca_bundle_path)
if __name__ == "__main__":
# Run tests
pytest.main([__file__, "-v", "-s"])

View file

@ -13,7 +13,6 @@ sys.path.insert(
import litellm
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
from litellm.llms.vertex_ai.common_utils import _get_gemini_url
def run_sync(coro):
@ -1049,139 +1048,3 @@ class TestVertexBase:
MockCredentials.from_info.assert_called_once_with(json_obj)
mock_creds.with_scopes.assert_called_once_with(scopes)
assert result == "scoped_creds"
def test_get_token_and_url_with_api_key(self):
"""Test that API key authentication routes to Google AI Studio endpoint"""
vertex_base = VertexBase()
# Test with API key and no credentials - should use Google AI Studio endpoint
auth_header, url = vertex_base._get_token_and_url(
model="gemini-2.0-flash-exp",
auth_header=None,
gemini_api_key="test-api-key-123",
vertex_project="test-project",
vertex_location="us-central1",
vertex_credentials=None, # No service account credentials
stream=False,
custom_llm_provider="vertex_ai",
api_base=None,
should_use_v1beta1_features=False,
mode="chat",
)
# Should route to Google AI Studio endpoint
assert "generativelanguage.googleapis.com" in url
assert "gemini-2.0-flash-exp" in url
assert "key=test-api-key-123" in url
assert auth_header is None # API key is in URL, not header
def test_get_token_and_url_with_credentials(self):
"""Test that service account credentials route to Vertex AI endpoint"""
vertex_base = VertexBase()
mock_creds = MagicMock()
mock_creds.token = "mock-bearer-token"
mock_creds.expired = False
with patch.object(
vertex_base, "_ensure_access_token", return_value=("mock-bearer-token", "test-project")
):
# Test with credentials - should use Vertex AI endpoint
auth_header, url = vertex_base._get_token_and_url(
model="gemini-2.0-flash-exp",
auth_header="mock-bearer-token",
gemini_api_key=None,
vertex_project="test-project",
vertex_location="us-central1",
vertex_credentials={"type": "service_account"},
stream=False,
custom_llm_provider="vertex_ai",
api_base=None,
should_use_v1beta1_features=False,
mode="chat",
)
# Should route to Vertex AI endpoint
assert "aiplatform.googleapis.com" in url
assert "projects/test-project" in url
assert "locations/us-central1" in url
assert auth_header == "mock-bearer-token"
def test_get_token_and_url_api_key_with_streaming(self):
"""Test API key authentication with streaming enabled"""
vertex_base = VertexBase()
auth_header, url = vertex_base._get_token_and_url(
model="gemini-2.0-flash-exp",
auth_header=None,
gemini_api_key="test-api-key-456",
vertex_project="test-project",
vertex_location="us-central1",
vertex_credentials=None,
stream=True, # Streaming enabled
custom_llm_provider="vertex_ai",
api_base=None,
should_use_v1beta1_features=False,
mode="chat",
)
# Should route to Google AI Studio endpoint with streaming
assert "generativelanguage.googleapis.com" in url
assert "streamGenerateContent" in url
assert "key=test-api-key-456" in url
assert "alt=sse" in url
assert auth_header is None
def test_get_token_and_url_api_key_priority(self):
"""Test that credentials take priority over API key when both are provided"""
vertex_base = VertexBase()
# When both API key and credentials are provided, credentials take priority
mock_creds = MagicMock()
mock_creds.token = "mock-bearer-token"
mock_creds.expired = False
with patch.object(
vertex_base, "_ensure_access_token", return_value=("mock-bearer-token", "test-project")
):
auth_header, url = vertex_base._get_token_and_url(
model="gemini-2.0-flash-exp",
auth_header="mock-bearer-token",
gemini_api_key="test-api-key-789",
vertex_project="test-project",
vertex_location="us-central1",
vertex_credentials={"type": "service_account"}, # Credentials provided
stream=False,
custom_llm_provider="vertex_ai",
api_base=None,
should_use_v1beta1_features=False,
mode="chat",
)
# Should use Vertex AI endpoint with Bearer token (credentials take priority)
assert "aiplatform.googleapis.com" in url
assert auth_header == "mock-bearer-token"
def test_get_token_and_url_with_embedding_mode(self):
"""Test API key authentication with embedding mode"""
vertex_base = VertexBase()
auth_header, url = vertex_base._get_token_and_url(
model="text-embedding-004",
auth_header=None,
gemini_api_key="test-embedding-key",
vertex_project="test-project",
vertex_location="us-central1",
vertex_credentials=None,
stream=False,
custom_llm_provider="vertex_ai",
api_base=None,
should_use_v1beta1_features=False,
mode="embedding",
)
# Should route to Google AI Studio endpoint for embeddings
assert "generativelanguage.googleapis.com" in url
assert "embedContent" in url
assert "key=test-embedding-key" in url
assert auth_header is None