diff --git a/docs/my-website/docs/providers/manus.md b/docs/my-website/docs/providers/manus.md index 2981ec6d247..92bf2b9b966 100644 --- a/docs/my-website/docs/providers/manus.md +++ b/docs/my-website/docs/providers/manus.md @@ -9,7 +9,7 @@ Use Manus AI agents through LiteLLM's OpenAI-compatible Responses API. |----------|---------| | Description | Manus is an AI agent platform for complex reasoning tasks, document analysis, and multi-step workflows with asynchronous task execution. | | Provider Route on LiteLLM | `manus/{agent_profile}` | -| Supported Operations | `/responses` (Responses API) | +| Supported Operations | `/responses` (Responses API), `/files` (Files API) | | Provider Doc | [Manus API ↗](https://open.manus.im/docs/openai-compatibility) | ## Model Format @@ -188,7 +188,182 @@ For production applications, use [webhooks](https://open.manus.im/docs/webhooks) | `max_output_tokens` | āœ… | Limits response length | | `previous_response_id` | āœ… | For multi-turn conversations | +## Files API + +Manus supports file uploads for document analysis and processing. Files can be uploaded and then referenced in Responses API calls. + +### LiteLLM Python SDK + +```python showLineNumbers title="Upload, Use, Retrieve, and Delete Files" +import litellm +import os + +# Set API key +os.environ["MANUS_API_KEY"] = "your-manus-api-key" + +# Upload file +file_content = b"This is a document for analysis." +created_file = await litellm.acreate_file( + file=("document.txt", file_content), + purpose="assistants", + custom_llm_provider="manus", +) +print(f"Uploaded file: {created_file.id}") + +# Use file with Responses API +response = await litellm.aresponses( + model="manus/manus-1.6", + input=[ + { + "role": "user", + "content": [ + {"type": "input_text", "text": "Summarize this document."}, + {"type": "input_file", "file_id": created_file.id}, + ], + }, + ], + extra_body={"task_mode": "agent", "agent_profile": "manus-1.6-agent"}, +) +print(f"Response: {response.id}") + +# Retrieve file +retrieved_file = await litellm.afile_retrieve( + file_id=created_file.id, + custom_llm_provider="manus", +) +print(f"File details: {retrieved_file.filename}, {retrieved_file.bytes} bytes") + +# Delete file +deleted_file = await litellm.afile_delete( + file_id=created_file.id, + custom_llm_provider="manus", +) +print(f"Deleted: {deleted_file.deleted}") +``` + +### LiteLLM AI Gateway + + + + +```bash showLineNumbers title="Upload File" +# Upload file +curl -X POST http://localhost:4000/v1/files \ + -H "Authorization: Bearer your-proxy-key" \ + -F "file=@document.txt" \ + -F "purpose=assistants" \ + -F "custom_llm_provider=manus" + +# Response +{ + "id": "file_abc123", + "object": "file", + "bytes": 1024, + "created_at": 1234567890, + "filename": "document.txt", + "purpose": "assistants", + "status": "uploaded" +} +``` + +```bash showLineNumbers title="Use File with Responses API" +# Create response with file +curl -X POST http://localhost:4000/responses \ + -H "Authorization: Bearer your-proxy-key" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "manus-agent", + "input": [ + { + "role": "user", + "content": [ + {"type": "input_text", "text": "Summarize this document."}, + {"type": "input_file", "file_id": "file_abc123"} + ] + } + ] + }' +``` + +```bash showLineNumbers title="Retrieve File" +# Get file details +curl http://localhost:4000/v1/files/file_abc123 \ + -H "Authorization: Bearer your-proxy-key" + +# Response +{ + "id": "file_abc123", + "object": "file", + "bytes": 1024, + "created_at": 1234567890, + "filename": "document.txt", + "purpose": "assistants", + "status": "uploaded" +} +``` + +```bash showLineNumbers title="Delete File" +# Delete file +curl -X DELETE http://localhost:4000/v1/files/file_abc123 \ + -H "Authorization: Bearer your-proxy-key" + +# Response +{ + "id": "file_abc123", + "object": "file", + "deleted": true +} +``` + + + + +```python showLineNumbers title="Upload, Use, Retrieve, and Delete Files" +import openai + +client = openai.OpenAI( + base_url="http://localhost:4000", + api_key="your-proxy-key" +) + +# Upload file +with open("document.txt", "rb") as f: + created_file = client.files.create( + file=f, + purpose="assistants", + extra_body={"custom_llm_provider": "manus"} + ) +print(f"Uploaded file: {created_file.id}") + +# Use file with Responses API +response = client.responses.create( + model="manus-agent", + input=[ + { + "role": "user", + "content": [ + {"type": "input_text", "text": "Summarize this document."}, + {"type": "input_file", "file_id": created_file.id} + ] + } + ] +) +print(f"Response: {response.id}") + +# Retrieve file +retrieved_file = client.files.retrieve(created_file.id) +print(f"File: {retrieved_file.filename}, {retrieved_file.bytes} bytes") + +# Delete file +deleted_file = client.files.delete(created_file.id) +print(f"Deleted: {deleted_file.deleted}") +``` + + + + ## Related Documentation - [LiteLLM Responses API](/docs/response_api) +- [LiteLLM Files API](/docs/proxy/litellm_managed_files) - [Manus OpenAI Compatibility](https://open.manus.im/docs/openai-compatibility) diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index f46608aa57c..33ebf535d29 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -35,8 +35,6 @@ import json # !gcloud auth application-default login - run this to add vertex credentials to your env ## OR ## file_path = 'path/to/vertex_ai_service_account.json' -## OR ## -export VERTEXAI_API_KEY="your-api-key" # Load the JSON file with open(file_path, 'r') as file: @@ -49,7 +47,7 @@ vertex_credentials_json = json.dumps(vertex_credentials) response = completion( model="vertex_ai/gemini-2.5-pro", messages=[{ "content": "Hello, how are you?","role": "user"}], - vertex_credentials=vertex_credentials_json # Can remove this is added VERTEXAI_API_KEY in env + vertex_credentials=vertex_credentials_json ) ``` @@ -1331,41 +1329,15 @@ Here's how to use Vertex AI with the LiteLLM Proxy Server ## Authentication - vertex_project, vertex_location, etc. -LiteLLM supports two authentication methods for Vertex AI: - -1. **API Key Authentication** (Recommended for getting started) -2. **Service Account Credentials** (Recommended for production) - Set your vertex credentials via: - dynamic params OR - env vars -### **Authentication Method 1: -The simplest way to authenticate with Vertex AI. You can set: -- `api_key` (str) - Your Vertex AI API key +### **Dynamic Params** -**Environment Variables:** -```bash -export VERTEXAI_API_KEY="your-api-key" -``` - -**Or pass as parameters:** -```python -from litellm import completion - -response = completion( - model="vertex_ai/gemini-2.0-flash-exp", - messages=[{"role": "user", "content": "Hello!"}], - api_key="your-vertex-api-key", - -) -``` - -### **Authentication Method 2: Service Account Credentials** - -For production environments with fine-grained access control. You can set: +You can set: - `vertex_credentials` (str) - can be a json string or filepath to your vertex ai service account.json - `vertex_location` (str) - place where vertex model is deployed (us-central1, asia-southeast1, etc.). Some models support the global location, please see [Vertex AI documentation](https://cloud.google.com/vertex-ai/generative-ai/docs/learn/locations#supported_models) - `vertex_project` Optional[str] - use if vertex project different from the one in vertex_credentials @@ -1420,16 +1392,7 @@ model_list: ### **Environment Variables** -#### For API Key Authentication: - -- `VERTEXAI_API_KEY` or `VERTEX_API_KEY` - Your Vertex AI API key - -```bash -export VERTEXAI_API_KEY="your-vertex-api-key" -``` - -#### For Service Account Authentication: - +You can set: - `GOOGLE_APPLICATION_CREDENTIALS` - store the filepath for your service_account.json in here (used by vertex sdk directly). - VERTEXAI_LOCATION - place where vertex model is deployed (us-central1, asia-southeast1, etc.) - VERTEXAI_PROJECT - Optional[str] - use if vertex project different from the one in vertex_credentials diff --git a/docs/my-website/img/ui_endpoint_activity.png b/docs/my-website/img/ui_endpoint_activity.png new file mode 100644 index 00000000000..e2550066a07 Binary files /dev/null and b/docs/my-website/img/ui_endpoint_activity.png differ diff --git a/docs/my-website/release_notes/v1.80.11-stable/index.md b/docs/my-website/release_notes/v1.80.11-stable/index.md index b671b795602..ea0e6083dfc 100644 --- a/docs/my-website/release_notes/v1.80.11-stable/index.md +++ b/docs/my-website/release_notes/v1.80.11-stable/index.md @@ -1,5 +1,5 @@ --- -title: "[Preview] v1.80.11 - Google Interactions API" +title: "v1.80.11 - Google Interactions API" slug: "v1-80-11" date: 2025-12-20T10:00:00 authors: @@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem'; docker run \ -e STORE_MODEL_IN_DB=True \ -p 4000:4000 \ -docker.litellm.ai/berriai/litellm:v1.80.11.rc.1 +docker.litellm.ai/berriai/litellm:v1.80.11-stable ``` diff --git a/docs/my-website/release_notes/v1.80.14/index.md b/docs/my-website/release_notes/v1.80.14/index.md new file mode 100644 index 00000000000..38751f80aeb --- /dev/null +++ b/docs/my-website/release_notes/v1.80.14/index.md @@ -0,0 +1,589 @@ +--- +title: "v1.80.14 - Manus API Support" +slug: "v1-80-14" +date: 2026-01-10T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg + - name: Ishaan Jaff + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg +hide_table_of_contents: false +--- + +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +## Deploy this version + + + + +``` showLineNumbers title="docker run litellm" +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +docker.litellm.ai/berriai/litellm:v1.80.14-stable +``` + + + + + +``` showLineNumbers title="pip install litellm" +pip install litellm==1.80.14 +``` + + + + +--- + +## Key Highlights + +- **Manus API Support** - [New provider support for Manus API on /responses and GET /responses endpoints](../../docs/providers/manus) +- **MiniMax Provider** - [Full support for MiniMax chat completions, TTS, and Anthropic native endpoint](../../docs/providers/minimax) +- **AWS Polly TTS** - [New TTS provider using AWS Polly API](../../docs/providers/aws_polly) +- **SSO Role Mapping** - Configure role mappings for SSO providers directly in the UI +- **Cost Estimator** - New UI tool for estimating costs across multiple models and requests +- **MCP Global Mode** - [Configure MCP servers globally with visibility controls](../../docs/mcp) +- **Interactions API Bridge** - [Use all LiteLLM providers with the Interactions API](../../docs/interactions) +- **RAG Query Endpoint** - [New RAG Search/Query endpoint for retrieval-augmented generation](../../docs/search/index) +- **92.7% Faster Provider Config Lookup** - Major performance improvement for provider configuration +- **UI Usage - Endpoint Activity** - Users can now see Endpoint Activity Metrics in the UI. + + +--- + +### UI Usage - Endpoint Activity + + + +Users can now see Endpoint Activity Metrics in the UI. + +--- + +## New Providers and Endpoints + +### New Providers (11 new providers) + +| Provider | Supported LiteLLM Endpoints | Description | +| -------- | ------------------- | ----------- | +| [Manus](../../docs/providers/manus) | `/responses` | Manus API for agentic workflows | +| [Manus](../../docs/providers/manus) | `GET /responses` | Manus API for retrieving responses | +| [Manus](../../docs/providers/manus) | `/files` | Manus API for file management | +| [MiniMax](../../docs/providers/minimax) | `/chat/completions` | MiniMax chat completions | +| [MiniMax](../../docs/providers/minimax) | `/audio/speech` | MiniMax text-to-speech | +| [AWS Polly](../../docs/providers/aws_polly) | `/audio/speech` | AWS Polly text-to-speech API | +| [GigaChat](../../docs/providers/gigachat) | `/chat/completions` | GigaChat provider for Russian language AI | +| [LlamaGate](../../docs/providers/llamagate) | `/chat/completions` | LlamaGate chat completions | +| [LlamaGate](../../docs/providers/llamagate) | `/embeddings` | LlamaGate embeddings | +| [Abliteration AI](../../docs/providers/abliteration) | `/chat/completions` | Abliteration.ai provider support | +| [Bedrock](../../docs/providers/bedrock) | `/v1/messages/count_tokens` | Bedrock as new provider for token counting | + +### New LLM API Endpoints (3 new endpoints) + +| Endpoint | Method | Description | Documentation | +| -------- | ------ | ----------- | ------------- | +| `/responses/compact` | POST | Compact responses API endpoint | [Docs](../../docs/response_api) | +| `/rag/query` | POST | RAG Search/Query endpoint | [Docs](../../docs/search/index) | +| `/containers/{id}/files` | POST | Upload files to containers | [Docs](../../docs/container_files) | + +--- + +## New Models / Updated Models + +#### New Model Support (100+ new models) + +| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Features | +| -------- | ----- | -------------- | ------------------- | -------------------- | -------- | +| Azure | `azure/gpt-5.2` | 400K | $1.75 | $14.00 | Reasoning, vision, caching | +| Azure | `azure/gpt-5.2-chat` | 128K | $1.75 | $14.00 | Reasoning, vision | +| Azure | `azure/gpt-5.2-pro` | 400K | $21.00 | $168.00 | Reasoning, vision, web search | +| Azure | `azure/gpt-image-1.5` | - | Token-based | Token-based | Image generation/editing | +| Azure AI | `azure_ai/gpt-oss-120b` | 131K | $0.15 | $0.60 | Function calling | +| Azure AI | `azure_ai/flux.2-pro` | - | - | $0.04/image | Image generation | +| Azure AI | `azure_ai/deepseek-v3.2` | 164K | $0.58 | $1.68 | Reasoning, function calling | +| Bedrock | `amazon.nova-2-multimodal-embeddings-v1:0` | 8K | $0.135 | - | Multimodal embeddings | +| Bedrock | `writer.palmyra-x4-v1:0` | 128K | $2.50 | $10.00 | Function calling, PDF | +| Bedrock | `writer.palmyra-x5-v1:0` | 1M | $0.60 | $6.00 | Function calling, PDF | +| Bedrock | `moonshot.kimi-k2-v1:0` | - | - | - | Kimi K2 model | +| Cerebras | `cerebras/zai-glm-4.6` | 128K | $2.25 | $2.75 | Reasoning, function calling | +| GigaChat | `gigachat/GigaChat-2-Lite` | - | - | - | Chat completions | +| GigaChat | `gigachat/GigaChat-2-Max` | - | - | - | Chat completions | +| GigaChat | `gigachat/GigaChat-2-Pro` | - | - | - | Chat completions | +| Gemini | `gemini/veo-3.1-generate-001` | - | - | - | Video generation | +| Gemini | `gemini/veo-3.1-fast-generate-001` | - | - | - | Video generation | +| GitHub Copilot | 25+ models | Various | - | - | Chat completions | +| LlamaGate | 15+ models | Various | - | - | Chat, vision, embeddings | +| MiniMax | `minimax/abab7-chat-preview` | - | - | - | Chat completions | +| Novita | 80+ models | Various | Various | Various | Chat, vision, embeddings | +| OpenRouter | `openrouter/google/gemini-3-flash-preview` | - | - | - | Chat completions | +| Together AI | Multiple models | Various | Various | Various | Response schema support | +| Vertex AI | `vertex_ai/zai-glm-4.7` | - | - | - | GLM 4.7 support | + +#### Features + +- **[Gemini](../../docs/providers/gemini)** + - Add image tokens in chat completion - [PR #18327](https://github.com/BerriAI/litellm/pull/18327) + - Add usage object in image generation - [PR #18328](https://github.com/BerriAI/litellm/pull/18328) + - Add thought signature support via tool call id - [PR #18374](https://github.com/BerriAI/litellm/pull/18374) + - Add thought signature for non tool call requests - [PR #18581](https://github.com/BerriAI/litellm/pull/18581) + - Preserve system instructions - [PR #18585](https://github.com/BerriAI/litellm/pull/18585) + - Fix Gemini 3 images in tool response - [PR #18190](https://github.com/BerriAI/litellm/pull/18190) + - Support snake_case for google_search tool parameters - [PR #18451](https://github.com/BerriAI/litellm/pull/18451) + - Google GenAI adapter inline data support - [PR #18477](https://github.com/BerriAI/litellm/pull/18477) + - Add deprecation_date for discontinued Google models - [PR #18550](https://github.com/BerriAI/litellm/pull/18550) +- **[Vertex AI](../../docs/providers/vertex)** + - Add centralized get_vertex_base_url() helper for global location support - [PR #18410](https://github.com/BerriAI/litellm/pull/18410) + - Convert image URLs to base64 for Vertex AI Anthropic - [PR #18497](https://github.com/BerriAI/litellm/pull/18497) + - Separate Tool objects for each tool type per API spec - [PR #18514](https://github.com/BerriAI/litellm/pull/18514) + - Add thought_signatures to VertexGeminiConfig - [PR #18853](https://github.com/BerriAI/litellm/pull/18853) + - Add support for Vertex AI API keys - [PR #18806](https://github.com/BerriAI/litellm/pull/18806) + - Add zai glm-4.7 model support - [PR #18782](https://github.com/BerriAI/litellm/pull/18782) +- **[Azure](../../docs/providers/azure/azure)** + - Add Azure gpt-image-1.5 pricing to cost map - [PR #18347](https://github.com/BerriAI/litellm/pull/18347) + - Add azure/gpt-5.2-chat model - [PR #18361](https://github.com/BerriAI/litellm/pull/18361) + - Add support for image generation via Azure AD token - [PR #18413](https://github.com/BerriAI/litellm/pull/18413) + - Add logprobs support for Azure OpenAI GPT-5.2 model - [PR #18856](https://github.com/BerriAI/litellm/pull/18856) + - Add Azure BFL Flux 2 models for image generation and editing - [PR #18764](https://github.com/BerriAI/litellm/pull/18764), [PR #18766](https://github.com/BerriAI/litellm/pull/18766) +- **[Bedrock](../../docs/providers/bedrock)** + - Add Bedrock Kimi K2 model support - [PR #18797](https://github.com/BerriAI/litellm/pull/18797) + - Add support for model id in bedrock passthrough - [PR #18800](https://github.com/BerriAI/litellm/pull/18800) + - Fix Nova model detection for Bedrock provider - [PR #18250](https://github.com/BerriAI/litellm/pull/18250) + - Ensure toolUse.input is always a dict when converting from OpenAI format - [PR #18414](https://github.com/BerriAI/litellm/pull/18414) +- **[Databricks](../../docs/providers/databricks)** + - Add enhanced authentication, security features, and custom user-agent support - [PR #18349](https://github.com/BerriAI/litellm/pull/18349) +- **[MiniMax](../../docs/providers/minimax)** + - Add MiniMax chat completion support - [PR #18380](https://github.com/BerriAI/litellm/pull/18380) + - Add Anthropic native endpoint support for MiniMax - [PR #18377](https://github.com/BerriAI/litellm/pull/18377) + - Add support for MiniMax TTS - [PR #18334](https://github.com/BerriAI/litellm/pull/18334) + - Add MiniMax provider support to UI dashboard - [PR #18496](https://github.com/BerriAI/litellm/pull/18496) +- **[Together AI](../../docs/providers/togetherai)** + - Add supports_response_schema to all supported Together AI models - [PR #18368](https://github.com/BerriAI/litellm/pull/18368) +- **[OpenRouter](../../docs/providers/openrouter)** + - Add OpenRouter embeddings API support - [PR #18391](https://github.com/BerriAI/litellm/pull/18391) +- **[Anthropic](../../docs/providers/anthropic)** + - Pass server_tool_use and tool_search_tool_result blocks - [PR #18770](https://github.com/BerriAI/litellm/pull/18770) + - Add Anthropic cache control option to image tool call results - [PR #18674](https://github.com/BerriAI/litellm/pull/18674) +- **[Ollama](../../docs/providers/ollama)** + - Add dimensions for ollama embedding - [PR #18536](https://github.com/BerriAI/litellm/pull/18536) + - Extract pure base64 data from data URLs for Ollama - [PR #18465](https://github.com/BerriAI/litellm/pull/18465) +- **[Watsonx](../../docs/providers/watsonx/index)** + - Add Watsonx fields support - [PR #18569](https://github.com/BerriAI/litellm/pull/18569) + - Fix Watsonx Audio Transcription - filter model field - [PR #18810](https://github.com/BerriAI/litellm/pull/18810) +- **[SAP](../../docs/providers/sap)** + - Add SAP creds for list in proxy UI - [PR #18375](https://github.com/BerriAI/litellm/pull/18375) + - Pass through extra params from allowed_openai_params - [PR #18432](https://github.com/BerriAI/litellm/pull/18432) + - Add client header for SAP AI Core Tracking - [PR #18714](https://github.com/BerriAI/litellm/pull/18714) +- **[Fireworks AI](../../docs/providers/fireworks_ai)** + - Correct deepseek-v3p2 pricing - [PR #18483](https://github.com/BerriAI/litellm/pull/18483) +- **[ZAI](../../docs/providers/zai)** + - Add GLM-4.7 model with reasoning support - [PR #18476](https://github.com/BerriAI/litellm/pull/18476) +- **[Codestral](../../docs/providers/codestral)** + - Correctly route codestral chat and FIM endpoints - [PR #18467](https://github.com/BerriAI/litellm/pull/18467) +- **[Azure AI](../../docs/providers/azure_ai)** + - Fix authentication errors at messages API via azure_ai - [PR #18500](https://github.com/BerriAI/litellm/pull/18500) + +#### New Provider Support + +- **[AWS Polly](../../docs/providers/aws_polly)** - Add AWS Polly API for TTS - [PR #18326](https://github.com/BerriAI/litellm/pull/18326) +- **[GigaChat](../../docs/providers/gigachat)** - Add GigaChat provider support - [PR #18564](https://github.com/BerriAI/litellm/pull/18564) +- **[LlamaGate](../../docs/providers/llamagate)** - Add LlamaGate as a new provider - [PR #18673](https://github.com/BerriAI/litellm/pull/18673) +- **[Abliteration AI](../../docs/providers/abliteration)** - Add abliteration.ai provider - [PR #18678](https://github.com/BerriAI/litellm/pull/18678) +- **[Manus](../../docs/providers/manus)** - Add Manus API support on /responses, GET /responses - [PR #18804](https://github.com/BerriAI/litellm/pull/18804) +- **5 AI Providers via openai_like** - Add 5 AI providers using openai_like - [PR #18362](https://github.com/BerriAI/litellm/pull/18362) + +### Bug Fixes + +- **[Gemini](../../docs/providers/gemini)** + - Properly catch context window exceeded errors - [PR #18283](https://github.com/BerriAI/litellm/pull/18283) + - Remove prompt caching headers as support has been removed - [PR #18579](https://github.com/BerriAI/litellm/pull/18579) + - Fix generate content request with audio file id - [PR #18745](https://github.com/BerriAI/litellm/pull/18745) + - Fix google_genai streaming adapter provider handling - [PR #18845](https://github.com/BerriAI/litellm/pull/18845) +- **[Groq](../../docs/providers/groq)** + - Remove deprecated Groq models and update model registry - [PR #18062](https://github.com/BerriAI/litellm/pull/18062) +- **[Vertex AI](../../docs/providers/vertex)** + - Handle unsupported region for Vertex AI count tokens endpoint - [PR #18665](https://github.com/BerriAI/litellm/pull/18665) +- **General** + - Fix request body for image embedding request - [PR #18336](https://github.com/BerriAI/litellm/pull/18336) + - Fix lost tool_calls when streaming has both text and tool_calls - [PR #18316](https://github.com/BerriAI/litellm/pull/18316) + - Add all resolution for gpt-image-1.5 - [PR #18586](https://github.com/BerriAI/litellm/pull/18586) + - Fix gpt-image-1 cost calculation using token-based pricing - [PR #17906](https://github.com/BerriAI/litellm/pull/17906) + - Fix response_format leaking into extra_body - [PR #18859](https://github.com/BerriAI/litellm/pull/18859) + - Align max_tokens with max_output_tokens for consistency - [PR #18820](https://github.com/BerriAI/litellm/pull/18820) + +--- + +## LLM API Endpoints + +#### Features + +- **[Responses API](../../docs/response_api)** + - Add new compact endpoint (v1/responses/compact) - [PR #18697](https://github.com/BerriAI/litellm/pull/18697) + - Support more streaming callback hooks - [PR #18513](https://github.com/BerriAI/litellm/pull/18513) + - Add mapping for reasoning effort to summary param - [PR #18635](https://github.com/BerriAI/litellm/pull/18635) + - Add output_text property to ResponsesAPIResponse - [PR #18491](https://github.com/BerriAI/litellm/pull/18491) + - Add annotations to completions responses API bridge - [PR #18754](https://github.com/BerriAI/litellm/pull/18754) +- **[Interactions API](../../docs/interactions)** + - Allow using all LiteLLM providers (interactions -> responses API bridge) - [PR #18373](https://github.com/BerriAI/litellm/pull/18373) +- **[RAG Search API](../../docs/search/index)** + - Add RAG Search/Query endpoint - [PR #18376](https://github.com/BerriAI/litellm/pull/18376) +- **[CountTokens API](../../docs/anthropic_count_tokens)** + - Add Bedrock as a new provider for `/v1/messages/count_tokens` - [PR #18858](https://github.com/BerriAI/litellm/pull/18858) +- **[Generate Content](../../docs/providers/gemini)** + - Add generate content in LLM route - [PR #18405](https://github.com/BerriAI/litellm/pull/18405) +- **General** + - Enable async_post_call_failure_hook to transform error responses - [PR #18348](https://github.com/BerriAI/litellm/pull/18348) + - Calculate total_tokens manually if missing and can be calculated - [PR #18445](https://github.com/BerriAI/litellm/pull/18445) + - Add custom llm provider to get_llm_provider when sent via UI - [PR #18638](https://github.com/BerriAI/litellm/pull/18638) + +#### Bugs + +- **General** + - Handle empty error objects in response conversion - [PR #18493](https://github.com/BerriAI/litellm/pull/18493) + - Preserve client error status codes in streaming mode - [PR #18698](https://github.com/BerriAI/litellm/pull/18698) + - Return json error response instead of SSE format for initial streaming errors - [PR #18757](https://github.com/BerriAI/litellm/pull/18757) + - Fix auth header for custom api base in generateContent request - [PR #18637](https://github.com/BerriAI/litellm/pull/18637) + - Tool content should be string for Deepinfra - [PR #18739](https://github.com/BerriAI/litellm/pull/18739) + - Fix incomplete usage in response object passed - [PR #18799](https://github.com/BerriAI/litellm/pull/18799) + - Unify model names to provider-defined names - [PR #18573](https://github.com/BerriAI/litellm/pull/18573) + +--- + +## Management Endpoints / UI + +#### Features + +- **SSO Configuration** + - Add SSO Role Mapping feature - [PR #18090](https://github.com/BerriAI/litellm/pull/18090) + - Add SSO Settings Page - [PR #18600](https://github.com/BerriAI/litellm/pull/18600) + - Allow adding role mappings for SSO - [PR #18593](https://github.com/BerriAI/litellm/pull/18593) + - SSO Settings Page Add Role Mappings - [PR #18677](https://github.com/BerriAI/litellm/pull/18677) + - SSO Settings Loading State + Deprecate Previous SSO Flow - [PR #18617](https://github.com/BerriAI/litellm/pull/18617) +- **Virtual Keys** + - Allow deleting key expiry - [PR #18278](https://github.com/BerriAI/litellm/pull/18278) + - Add optional query param "expand" to /key/list - [PR #18502](https://github.com/BerriAI/litellm/pull/18502) + - Key Table Loading Skeleton - [PR #18527](https://github.com/BerriAI/litellm/pull/18527) + - Allow column resizing on Keys Table - [PR #18424](https://github.com/BerriAI/litellm/pull/18424) + - Virtual Keys Table Loading State Between Pages - [PR #18619](https://github.com/BerriAI/litellm/pull/18619) + - Key and Team Router Setting - [PR #18790](https://github.com/BerriAI/litellm/pull/18790) + - Allow router_settings on Keys and Teams - [PR #18675](https://github.com/BerriAI/litellm/pull/18675) + - Use timedelta to calculate key expiry on generate - [PR #18666](https://github.com/BerriAI/litellm/pull/18666) +- **Models + Endpoints** + - Add Model Clearer Flow For Team Admins - [PR #18532](https://github.com/BerriAI/litellm/pull/18532) + - Model Page Loading State - [PR #18574](https://github.com/BerriAI/litellm/pull/18574) + - Model Page Model Provider Select Performance - [PR #18425](https://github.com/BerriAI/litellm/pull/18425) + - Model Page Sorting Sorts Entire Set - [PR #18420](https://github.com/BerriAI/litellm/pull/18420) + - Refactor Model Hub Page - [PR #18568](https://github.com/BerriAI/litellm/pull/18568) + - Add request provider form on UI - [PR #18704](https://github.com/BerriAI/litellm/pull/18704) +- **Organizations & Teams** + - Allow Organization Admins to See Organization Tab - [PR #18400](https://github.com/BerriAI/litellm/pull/18400) + - Resolve Organization Alias on Team Table - [PR #18401](https://github.com/BerriAI/litellm/pull/18401) + - Resolve Team Alias in Organization Info View - [PR #18404](https://github.com/BerriAI/litellm/pull/18404) + - Allow Organization Admins to View Their Organization Info - [PR #18417](https://github.com/BerriAI/litellm/pull/18417) + - Allow editing team_member_budget_duration in /team/update - [PR #18735](https://github.com/BerriAI/litellm/pull/18735) + - Reusable Duration Select + Team Update Member Budget Duration - [PR #18736](https://github.com/BerriAI/litellm/pull/18736) +- **Usage & Spend** + - Add Error Code Filtering on Spend Logs - [PR #18359](https://github.com/BerriAI/litellm/pull/18359) + - Add Error Code Filtering on UI - [PR #18366](https://github.com/BerriAI/litellm/pull/18366) + - Usage Page User Max Budget fix - [PR #18555](https://github.com/BerriAI/litellm/pull/18555) + - Add endpoint to Daily Activity Tables - [PR #18729](https://github.com/BerriAI/litellm/pull/18729) + - Endpoint Activity in Usage - [PR #18798](https://github.com/BerriAI/litellm/pull/18798) +- **Cost Estimator** + - Add Cost Estimator for AI Gateway - [PR #18643](https://github.com/BerriAI/litellm/pull/18643) + - Add view for estimating costs across requests - [PR #18645](https://github.com/BerriAI/litellm/pull/18645) + - Allow selecting many models for cost estimator - [PR #18653](https://github.com/BerriAI/litellm/pull/18653) +- **CloudZero** + - Improve Create and Delete Path for CloudZero - [PR #18263](https://github.com/BerriAI/litellm/pull/18263) + - Add CloudZero UI Docs - [PR #18350](https://github.com/BerriAI/litellm/pull/18350) +- **Playground** + - Add MCP test support to completions on Playground - [PR #18440](https://github.com/BerriAI/litellm/pull/18440) + - Add selectable MCP servers to the playground - [PR #18578](https://github.com/BerriAI/litellm/pull/18578) + - Add custom proxy base URL support to Playground - [PR #18661](https://github.com/BerriAI/litellm/pull/18661) +- **General UI** + - UI styling improvements and fixes - [PR #18310](https://github.com/BerriAI/litellm/pull/18310) + - Add reusable "New" badge component for feature highlights - [PR #18537](https://github.com/BerriAI/litellm/pull/18537) + - Hide New Badges - [PR #18547](https://github.com/BerriAI/litellm/pull/18547) + - Change Budget page to Have Tabs - [PR #18576](https://github.com/BerriAI/litellm/pull/18576) + - Clicking on Logo Directs to Correct URL - [PR #18575](https://github.com/BerriAI/litellm/pull/18575) + - Add UI support for configuring meta URLs - [PR #18580](https://github.com/BerriAI/litellm/pull/18580) + - Expire Previous UI Session Tokens on Login - [PR #18557](https://github.com/BerriAI/litellm/pull/18557) + - Add license endpoint - [PR #18311](https://github.com/BerriAI/litellm/pull/18311) + - Router Fields Endpoint + React Query for Router Fields - [PR #18880](https://github.com/BerriAI/litellm/pull/18880) + +#### Bugs + +- **UI Fixes** + - Fix Key Creation MCP Settings Submit Form Unintentionally - [PR #18355](https://github.com/BerriAI/litellm/pull/18355) + - Fix UI Disappears in Development Environments - [PR #18399](https://github.com/BerriAI/litellm/pull/18399) + - Fix Disable Admin UI Flag - [PR #18397](https://github.com/BerriAI/litellm/pull/18397) + - Remove Model Analytics From Model Page - [PR #18552](https://github.com/BerriAI/litellm/pull/18552) + - Useful Links Remove Modal on Adding Links - [PR #18602](https://github.com/BerriAI/litellm/pull/18602) + - SSO Edit Modal Clear Role Mapping Values on Provider Change - [PR #18680](https://github.com/BerriAI/litellm/pull/18680) + - UI Login Case Sensitivity fix - [PR #18877](https://github.com/BerriAI/litellm/pull/18877) +- **API Fixes** + - Fix User Invite & Key Generation Email Notification Logic - [PR #18524](https://github.com/BerriAI/litellm/pull/18524) + - Normalize Proxy Config Callback - [PR #18775](https://github.com/BerriAI/litellm/pull/18775) + - Return empty data array instead of 500 when no models configured - [PR #18556](https://github.com/BerriAI/litellm/pull/18556) + - Enforce org level max budget - [PR #18813](https://github.com/BerriAI/litellm/pull/18813) + +--- + +## AI Integrations + +### New Integrations (4 new integrations) + +| Integration | Type | Description | +| ----------- | ---- | ----------- | +| [Focus](../../docs/observability/focus) | Logging | Focus export support for observability - [PR #18802](https://github.com/BerriAI/litellm/pull/18802) | +| [SigNoz](../../docs/observability/signoz) | Logging | SigNoz integration for observability - [PR #18726](https://github.com/BerriAI/litellm/pull/18726) | +| [Qualifire](../../docs/proxy/guardrails/qualifire) | Guardrails | Qualifire guardrails and eval webhook - [PR #18594](https://github.com/BerriAI/litellm/pull/18594) | +| [Levo AI](../../docs/observability/levo_integration) | Guardrails | Levo AI integration for security - [PR #18529](https://github.com/BerriAI/litellm/pull/18529) | + +### Logging + +- **[DataDog](../../docs/proxy/logging#datadog)** + - Fix span kind fallback when parent_id missing - [PR #18418](https://github.com/BerriAI/litellm/pull/18418) +- **[Langfuse](../../docs/proxy/logging#langfuse)** + - Map Gemini cached_tokens to Langfuse cache_read_input_tokens - [PR #18614](https://github.com/BerriAI/litellm/pull/18614) +- **[Prometheus](../../docs/proxy/logging#prometheus)** + - Align prometheus metric names with DEFINED_PROMETHEUS_METRICS - [PR #18463](https://github.com/BerriAI/litellm/pull/18463) + - Add Prometheus metrics for request queue time and guardrails - [PR #17973](https://github.com/BerriAI/litellm/pull/17973) + - Add caching metrics for cache hits, misses, and tokens - [PR #18755](https://github.com/BerriAI/litellm/pull/18755) + - Skip metrics for invalid API key requests - [PR #18788](https://github.com/BerriAI/litellm/pull/18788) +- **[Braintrust](../../docs/proxy/logging#braintrust)** + - Pass span_attributes in async logging and skip tags on non-root spans - [PR #18409](https://github.com/BerriAI/litellm/pull/18409) +- **[CloudZero](../../docs/proxy/logging#cloudzero)** + - Add user email to CloudZero - [PR #18584](https://github.com/BerriAI/litellm/pull/18584) +- **[OpenTelemetry](../../docs/proxy/logging#opentelemetry)** + - Use already configured opentelemetry providers - [PR #18279](https://github.com/BerriAI/litellm/pull/18279) + - Prevent LiteLLM from closing external OTEL spans - [PR #18553](https://github.com/BerriAI/litellm/pull/18553) + - Allow configuring arize project name for OpenTelemetry service name - [PR #18738](https://github.com/BerriAI/litellm/pull/18738) +- **[LangSmith](../../docs/proxy/logging#langsmith)** + - Add support for LangSmith organization-scoped API keys with tenant ID - [PR #18623](https://github.com/BerriAI/litellm/pull/18623) +- **[Generic API Logger](../../docs/proxy/logging#generic-api-logger)** + - Add log_format option to GenericAPILogger - [PR #18587](https://github.com/BerriAI/litellm/pull/18587) + +### Guardrails + +- **[Content Filter](../../docs/proxy/guardrails/litellm_content_filter)** + - Add content filter logs page - [PR #18335](https://github.com/BerriAI/litellm/pull/18335) + - Log actual event type for guardrails - [PR #18489](https://github.com/BerriAI/litellm/pull/18489) +- **[Qualifire](../../docs/proxy/guardrails/qualifire)** + - Add Qualifire eval webhook - [PR #18836](https://github.com/BerriAI/litellm/pull/18836) +- **[Lasso Security](../../docs/proxy/guardrails/lasso_security)** + - Add Lasso guardrail API docs - [PR #18652](https://github.com/BerriAI/litellm/pull/18652) +- **[Noma Security](../../docs/proxy/guardrails/noma_security)** + - Add MCP guardrail support for Noma - [PR #18668](https://github.com/BerriAI/litellm/pull/18668) +- **[Bedrock Guardrails](../../docs/proxy/guardrails/bedrock)** + - Remove redundant Bedrock guardrail block handling - [PR #18634](https://github.com/BerriAI/litellm/pull/18634) +- **General** + - Generic guardrail API update - [PR #18647](https://github.com/BerriAI/litellm/pull/18647) + - Prevent proxy startup failures from case-sensitive tool permission guardrail validation - [PR #18662](https://github.com/BerriAI/litellm/pull/18662) + - Extend case normalization to ALL guardrail types - [PR #18664](https://github.com/BerriAI/litellm/pull/18664) + - Fix MCP handling in unified guardrail - [PR #18630](https://github.com/BerriAI/litellm/pull/18630) + - Fix embeddings calltype for guardrail precallhook - [PR #18740](https://github.com/BerriAI/litellm/pull/18740) + +--- + +## Spend Tracking, Budgets and Rate Limiting + +- **Platform Fee / Margins** - Add support for Platform Fee / Margins - [PR #18427](https://github.com/BerriAI/litellm/pull/18427) +- **Negative Budget Validation** - Add validation for negative budget - [PR #18583](https://github.com/BerriAI/litellm/pull/18583) +- **Cost Calculation Fixes** + - Correct cost calculation when reasoning_tokens are without text_tokens - [PR #18607](https://github.com/BerriAI/litellm/pull/18607) + - Fix background cost tracking tests - [PR #18588](https://github.com/BerriAI/litellm/pull/18588) +- **Tag Routing** - Support toggling tag matching between ANY and ALL - [PR #18776](https://github.com/BerriAI/litellm/pull/18776) + +--- + +## MCP Gateway + +- **MCP Global Mode** - Add MCP global mode - [PR #18639](https://github.com/BerriAI/litellm/pull/18639) +- **MCP Server Visibility** - Add configurable MCP server visibility - [PR #18681](https://github.com/BerriAI/litellm/pull/18681) +- **MCP Registry** - Add MCP registry - [PR #18850](https://github.com/BerriAI/litellm/pull/18850) +- **MCP Stdio Header** - Support MCP stdio header env overrides - [PR #18324](https://github.com/BerriAI/litellm/pull/18324) +- **Parallel Tool Fetching** - Parallelize tool fetching from multiple MCP servers - [PR #18627](https://github.com/BerriAI/litellm/pull/18627) +- **Optimize MCP Server Listing** - Separate health checks for optimized listing - [PR #18530](https://github.com/BerriAI/litellm/pull/18530) +- **Auth Improvements** + - Require auth for MCP connection test endpoint - [PR #18290](https://github.com/BerriAI/litellm/pull/18290) + - Fix MCP gateway OAuth2 auth issues and ClosedResourceError - [PR #18281](https://github.com/BerriAI/litellm/pull/18281) +- **Bug Fixes** + - Fix MCP server health status reporting - [PR #18443](https://github.com/BerriAI/litellm/pull/18443) + - Fix OpenAPI to MCP tool conversion - [PR #18597](https://github.com/BerriAI/litellm/pull/18597) + - Remove exec() usage and handle invalid OpenAPI parameter names for security - [PR #18480](https://github.com/BerriAI/litellm/pull/18480) + - Fix MCP error when using multiple servers simultaneously - [PR #18855](https://github.com/BerriAI/litellm/pull/18855) +- **Migrate MCP Fetching Logic to React Query** - [PR #18352](https://github.com/BerriAI/litellm/pull/18352) + +--- + +## Performance / Loadbalancing / Reliability improvements + +- **92.7% Faster Provider Config Lookup** - LiteLLM now stresses LLM providers 2.5x more - [PR #18867](https://github.com/BerriAI/litellm/pull/18867) +- **Lazy Loading Improvements** + - Consolidate lazy import handlers with registry pattern - [PR #18389](https://github.com/BerriAI/litellm/pull/18389) + - Complete lazy loading migration for all 180+ LLM config classes - [PR #18392](https://github.com/BerriAI/litellm/pull/18392) + - Lazy load additional components (types, callbacks, utilities) - [PR #18396](https://github.com/BerriAI/litellm/pull/18396) + - Add lazy loading for get_llm_provider - [PR #18591](https://github.com/BerriAI/litellm/pull/18591) + - Lazy-load heavy audio library and loggers - [PR #18592](https://github.com/BerriAI/litellm/pull/18592) + - Lazy load 9 heavy imports in litellm/utils.py - [PR #18595](https://github.com/BerriAI/litellm/pull/18595) + - Lazy load heavy imports to improve import time and memory usage - [PR #18610](https://github.com/BerriAI/litellm/pull/18610) + - Implement lazy loading for provider configs, model info classes, streaming handlers - [PR #18611](https://github.com/BerriAI/litellm/pull/18611) + - Lazy load 15 additional imports - [PR #18613](https://github.com/BerriAI/litellm/pull/18613) + - Lazy load 15+ unused imports - [PR #18616](https://github.com/BerriAI/litellm/pull/18616) + - Lazy load DatadogLLMObsInitParams - [PR #18658](https://github.com/BerriAI/litellm/pull/18658) + - Migrate utils.py lazy imports to registry pattern - [PR #18657](https://github.com/BerriAI/litellm/pull/18657) + - Lazy load get_llm_provider and remove_index_from_tool_calls - [PR #18608](https://github.com/BerriAI/litellm/pull/18608) +- **Router Improvements** + - Validate routing_strategy at startup to fail fast with helpful error - [PR #18624](https://github.com/BerriAI/litellm/pull/18624) + - Correct num_retries tracking in retry logic - [PR #18712](https://github.com/BerriAI/litellm/pull/18712) + - Improve error messages and validation for wildcard routing with multiple credentials - [PR #18629](https://github.com/BerriAI/litellm/pull/18629) +- **Memory Improvements** + - Add memory pattern detection test and fix bad memory patterns - [PR #18589](https://github.com/BerriAI/litellm/pull/18589) + - Add unbounded data structure detection to memory test - [PR #18590](https://github.com/BerriAI/litellm/pull/18590) + - Add memory leak detection tests with CI integration - [PR #18881](https://github.com/BerriAI/litellm/pull/18881) +- **Database** + - Add idx on LOWER(user_email) for faster duplicate email checks - [PR #18828](https://github.com/BerriAI/litellm/pull/18828) + - Proactive RDS IAM token refresh to prevent 15-min connection failed - [PR #18795](https://github.com/BerriAI/litellm/pull/18795) + - Clarify database_connection_pool_limit applies per worker - [PR #18780](https://github.com/BerriAI/litellm/pull/18780) + - Make base_connection_pool_limit default value the same - [PR #18721](https://github.com/BerriAI/litellm/pull/18721) +- **Docker** + - Add libsndfile to database Docker image for audio processing - [PR #18612](https://github.com/BerriAI/litellm/pull/18612) + - Add line_profiler support for performance analysis and fix Windows CRLF issues - [PR #18773](https://github.com/BerriAI/litellm/pull/18773) +- **Helm** + - Add lifecycle support to Helm charts - [PR #18517](https://github.com/BerriAI/litellm/pull/18517) +- **Authentication** + - Add Kubernetes ServiceAccount JWT authentication support - [PR #18055](https://github.com/BerriAI/litellm/pull/18055) + - Use async anthropic client to prevent event loop blocking - [PR #18435](https://github.com/BerriAI/litellm/pull/18435) +- **Logging Worker** + - Handle event loop changes in multiprocessing - [PR #18423](https://github.com/BerriAI/litellm/pull/18423) +- **Security** + - Prevent expired key plaintext leak in error response - [PR #18860](https://github.com/BerriAI/litellm/pull/18860) + - Mask extra header secrets in model info - [PR #18822](https://github.com/BerriAI/litellm/pull/18822) + - Prevent duplicate User-Agent tags in request_tags - [PR #18723](https://github.com/BerriAI/litellm/pull/18723) + - Properly use litellm api keys - [PR #18832](https://github.com/BerriAI/litellm/pull/18832) +- **Misc** + - Remove double imports in main.py - [PR #18406](https://github.com/BerriAI/litellm/pull/18406) + - Add LITELLM_DISABLE_LAZY_LOADING env var to fix VCR cassette creation issue - [PR #18725](https://github.com/BerriAI/litellm/pull/18725) + - Add xiaomi_mimo to LlmProviders enum to fix router support - [PR #18819](https://github.com/BerriAI/litellm/pull/18819) + - Allow installation with current grpcio on old Python - [PR #18473](https://github.com/BerriAI/litellm/pull/18473) + - Add Custom CA certificates to boto3 clients - [PR #18852](https://github.com/BerriAI/litellm/pull/18852) + - Fix bedrock_cache, metadata and max_model_budget - [PR #18872](https://github.com/BerriAI/litellm/pull/18872) + - Fix LiteLLM SDK embedding headers missing field - [PR #18844](https://github.com/BerriAI/litellm/pull/18844) + - Put automatic reasoning summary inclusion behind feat flag - [PR #18688](https://github.com/BerriAI/litellm/pull/18688) + - turn_off_message_logging Does Not Redact Request Messages in proxy_server_request Field - [PR #18897](https://github.com/BerriAI/litellm/pull/18897) + +--- + +## Documentation Updates + +- **Provider Documentation** + - Update MiniMax docs to be in proper format - [PR #18403](https://github.com/BerriAI/litellm/pull/18403) + - Add docs for 5 AI providers - [PR #18388](https://github.com/BerriAI/litellm/pull/18388) + - Fix gpt-5-mini reasoning_effort supported values - [PR #18346](https://github.com/BerriAI/litellm/pull/18346) + - Fix PDF documentation inconsistency in Anthropic page - [PR #18816](https://github.com/BerriAI/litellm/pull/18816) + - Update OpenRouter docs to include embedding support - [PR #18874](https://github.com/BerriAI/litellm/pull/18874) + - Add LITELLM_REASONING_AUTO_SUMMARY in doc - [PR #18705](https://github.com/BerriAI/litellm/pull/18705) +- **MCP Documentation** + - Agentcore MCP server docs - [PR #18603](https://github.com/BerriAI/litellm/pull/18603) + - Mention MCP prompt/resources types in overview - [PR #18669](https://github.com/BerriAI/litellm/pull/18669) + - Add Focus docs - [PR #18837](https://github.com/BerriAI/litellm/pull/18837) +- **Guardrails Documentation** + - Qualifire docs hotfix - [PR #18724](https://github.com/BerriAI/litellm/pull/18724) +- **Infrastructure Documentation** + - IAM Roles Anywhere docs - [PR #18559](https://github.com/BerriAI/litellm/pull/18559) + - Fix formatting in proxy configs documentation - [PR #18498](https://github.com/BerriAI/litellm/pull/18498) + - Fix GCS cache docs missing for proxy mode - [PR #13328](https://github.com/BerriAI/litellm/pull/13328) + - Fix how to execute cloudzero sql - [PR #18841](https://github.com/BerriAI/litellm/pull/18841) +- **General** + - LiteLLM adopters section - [PR #18605](https://github.com/BerriAI/litellm/pull/18605) + - Remove redundant comments about setting litellm.callbacks - [PR #18711](https://github.com/BerriAI/litellm/pull/18711) + - Update header to be markdown bold by removing space - [PR #18846](https://github.com/BerriAI/litellm/pull/18846) + - Manus docs - new provider - [PR #18817](https://github.com/BerriAI/litellm/pull/18817) + +--- + +## New Contributors + +* @prasadkona made their first contribution in [PR #18349](https://github.com/BerriAI/litellm/pull/18349) +* @lucasrothman made their first contribution in [PR #18283](https://github.com/BerriAI/litellm/pull/18283) +* @aggeentik made their first contribution in [PR #18317](https://github.com/BerriAI/litellm/pull/18317) +* @mihidumh made their first contribution in [PR #18361](https://github.com/BerriAI/litellm/pull/18361) +* @Prazeina made their first contribution in [PR #18498](https://github.com/BerriAI/litellm/pull/18498) +* @systec-dk made their first contribution in [PR #18500](https://github.com/BerriAI/litellm/pull/18500) +* @xuan07t2 made their first contribution in [PR #18514](https://github.com/BerriAI/litellm/pull/18514) +* @RensDimmendaal made their first contribution in [PR #18190](https://github.com/BerriAI/litellm/pull/18190) +* @yurekami made their first contribution in [PR #18483](https://github.com/BerriAI/litellm/pull/18483) +* @agertz7 made their first contribution in [PR #18556](https://github.com/BerriAI/litellm/pull/18556) +* @yudelevi made their first contribution in [PR #18550](https://github.com/BerriAI/litellm/pull/18550) +* @smallp made their first contribution in [PR #18536](https://github.com/BerriAI/litellm/pull/18536) +* @kevinpauer made their first contribution in [PR #18569](https://github.com/BerriAI/litellm/pull/18569) +* @cansakiroglu made their first contribution in [PR #18517](https://github.com/BerriAI/litellm/pull/18517) +* @dee-walia20 made their first contribution in [PR #18432](https://github.com/BerriAI/litellm/pull/18432) +* @luxinfeng made their first contribution in [PR #18477](https://github.com/BerriAI/litellm/pull/18477) +* @cantalupo555 made their first contribution in [PR #18476](https://github.com/BerriAI/litellm/pull/18476) +* @andersk made their first contribution in [PR #18473](https://github.com/BerriAI/litellm/pull/18473) +* @majiayu000 made their first contribution in [PR #18467](https://github.com/BerriAI/litellm/pull/18467) +* @amangupta-20 made their first contribution in [PR #18529](https://github.com/BerriAI/litellm/pull/18529) +* @hamzaq453 made their first contribution in [PR #18480](https://github.com/BerriAI/litellm/pull/18480) +* @ktsaou made their first contribution in [PR #18627](https://github.com/BerriAI/litellm/pull/18627) +* @FlibbertyGibbitz made their first contribution in [PR #18624](https://github.com/BerriAI/litellm/pull/18624) +* @drorIvry made their first contribution in [PR #18594](https://github.com/BerriAI/litellm/pull/18594) +* @urainshah made their first contribution in [PR #18524](https://github.com/BerriAI/litellm/pull/18524) +* @mangabits made their first contribution in [PR #18279](https://github.com/BerriAI/litellm/pull/18279) +* @0717376 made their first contribution in [PR #18564](https://github.com/BerriAI/litellm/pull/18564) +* @nmgarza5 made their first contribution in [PR #17330](https://github.com/BerriAI/litellm/pull/17330) +* @wileykestner made their first contribution in [PR #18445](https://github.com/BerriAI/litellm/pull/18445) +* @minijeong-log made their first contribution in [PR #14440](https://github.com/BerriAI/litellm/pull/14440) +* @Isaac4real made their first contribution in [PR #18710](https://github.com/BerriAI/litellm/pull/18710) +* @marukaz made their first contribution in [PR #18711](https://github.com/BerriAI/litellm/pull/18711) +* @rohitravirane made their first contribution in [PR #18712](https://github.com/BerriAI/litellm/pull/18712) +* @lizzzcai made their first contribution in [PR #18714](https://github.com/BerriAI/litellm/pull/18714) +* @hkd987 made their first contribution in [PR #18673](https://github.com/BerriAI/litellm/pull/18673) +* @Mr-Pepe made their first contribution in [PR #18674](https://github.com/BerriAI/litellm/pull/18674) +* @gkarthi-signoz made their first contribution in [PR #18726](https://github.com/BerriAI/litellm/pull/18726) +* @Tianduo16 made their first contribution in [PR #18723](https://github.com/BerriAI/litellm/pull/18723) +* @wilsonjr made their first contribution in [PR #18721](https://github.com/BerriAI/litellm/pull/18721) +* @abliteration-ai made their first contribution in [PR #18678](https://github.com/BerriAI/litellm/pull/18678) +* @danialkhan02 made their first contribution in [PR #18770](https://github.com/BerriAI/litellm/pull/18770) +* @ihower made their first contribution in [PR #18409](https://github.com/BerriAI/litellm/pull/18409) +* @elkkhan made their first contribution in [PR #18391](https://github.com/BerriAI/litellm/pull/18391) +* @runixer made their first contribution in [PR #18435](https://github.com/BerriAI/litellm/pull/18435) +* @choby-shun made their first contribution in [PR #18776](https://github.com/BerriAI/litellm/pull/18776) +* @jutaz made their first contribution in [PR #18853](https://github.com/BerriAI/litellm/pull/18853) +* @sjmatta made their first contribution in [PR #18250](https://github.com/BerriAI/litellm/pull/18250) +* @andres-ortizl made their first contribution in [PR #18856](https://github.com/BerriAI/litellm/pull/18856) +* @gauthiermartin made their first contribution in [PR #18844](https://github.com/BerriAI/litellm/pull/18844) +* @mel2oo made their first contribution in [PR #18845](https://github.com/BerriAI/litellm/pull/18845) +* @DominikHallab made their first contribution in [PR #18846](https://github.com/BerriAI/litellm/pull/18846) +* @ji-chuan-che made their first contribution in [PR #18540](https://github.com/BerriAI/litellm/pull/18540) +* @raghav-stripe made their first contribution in [PR #18858](https://github.com/BerriAI/litellm/pull/18858) +* @akraines made their first contribution in [PR #18629](https://github.com/BerriAI/litellm/pull/18629) +* @otaviofbrito made their first contribution in [PR #18665](https://github.com/BerriAI/litellm/pull/18665) +* @chetanchoudhary-sumo made their first contribution in [PR #18587](https://github.com/BerriAI/litellm/pull/18587) +* @pascalwhoop made their first contribution in [PR #13328](https://github.com/BerriAI/litellm/pull/13328) +* @orgersh92 made their first contribution in [PR #18652](https://github.com/BerriAI/litellm/pull/18652) +* @DevajMody made their first contribution in [PR #18497](https://github.com/BerriAI/litellm/pull/18497) +* @matt-greathouse made their first contribution in [PR #18247](https://github.com/BerriAI/litellm/pull/18247) +* @emerzon made their first contribution in [PR #18290](https://github.com/BerriAI/litellm/pull/18290) +* @Eric84626 made their first contribution in [PR #18281](https://github.com/BerriAI/litellm/pull/18281) +* @LukasdeBoer made their first contribution in [PR #18055](https://github.com/BerriAI/litellm/pull/18055) +* @LingXuanYin made their first contribution in [PR #18513](https://github.com/BerriAI/litellm/pull/18513) +* @krisxia0506 made their first contribution in [PR #18698](https://github.com/BerriAI/litellm/pull/18698) +* @LouisShark made their first contribution in [PR #18414](https://github.com/BerriAI/litellm/pull/18414) + +--- + +## Full Changelog + +**[View complete changelog on GitHub](https://github.com/BerriAI/litellm/compare/v1.80.11.rc.1...v1.80.14.rc.1)** + + diff --git a/litellm/files/main.py b/litellm/files/main.py index a7c82290c29..913ec84626d 100644 --- a/litellm/files/main.py +++ b/litellm/files/main.py @@ -8,6 +8,7 @@ https://platform.openai.com/docs/api-reference/files import asyncio import contextvars import os +import time from functools import partial from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast @@ -60,7 +61,7 @@ async def acreate_file( file: FileTypes, purpose: Literal["assistants", "batch", "fine-tune"], expires_after: Optional[FileExpiresAfter] = None, - custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai", + custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "manus"] = "openai", extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -105,7 +106,7 @@ def create_file( file: FileTypes, purpose: Literal["assistants", "batch", "fine-tune"], expires_after: Optional[FileExpiresAfter] = None, - custom_llm_provider: Optional[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"]] = None, + custom_llm_provider: Optional[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "manus"]] = None, extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -274,7 +275,7 @@ def create_file( ) else: raise litellm.exceptions.BadRequestError( - message="LiteLLM doesn't support {} for 'create_file'. Only ['openai', 'azure', 'vertex_ai'] are supported.".format( + message="LiteLLM doesn't support {} for 'create_file'. Only ['openai', 'azure', 'vertex_ai', 'manus'] are supported.".format( custom_llm_provider ), model="n/a", @@ -293,7 +294,7 @@ def create_file( @client async def afile_retrieve( file_id: str, - custom_llm_provider: Literal["openai", "azure", "hosted_vllm"] = "openai", + custom_llm_provider: Literal["openai", "azure", "hosted_vllm", "manus"] = "openai", extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -334,7 +335,7 @@ async def afile_retrieve( @client def file_retrieve( file_id: str, - custom_llm_provider: Literal["openai", "azure", "hosted_vllm"] = "openai", + custom_llm_provider: Literal["openai", "azure", "hosted_vllm", "manus"] = "openai", extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -428,18 +429,60 @@ def file_retrieve( file_id=file_id, ) else: - raise litellm.exceptions.BadRequestError( - message="LiteLLM doesn't support {} for 'file_retrieve'. Only 'openai' and 'azure' are supported.".format( - custom_llm_provider - ), - model="n/a", - llm_provider=custom_llm_provider, - response=httpx.Response( - status_code=400, - content="Unsupported provider", - request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore - ), + # Try using provider config pattern (for Manus, Bedrock, etc.) + provider_config = ProviderConfigManager.get_provider_files_config( + model="", + provider=LlmProviders(custom_llm_provider), ) + if provider_config is not None: + litellm_params_dict = get_litellm_params(**kwargs) + litellm_params_dict["api_key"] = optional_params.api_key + litellm_params_dict["api_base"] = optional_params.api_base + + logging_obj = kwargs.get("litellm_logging_obj") + if logging_obj is None: + from litellm.litellm_core_utils.litellm_logging import ( + Logging as LiteLLMLoggingObj, + ) + logging_obj = LiteLLMLoggingObj( + model="", + messages=[], + stream=False, + call_type="afile_retrieve" if _is_async else "file_retrieve", + start_time=time.time(), + litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())), + function_id=str(kwargs.get("id") or ""), + ) + + client = kwargs.get("client") + response = base_llm_http_handler.retrieve_file( + file_id=file_id, + provider_config=provider_config, + litellm_params=litellm_params_dict, + headers=extra_headers or {}, + logging_obj=logging_obj, + _is_async=_is_async, + client=( + client + if client is not None + and isinstance(client, (HTTPHandler, AsyncHTTPHandler)) + else None + ), + timeout=timeout, + ) + else: + raise litellm.exceptions.BadRequestError( + message="LiteLLM doesn't support {} for 'file_retrieve'. Only 'openai', 'azure', and 'manus' are supported.".format( + custom_llm_provider + ), + model="n/a", + llm_provider=custom_llm_provider, + response=httpx.Response( + status_code=400, + content="Unsupported provider", + request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore + ), + ) return cast(FileObject, response) except Exception as e: @@ -450,7 +493,7 @@ def file_retrieve( @client async def afile_delete( file_id: str, - custom_llm_provider: Literal["openai", "azure"] = "openai", + custom_llm_provider: Literal["openai", "azure", "manus"] = "openai", extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -494,7 +537,7 @@ async def afile_delete( def file_delete( file_id: str, model: Optional[str] = None, - custom_llm_provider: Union[Literal["openai", "azure"], str] = "openai", + custom_llm_provider: Union[Literal["openai", "azure", "manus"], str] = "openai", extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -596,18 +639,58 @@ def file_delete( litellm_params=litellm_params_dict, ) else: - raise litellm.exceptions.BadRequestError( - message="LiteLLM doesn't support {} for 'delete_batch'. Only 'openai' is supported.".format( - custom_llm_provider - ), - model="n/a", - llm_provider=custom_llm_provider, - response=httpx.Response( - status_code=400, - content="Unsupported provider", - request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore - ), + # Try using provider config pattern (for Manus, Bedrock, etc.) + provider_config = ProviderConfigManager.get_provider_files_config( + model="", + provider=LlmProviders(custom_llm_provider), ) + if provider_config is not None: + litellm_params_dict["api_key"] = optional_params.api_key + litellm_params_dict["api_base"] = optional_params.api_base + + logging_obj = kwargs.get("litellm_logging_obj") + if logging_obj is None: + from litellm.litellm_core_utils.litellm_logging import ( + Logging as LiteLLMLoggingObj, + ) + logging_obj = LiteLLMLoggingObj( + model="", + messages=[], + stream=False, + call_type="afile_delete" if _is_async else "file_delete", + start_time=time.time(), + litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())), + function_id=str(kwargs.get("id") or ""), + ) + + response = base_llm_http_handler.delete_file( + file_id=file_id, + provider_config=provider_config, + litellm_params=litellm_params_dict, + headers=extra_headers or {}, + logging_obj=logging_obj, + _is_async=_is_async, + client=( + client + if client is not None + and isinstance(client, (HTTPHandler, AsyncHTTPHandler)) + else None + ), + timeout=timeout, + ) + else: + raise litellm.exceptions.BadRequestError( + message="LiteLLM doesn't support {} for 'file_delete'. Only 'openai', 'azure', and 'manus' are supported.".format( + custom_llm_provider + ), + model="n/a", + llm_provider=custom_llm_provider, + response=httpx.Response( + status_code=400, + content="Unsupported provider", + request=httpx.Request(method="create_thread", url="https://github.com/BerriAI/litellm"), # type: ignore + ), + ) return cast(FileDeleted, response) except Exception as e: raise e @@ -616,7 +699,7 @@ def file_delete( # List files @client async def afile_list( - custom_llm_provider: Literal["openai", "azure"] = "openai", + custom_llm_provider: Literal["openai", "azure", "manus"] = "openai", purpose: Optional[str] = None, extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, @@ -657,7 +740,7 @@ async def afile_list( @client def file_list( - custom_llm_provider: Literal["openai", "azure"] = "openai", + custom_llm_provider: Literal["openai", "azure", "manus"] = "openai", purpose: Optional[str] = None, extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, @@ -687,7 +770,50 @@ def file_list( timeout = 600.0 _is_async = kwargs.pop("is_async", False) is True - if custom_llm_provider in OPENAI_COMPATIBLE_BATCH_AND_FILES_PROVIDERS: + + # Check if provider has a custom files config (e.g., Manus, Bedrock, Vertex AI) + provider_config = ProviderConfigManager.get_provider_files_config( + model="", + provider=LlmProviders(custom_llm_provider), + ) + if provider_config is not None: + litellm_params_dict = get_litellm_params(**kwargs) + litellm_params_dict["api_key"] = optional_params.api_key + litellm_params_dict["api_base"] = optional_params.api_base + + logging_obj = kwargs.get("litellm_logging_obj") + if logging_obj is None: + from litellm.litellm_core_utils.litellm_logging import ( + Logging as LiteLLMLoggingObj, + ) + logging_obj = LiteLLMLoggingObj( + model="", + messages=[], + stream=False, + call_type="afile_list" if _is_async else "file_list", + start_time=time.time(), + litellm_call_id=kwargs.get("litellm_call_id", str(uuid.uuid4())), + function_id=str(kwargs.get("id", "")), + ) + + client = kwargs.get("client") + response = base_llm_http_handler.list_files( + purpose=purpose, + provider_config=provider_config, + litellm_params=litellm_params_dict, + headers=extra_headers or {}, + logging_obj=logging_obj, + _is_async=_is_async, + client=( + client + if client is not None + and isinstance(client, (HTTPHandler, AsyncHTTPHandler)) + else None + ), + timeout=timeout, + ) + return response + elif custom_llm_provider in OPENAI_COMPATIBLE_BATCH_AND_FILES_PROVIDERS: # for deepinfra/perplexity/anyscale/groq we check in get_llm_provider and pass in the api base from there api_base = ( optional_params.api_base @@ -752,7 +878,7 @@ def file_list( ) else: raise litellm.exceptions.BadRequestError( - message="LiteLLM doesn't support {} for 'file_list'. Only 'openai' and 'azure' are supported.".format( + message="LiteLLM doesn't support {} for 'file_list'. Only 'openai', 'azure', and 'manus' are supported.".format( custom_llm_provider ), model="n/a", @@ -771,7 +897,7 @@ def file_list( @client async def afile_content( file_id: str, - custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic"] = "openai", + custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic", "manus"] = "openai", extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -816,7 +942,7 @@ def file_content( file_id: str, model: Optional[str] = None, custom_llm_provider: Optional[ - Union[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic"], str] + Union[Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm", "anthropic", "manus"], str] ] = None, extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, @@ -977,7 +1103,7 @@ def file_content( ) else: raise litellm.exceptions.BadRequestError( - message="LiteLLM doesn't support {} for 'custom_llm_provider'. Supported providers are 'openai', 'azure', 'vertex_ai', 'bedrock'.".format( + message="LiteLLM doesn't support {} for 'file_content'. Supported providers are 'openai', 'azure', 'vertex_ai', 'bedrock', 'manus'.".format( custom_llm_provider ), model="n/a", diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index c32a7b75c51..bb9d6166b01 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -875,16 +875,7 @@ class PrometheusLogger(CustomLogger): # Include top-level metadata fields (excluding nested dictionaries) # This allows accessing fields like requester_ip_address from top-level metadata top_level_metadata = standard_logging_payload.get("metadata", {}) - top_level_fields: Dict[str, Any] = {} - if isinstance(top_level_metadata, dict): - top_level_fields = { - k: v - for k, v in top_level_metadata.items() - if not isinstance(v, dict) # Exclude nested dicts to avoid conflicts - } - combined_metadata: Dict[str, Any] = { - **top_level_fields, # Include top-level fields first **(_requester_metadata if _requester_metadata else {}), **(user_api_key_auth_metadata if user_api_key_auth_metadata else {}), } diff --git a/litellm/llms/base_llm/files/transformation.py b/litellm/llms/base_llm/files/transformation.py index 35b76479cdc..7b0a1868f19 100644 --- a/litellm/llms/base_llm/files/transformation.py +++ b/litellm/llms/base_llm/files/transformation.py @@ -2,11 +2,14 @@ from abc import ABC, abstractmethod from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union import httpx +from openai.types.file_deleted import FileDeleted from litellm.proxy._types import UserAPIKeyAuth +from litellm.types.files import TwoStepFileUploadConfig from litellm.types.llms.openai import ( AllMessageValues, CreateFileRequest, + FileContentRequest, OpenAICreateFileRequestOptionalParams, OpenAIFileObject, OpenAIFilesPurpose, @@ -75,7 +78,15 @@ class BaseFilesConfig(BaseConfig): create_file_data: CreateFileRequest, optional_params: dict, litellm_params: dict, - ) -> Union[dict, str, bytes]: + ) -> Union[dict, str, bytes, "TwoStepFileUploadConfig"]: + """ + Transform OpenAI-style file creation request into provider-specific format. + + Returns: + - dict: For pre-signed single-step uploads (e.g., Bedrock S3) + - str/bytes: For traditional file uploads + - TwoStepFileUploadConfig: For two-step upload process (e.g., Manus, GCS) + """ pass @abstractmethod @@ -88,6 +99,86 @@ class BaseFilesConfig(BaseConfig): ) -> OpenAIFileObject: pass + @abstractmethod + def transform_retrieve_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Transform file retrieve request into provider-specific format.""" + pass + + @abstractmethod + def transform_retrieve_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> OpenAIFileObject: + """Transform file retrieve response into OpenAI format.""" + pass + + @abstractmethod + def transform_delete_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Transform file delete request into provider-specific format.""" + pass + + @abstractmethod + def transform_delete_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> "FileDeleted": + """Transform file delete response into OpenAI format.""" + pass + + @abstractmethod + def transform_list_files_request( + self, + purpose: Optional[str], + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Transform file list request into provider-specific format.""" + pass + + @abstractmethod + def transform_list_files_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> List[OpenAIFileObject]: + """Transform file list response into OpenAI format.""" + pass + + @abstractmethod + def transform_file_content_request( + self, + file_content_request: "FileContentRequest", + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Transform file content request into provider-specific format.""" + pass + + @abstractmethod + def transform_file_content_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> "HttpxBinaryResponseContent": + """Transform file content response into OpenAI format.""" + pass + def transform_request( self, model: str, diff --git a/litellm/llms/bedrock/base_aws_llm.py b/litellm/llms/bedrock/base_aws_llm.py index 18e9deb53b0..e9cea23ea4a 100644 --- a/litellm/llms/bedrock/base_aws_llm.py +++ b/litellm/llms/bedrock/base_aws_llm.py @@ -74,41 +74,6 @@ class BaseAWSLLM: "aws_external_id", ] - def _get_ssl_verify(self): - """ - Get SSL verification setting for boto3 clients. - - This ensures that custom CA certificates are properly used for all AWS API calls, - including STS and Bedrock services. - - Returns: - Union[bool, str]: SSL verification setting - False to disable, True to enable, - or a string path to a CA bundle file - """ - import litellm - from litellm.secret_managers.main import str_to_bool - - # Check environment variable first (highest priority) - ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify) - - # Convert string "False"/"True" to boolean - if isinstance(ssl_verify, str): - # Check if it's a file path - if os.path.exists(ssl_verify): - return ssl_verify - # Otherwise try to convert to boolean - ssl_verify_bool = str_to_bool(ssl_verify) - if ssl_verify_bool is not None: - ssl_verify = ssl_verify_bool - - # Check SSL_CERT_FILE environment variable for custom CA bundle - if ssl_verify is True or ssl_verify == "True": - ssl_cert_file = os.getenv("SSL_CERT_FILE") - if ssl_cert_file and os.path.exists(ssl_cert_file): - return ssl_cert_file - - return ssl_verify - def get_cache_key(self, credential_args: Dict[str, Optional[str]]) -> str: """ Generate a unique cache key based on the credential arguments. @@ -604,7 +569,6 @@ class BaseAWSLLM: "sts", region_name=aws_region_name, endpoint_url=sts_endpoint, - verify=self._get_ssl_verify(), ) # https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html @@ -661,7 +625,7 @@ class BaseAWSLLM: # Create an STS client without credentials with tracer.trace("boto3.client(sts) for manual IRSA"): - sts_client = boto3.client("sts", region_name=region, verify=self._get_ssl_verify()) + sts_client = boto3.client("sts", region_name=region) # Manually assume the IRSA role with the session name verbose_logger.debug( @@ -684,7 +648,6 @@ class BaseAWSLLM: aws_access_key_id=irsa_creds["AccessKeyId"], aws_secret_access_key=irsa_creds["SecretAccessKey"], aws_session_token=irsa_creds["SessionToken"], - verify=self._get_ssl_verify(), ) # Get current caller identity for debugging @@ -723,7 +686,7 @@ class BaseAWSLLM: verbose_logger.debug("Same account role assumption, using automatic IRSA") with tracer.trace("boto3.client(sts) with automatic IRSA"): - sts_client = boto3.client("sts", region_name=region, verify=self._get_ssl_verify()) + sts_client = boto3.client("sts", region_name=region) # Get current caller identity for debugging try: @@ -846,7 +809,7 @@ class BaseAWSLLM: # This allows the web identity token to work automatically if aws_access_key_id is None and aws_secret_access_key is None: with tracer.trace("boto3.client(sts)"): - sts_client = boto3.client("sts", verify=self._get_ssl_verify()) + sts_client = boto3.client("sts") else: with tracer.trace("boto3.client(sts)"): sts_client = boto3.client( @@ -854,7 +817,6 @@ class BaseAWSLLM: aws_access_key_id=aws_access_key_id, aws_secret_access_key=aws_secret_access_key, aws_session_token=aws_session_token, - verify=self._get_ssl_verify(), ) assume_role_params = { diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index f4b5de8f7c0..d62a8bae425 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -260,7 +260,7 @@ def init_bedrock_client( status_code=401, ) - sts_client = boto3.client("sts", verify=ssl_verify) + sts_client = boto3.client("sts") # https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html # https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts/client/assume_role_with_web_identity.html diff --git a/litellm/llms/bedrock/files/handler.py b/litellm/llms/bedrock/files/handler.py index 0350271dc44..d6177e090d5 100644 --- a/litellm/llms/bedrock/files/handler.py +++ b/litellm/llms/bedrock/files/handler.py @@ -142,7 +142,6 @@ class BedrockFilesHandler(BaseAWSLLM): aws_secret_access_key=credentials.secret_key, aws_session_token=credentials.token, region_name=aws_region_name, - verify=self._get_ssl_verify(), ) # Download file from S3 diff --git a/litellm/llms/bedrock/files/transformation.py b/litellm/llms/bedrock/files/transformation.py index 0a95cf9168f..fdcbe1a8242 100644 --- a/litellm/llms/bedrock/files/transformation.py +++ b/litellm/llms/bedrock/files/transformation.py @@ -1,12 +1,14 @@ import json import os import time -from litellm._uuid import uuid from typing import Any, Dict, List, Optional, Tuple, Union +import httpx from httpx import Headers, Response +from openai.types.file_deleted import FileDeleted from litellm._logging import verbose_logger +from litellm._uuid import uuid from litellm.files.utils import FilesAPIUtils from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data from litellm.llms.base_llm.chat.transformation import BaseLLMException @@ -18,6 +20,7 @@ from litellm.types.llms.openai import ( AllMessageValues, CreateFileRequest, FileTypes, + HttpxBinaryResponseContent, OpenAICreateFileRequestOptionalParams, OpenAIFileObject, PathLike, @@ -539,6 +542,70 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig): status_code=status_code, message=error_message, headers=headers ) + def transform_retrieve_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("BedrockFilesConfig does not support file retrieval") + + def transform_retrieve_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> OpenAIFileObject: + raise NotImplementedError("BedrockFilesConfig does not support file retrieval") + + def transform_delete_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("BedrockFilesConfig does not support file deletion") + + def transform_delete_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> FileDeleted: + raise NotImplementedError("BedrockFilesConfig does not support file deletion") + + def transform_list_files_request( + self, + purpose: Optional[str], + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("BedrockFilesConfig does not support file listing") + + def transform_list_files_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> List[OpenAIFileObject]: + raise NotImplementedError("BedrockFilesConfig does not support file listing") + + def transform_file_content_request( + self, + file_content_request, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("BedrockFilesConfig does not support file content retrieval") + + def transform_file_content_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> HttpxBinaryResponseContent: + raise NotImplementedError("BedrockFilesConfig does not support file content retrieval") + class BedrockJsonlFilesTransformation: """ diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index ea740400664..1da6e61252f 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -14,6 +14,7 @@ from typing import ( ) import httpx # type: ignore +from openai.types.file_deleted import FileDeleted import litellm import litellm.litellm_core_utils @@ -71,6 +72,7 @@ from litellm.types.containers.main import ( ContainerObject, DeleteContainerResult, ) +from litellm.types.files import TwoStepFileUploadConfig from litellm.types.llms.anthropic_messages.anthropic_response import ( AnthropicMessagesResponse, ) @@ -82,6 +84,7 @@ from litellm.types.llms.anthropic_skills import ( from litellm.types.llms.openai import ( CreateBatchRequest, CreateFileRequest, + FileContentRequest, HttpxBinaryResponseContent, OpenAIFileObject, ResponseInputParam, @@ -2782,6 +2785,38 @@ class BaseLLMHTTPHandler: logging_obj=logging_obj, ) + def _extract_upload_url_from_response( + self, + response: httpx.Response, + upload_url_location: str, + upload_url_key: str = "upload_url", + ) -> tuple[Optional[str], Optional[dict]]: + """ + Extract upload URL from initial file creation response. + + Args: + response: HTTP response from initial file creation request + upload_url_location: Where to find URL ('headers' or 'body') + upload_url_key: Key name for URL in response body (default: 'upload_url') + + Returns: + Tuple of (upload_url, response_data) + - upload_url: The extracted upload URL, or None if not found + - response_data: Parsed response body (for 'body' location), or None + """ + if upload_url_location == "headers": + # Google Cloud Storage style - URL in X-Goog-Upload-URL header + upload_url = response.headers.get("X-Goog-Upload-URL") + return upload_url, None + else: + # Response body style (e.g., Manus, S3 presigned URLs) + try: + response_data = response.json() + upload_url = response_data.get(upload_url_key) + return upload_url, response_data if upload_url else None + except Exception: + return None, None + def create_file( self, create_file_data: CreateFileRequest, @@ -2844,14 +2879,58 @@ class BaseLLMHTTPHandler: else: sync_httpx_client = client - if isinstance(transformed_request, dict) and "method" in transformed_request: + if isinstance(transformed_request, dict) and "initial_request" in transformed_request: + # Handle two-step uploads (TwoStepFileUploadConfig) + # Used by providers like Manus, Google Cloud Storage + try: + # Step 1: Initial request to get upload URL + initial_response = sync_httpx_client.post( + url=api_base, + headers={ + **headers, + **transformed_request["initial_request"]["headers"], + }, + data=json.dumps(transformed_request["initial_request"]["data"]), + timeout=timeout, + ) + + # Extract upload URL from response + upload_url, initial_response_data = self._extract_upload_url_from_response( + response=initial_response, + upload_url_location=transformed_request.get("upload_url_location", "headers"), + upload_url_key=transformed_request.get("upload_url_key", "upload_url"), + ) + + if not upload_url: + raise ValueError("Failed to get upload URL from initial request") + + # Step 2: Upload the actual file + upload_method = transformed_request["upload_request"].get("method", "POST").lower() + upload_response = getattr(sync_httpx_client, upload_method)( + url=upload_url, + headers=transformed_request["upload_request"]["headers"], + data=transformed_request["upload_request"]["data"], + timeout=timeout, + ) + + # Store initial response for transformation + if initial_response_data: + litellm_params["initial_file_response"] = initial_response_data + except Exception as e: + raise self._handle_error( + e=e, + provider_config=provider_config, + ) + elif isinstance(transformed_request, dict) and "method" in transformed_request and "initial_request" not in transformed_request: # Handle pre-signed requests (e.g., from Bedrock S3 uploads) + # Type narrowing: this is a plain dict, not TwoStepFileUploadConfig + presigned_request = cast(Dict[str, Any], transformed_request) upload_response = getattr( - sync_httpx_client, transformed_request["method"].lower() + sync_httpx_client, presigned_request["method"].lower() )( - url=transformed_request["url"], - headers=transformed_request["headers"], - data=transformed_request["data"], + url=presigned_request["url"], + headers=presigned_request["headers"], + data=presigned_request["data"], timeout=timeout, ) elif isinstance(transformed_request, str) or isinstance( @@ -2879,36 +2958,7 @@ class BaseLLMHTTPHandler: timeout=timeout, ) else: - try: - # Step 1: Initial request to get upload URL - initial_response = sync_httpx_client.post( - url=api_base, - headers={ - **headers, - **transformed_request["initial_request"]["headers"], - }, - data=json.dumps(transformed_request["initial_request"]["data"]), - timeout=timeout, - ) - - # Extract upload URL from response headers - upload_url = initial_response.headers.get("X-Goog-Upload-URL") - - if not upload_url: - raise ValueError("Failed to get upload URL from initial request") - - # Step 2: Upload the actual file - upload_response = sync_httpx_client.post( - url=upload_url, - headers=transformed_request["upload_request"]["headers"], - data=transformed_request["upload_request"]["data"], - timeout=timeout, - ) - except Exception as e: - raise self._handle_error( - e=e, - provider_config=provider_config, - ) + raise ValueError(f"Unsupported transformed_request type: {type(transformed_request)}") # Store the upload URL in litellm_params for the transformation method litellm_params_with_url = dict(litellm_params) @@ -2923,7 +2973,7 @@ class BaseLLMHTTPHandler: async def async_create_file( self, - transformed_request: Union[bytes, str, dict], + transformed_request: Union[bytes, str, dict, "TwoStepFileUploadConfig"], litellm_params: dict, provider_config: BaseFilesConfig, headers: dict, @@ -2955,14 +3005,59 @@ class BaseLLMHTTPHandler: }, ) - if isinstance(transformed_request, dict) and "method" in transformed_request: + if isinstance(transformed_request, dict) and "initial_request" in transformed_request: + # Handle two-step uploads (TwoStepFileUploadConfig) + # Used by providers like Manus, Google Cloud Storage + try: + # Step 1: Initial request to get upload URL + initial_response = await async_httpx_client.post( + url=api_base, + headers={ + **headers, + **transformed_request["initial_request"]["headers"], + }, + data=json.dumps(transformed_request["initial_request"]["data"]), + timeout=timeout, + ) + + # Extract upload URL from response + upload_url, initial_response_data = self._extract_upload_url_from_response( + response=initial_response, + upload_url_location=transformed_request.get("upload_url_location", "headers"), + upload_url_key=transformed_request.get("upload_url_key", "upload_url"), + ) + + if not upload_url: + raise ValueError("Failed to get upload URL from initial request") + + # Step 2: Upload the actual file + upload_method = transformed_request["upload_request"].get("method", "POST").lower() + upload_response = await getattr(async_httpx_client, upload_method)( + url=upload_url, + headers=transformed_request["upload_request"]["headers"], + data=transformed_request["upload_request"]["data"], + timeout=timeout, + ) + + # Store initial response for transformation + if initial_response_data: + litellm_params["initial_file_response"] = initial_response_data + except Exception as e: + verbose_logger.exception(f"Error creating file: {e}") + raise self._handle_error( + e=e, + provider_config=provider_config, + ) + elif isinstance(transformed_request, dict) and "method" in transformed_request and "initial_request" not in transformed_request: # Handle pre-signed requests (e.g., from Bedrock S3 uploads) + # Type narrowing: this is a plain dict, not TwoStepFileUploadConfig + presigned_request = cast(Dict[str, Any], transformed_request) upload_response = await getattr( - async_httpx_client, transformed_request["method"].lower() + async_httpx_client, presigned_request["method"].lower() )( - url=transformed_request["url"], - headers=transformed_request["headers"], - data=transformed_request["data"], + url=presigned_request["url"], + headers=presigned_request["headers"], + data=presigned_request["data"], timeout=timeout, ) elif isinstance(transformed_request, str) or isinstance( @@ -2990,37 +3085,7 @@ class BaseLLMHTTPHandler: timeout=timeout, ) else: - try: - # Step 1: Initial request to get upload URL - initial_response = await async_httpx_client.post( - url=api_base, - headers={ - **headers, - **transformed_request["initial_request"]["headers"], - }, - data=json.dumps(transformed_request["initial_request"]["data"]), - timeout=timeout, - ) - - # Extract upload URL from response headers - upload_url = initial_response.headers.get("X-Goog-Upload-URL") - - if not upload_url: - raise ValueError("Failed to get upload URL from initial request") - - # Step 2: Upload the actual file - upload_response = await async_httpx_client.post( - url=upload_url, - headers=transformed_request["upload_request"]["headers"], - data=transformed_request["upload_request"]["data"], - timeout=timeout, - ) - except Exception as e: - verbose_logger.exception(f"Error creating file: {e}") - raise self._handle_error( - e=e, - provider_config=provider_config, - ) + raise ValueError(f"Unsupported transformed_request type: {type(transformed_request)}") return provider_config.transform_create_file_response( model=None, @@ -3734,29 +3799,525 @@ class BaseLLMHTTPHandler: logging_obj=logging_obj, ) - def list_files(self): + def retrieve_file( + self, + file_id: str, + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + _is_async: bool = False, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> Union[OpenAIFileObject, Coroutine[Any, Any, OpenAIFileObject]]: """ - Lists all files + Retrieve file metadata by ID """ - pass + if _is_async: + return self.async_retrieve_file( + file_id=file_id, + provider_config=provider_config, + litellm_params=litellm_params, + headers=headers, + logging_obj=logging_obj, + client=client, + timeout=timeout, + ) - def delete_file(self): - """ - Deletes a file - """ - pass + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client - def retrieve_file(self): - """ - Returns the metadata of the file - """ - pass + # Get URL and params from provider config + url, params = provider_config.transform_retrieve_file_request( + file_id=file_id, + optional_params={}, + litellm_params=litellm_params, + ) - def retrieve_file_content(self): + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "file_id": file_id, + }, + ) + + try: + response = sync_httpx_client.get( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_retrieve_file_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + async def async_retrieve_file( + self, + file_id: str, + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> OpenAIFileObject: """ - Returns the content of the file + Async retrieve file metadata by ID """ - pass + if client is None or not isinstance(client, AsyncHTTPHandler): + async_httpx_client = get_async_httpx_client( + llm_provider=provider_config.custom_llm_provider + ) + else: + async_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_retrieve_file_request( + file_id=file_id, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "file_id": file_id, + }, + ) + + try: + response = await async_httpx_client.get( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_retrieve_file_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + def delete_file( + self, + file_id: str, + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + _is_async: bool = False, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> Union["FileDeleted", Coroutine[Any, Any, "FileDeleted"]]: + """ + Delete a file by ID + """ + if _is_async: + return self.async_delete_file( + file_id=file_id, + provider_config=provider_config, + litellm_params=litellm_params, + headers=headers, + logging_obj=logging_obj, + client=client, + timeout=timeout, + ) + + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_delete_file_request( + file_id=file_id, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "file_id": file_id, + }, + ) + + try: + response = sync_httpx_client.delete( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_delete_file_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + async def async_delete_file( + self, + file_id: str, + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> "FileDeleted": + """ + Async delete a file by ID + """ + if client is None or not isinstance(client, AsyncHTTPHandler): + async_httpx_client = get_async_httpx_client( + llm_provider=provider_config.custom_llm_provider + ) + else: + async_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_delete_file_request( + file_id=file_id, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "file_id": file_id, + }, + ) + + try: + response = await async_httpx_client.delete( + url=url, headers=headers, params=params, timeout=timeout + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_delete_file_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + def list_files( + self, + purpose: Optional[str], + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + _is_async: bool = False, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> Union[List[OpenAIFileObject], Coroutine[Any, Any, List[OpenAIFileObject]]]: + """ + List all files + """ + if _is_async: + return self.async_list_files( + purpose=purpose, + provider_config=provider_config, + litellm_params=litellm_params, + headers=headers, + logging_obj=logging_obj, + client=client, + timeout=timeout, + ) + + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_list_files_request( + purpose=purpose, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "purpose": purpose, + }, + ) + + try: + response = sync_httpx_client.get( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_list_files_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + async def async_list_files( + self, + purpose: Optional[str], + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> List[OpenAIFileObject]: + """ + Async list all files + """ + if client is None or not isinstance(client, AsyncHTTPHandler): + async_httpx_client = get_async_httpx_client( + llm_provider=provider_config.custom_llm_provider + ) + else: + async_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_list_files_request( + purpose=purpose, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "purpose": purpose, + }, + ) + + try: + response = await async_httpx_client.get( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_list_files_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + def retrieve_file_content( + self, + file_content_request: "FileContentRequest", + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + _is_async: bool = False, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> Union["HttpxBinaryResponseContent", Coroutine[Any, Any, "HttpxBinaryResponseContent"]]: + """ + Retrieve file content by ID + """ + if _is_async: + return self.async_retrieve_file_content( + file_content_request=file_content_request, + provider_config=provider_config, + litellm_params=litellm_params, + headers=headers, + logging_obj=logging_obj, + client=client, + timeout=timeout, + ) + + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_file_content_request( + file_content_request=file_content_request, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "file_id": file_content_request.get("file_id"), + }, + ) + + try: + response = sync_httpx_client.get( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_file_content_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + async def async_retrieve_file_content( + self, + file_content_request: "FileContentRequest", + provider_config: BaseFilesConfig, + litellm_params: dict, + headers: dict, + logging_obj: LiteLLMLoggingObj, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + timeout: Optional[Union[float, httpx.Timeout]] = None, + ) -> "HttpxBinaryResponseContent": + """ + Async retrieve file content by ID + """ + if client is None or not isinstance(client, AsyncHTTPHandler): + async_httpx_client = get_async_httpx_client( + llm_provider=provider_config.custom_llm_provider + ) + else: + async_httpx_client = client + + # Get URL and params from provider config + url, params = provider_config.transform_file_content_request( + file_content_request=file_content_request, + optional_params={}, + litellm_params=litellm_params, + ) + + # Validate environment and get headers + headers = provider_config.validate_environment( + api_key=litellm_params.get("api_key"), + headers=headers, + model="", + messages=[], + optional_params={}, + litellm_params=litellm_params, + ) + + logging_obj.pre_call( + input="", + api_key="", + additional_args={ + "api_base": url, + "headers": headers, + "file_id": file_content_request.get("file_id"), + }, + ) + + try: + response = await async_httpx_client.get( + url=url, headers=headers, params=params + ) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + + return provider_config.transform_file_content_response( + raw_response=response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) def _prepare_fake_stream_request( self, diff --git a/litellm/llms/gemini/common_utils.py b/litellm/llms/gemini/common_utils.py index 30c5b4f17c5..e53829d3329 100644 --- a/litellm/llms/gemini/common_utils.py +++ b/litellm/llms/gemini/common_utils.py @@ -150,15 +150,6 @@ def get_api_key_from_env() -> Optional[str]: return get_secret_str("GOOGLE_API_KEY") or get_secret_str("GEMINI_API_KEY") -def get_vertex_api_key_from_env() -> Optional[str]: - """ - Get API key from environment for Vertex AI. - Checks VERTEXAI_API_KEY and VERTEX_API_KEY environment variables. - This allows using Vertex AI with API keys instead of service account credentials. - """ - return get_secret_str("VERTEXAI_API_KEY") or get_secret_str("VERTEX_API_KEY") - - class GoogleAIStudioTokenCounter(BaseTokenCounter): """Token counter implementation for Google AI Studio provider.""" def should_use_token_counting_api( diff --git a/litellm/llms/gemini/files/transformation.py b/litellm/llms/gemini/files/transformation.py index e98e76dabc8..d9ebf69a97a 100644 --- a/litellm/llms/gemini/files/transformation.py +++ b/litellm/llms/gemini/files/transformation.py @@ -7,6 +7,7 @@ import time from typing import List, Optional import httpx +from openai.types.file_deleted import FileDeleted from litellm._logging import verbose_logger from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data @@ -17,6 +18,7 @@ from litellm.llms.base_llm.files.transformation import ( from litellm.types.llms.gemini import GeminiCreateFilesResponseObject from litellm.types.llms.openai import ( CreateFileRequest, + HttpxBinaryResponseContent, OpenAICreateFileRequestOptionalParams, OpenAIFileObject, ) @@ -171,3 +173,67 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig): except Exception as e: verbose_logger.exception(f"Error parsing file upload response: {str(e)}") raise ValueError(f"Error parsing file upload response: {str(e)}") + + def transform_retrieve_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file retrieval") + + def transform_retrieve_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> OpenAIFileObject: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file retrieval") + + def transform_delete_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file deletion") + + def transform_delete_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> FileDeleted: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file deletion") + + def transform_list_files_request( + self, + purpose: Optional[str], + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file listing") + + def transform_list_files_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> List[OpenAIFileObject]: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file listing") + + def transform_file_content_request( + self, + file_content_request, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file content retrieval") + + def transform_file_content_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> HttpxBinaryResponseContent: + raise NotImplementedError("GoogleAIStudioFilesHandler does not support file content retrieval") diff --git a/litellm/llms/manus/files/__init__.py b/litellm/llms/manus/files/__init__.py new file mode 100644 index 00000000000..66d23ca0340 --- /dev/null +++ b/litellm/llms/manus/files/__init__.py @@ -0,0 +1,2 @@ +# Manus Files API implementation + diff --git a/litellm/llms/manus/files/transformation.py b/litellm/llms/manus/files/transformation.py new file mode 100644 index 00000000000..a7965011969 --- /dev/null +++ b/litellm/llms/manus/files/transformation.py @@ -0,0 +1,439 @@ +""" +Manus Files API implementation. + +Manus has an OpenAI-compatible Files API with some differences: +- Uses API_KEY header instead of Authorization: Bearer +- File upload is a two-step process: + 1. Create file record to get upload URL + 2. Upload file content to the upload URL + +Reference: https://open.manus.im/docs/openai-compatibility#file-management +""" + +import time +from typing import Any, Dict, List, Optional, Union + +import httpx +from openai.types.file_deleted import FileDeleted + +import litellm +from litellm._logging import verbose_logger +from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.files.transformation import ( + BaseFilesConfig, + LiteLLMLoggingObj, +) +from litellm.llms.openai.common_utils import OpenAIError +from litellm.secret_managers.main import get_secret_str +from litellm.types.files import TwoStepFileUploadConfig, TwoStepFileUploadRequest +from litellm.types.llms.openai import ( + CreateFileRequest, + FileContentRequest, + HttpxBinaryResponseContent, + OpenAICreateFileRequestOptionalParams, + OpenAIFileObject, +) +from litellm.types.utils import LlmProviders + +MANUS_API_BASE = "https://api.manus.im" + + +class ManusFilesConfig(BaseFilesConfig): + """ + Configuration for Manus Files API. + + Manus uses: + - API_KEY header for authentication (not Authorization: Bearer) + - Two-step file upload process + - Content-Type: application/json for all requests + + Reference: https://open.manus.im/docs/openai-compatibility#file-management + """ + + def __init__(self): + pass + + @property + def custom_llm_provider(self) -> LlmProviders: + return LlmProviders.MANUS + + def validate_environment( + self, + headers: dict, + model: str, + messages: list, + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + """ + Validate environment and set up headers for Manus API. + + Manus uses API_KEY header instead of Authorization: Bearer. + For file uploads, don't set Content-Type - httpx will set it for multipart. + """ + api_key = ( + api_key + or litellm.api_key + or get_secret_str("MANUS_API_KEY") + ) + + if not api_key: + raise ValueError( + "Manus API key is required. Set MANUS_API_KEY environment variable or pass api_key parameter." + ) + + # Manus uses API_KEY header, not Authorization: Bearer + # Manus requires Content-Type: application/json for all requests (even GET) + headers.update( + { + "API_KEY": api_key, + "Content-Type": "application/json", + } + ) + return headers + + def get_supported_openai_params( + self, model: str + ) -> List[OpenAICreateFileRequestOptionalParams]: + """ + Return supported OpenAI file creation parameters for Manus. + Manus supports the standard 'purpose' parameter. + """ + return ["purpose"] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + """ + Map OpenAI parameters to Manus-specific parameters. + Manus is OpenAI-compatible, so no special mapping needed. + """ + return optional_params + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + Get the complete URL for Manus Files API endpoint. + + Returns: + str: The full URL for the Manus /v1/files endpoint + """ + api_base = ( + api_base + or litellm.api_base + or get_secret_str("MANUS_API_BASE") + or MANUS_API_BASE + ) + + # Remove trailing slashes + api_base = api_base.rstrip("/") + + # Manus API uses /v1/files endpoint + if api_base.endswith("/v1"): + return f"{api_base}/files" + return f"{api_base}/v1/files" + + def get_error_class( + self, + error_message: str, + status_code: int, + headers: Union[dict, httpx.Headers], + ) -> BaseLLMException: + """ + Return the appropriate error class for Manus API errors. + Uses OpenAIError since Manus is OpenAI-compatible. + """ + return OpenAIError( + status_code=status_code, + message=error_message, + headers=headers, + ) + + def transform_create_file_request( + self, + model: str, + create_file_data: CreateFileRequest, + optional_params: dict, + litellm_params: dict, + ) -> TwoStepFileUploadConfig: + """ + Transform OpenAI-style file creation request into Manus's two-step format. + + Manus API spec (https://open.manus.im/docs/openai-compatibility#file-management): + 1. POST /v1/files with JSON {"filename": "..."} → returns {"id": "...", "upload_url": "..."} + 2. PUT to upload_url with raw file content + """ + # Extract file data + file_data = create_file_data.get("file") + if file_data is None: + raise ValueError("File data is required") + + extracted_data = extract_file_data(file_data) + filename = extracted_data["filename"] or f"file_{int(time.time())}" + content = extracted_data["content"] + + # Get API base URL + api_base = self.get_complete_url( + api_base=litellm_params.get("api_base"), + api_key=litellm_params.get("api_key"), + model=model, + optional_params=optional_params, + litellm_params=litellm_params, + ) + + # Get API key + api_key = ( + litellm_params.get("api_key") + or litellm.api_key + or get_secret_str("MANUS_API_KEY") + ) + + if not api_key: + raise ValueError( + "Manus API key is required. Set MANUS_API_KEY environment variable or pass api_key parameter." + ) + + # Build typed two-step upload config + return TwoStepFileUploadConfig( + initial_request=TwoStepFileUploadRequest( + method="POST", + url=api_base, + headers={ + "API_KEY": api_key, + "Content-Type": "application/json", + }, + data={"filename": filename}, + ), + upload_request=TwoStepFileUploadRequest( + method="PUT", + url="", # Will be populated from initial_request response + headers={}, + data=content, + ), + upload_url_location="body", + upload_url_key="upload_url", + ) + + def transform_create_file_response( + self, + model: Optional[str], + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> OpenAIFileObject: + """ + Transform Manus's file upload response into OpenAI-style FileObject. + + For two-step uploads, the handler stores the initial response in litellm_params. + We need to return the file object from the initial POST, not the final PUT. + + Manus initial response format: + { + "id": "file-abc123xyz", + "object": "file", + "filename": "document.pdf", + "status": "pending", + "upload_url": "https://...", + "upload_expires_at": "...", + "created_at": "..." + } + """ + try: + # For two-step uploads, get the initial response from litellm_params + initial_response_data = litellm_params.get("initial_file_response") + if initial_response_data: + response_json = initial_response_data + else: + # Log raw response for debugging + verbose_logger.debug(f"Manus raw response text: {raw_response.text}") + response_json = raw_response.json() + + verbose_logger.debug(f"Manus file response: {response_json}") + + # Parse created_at timestamp + created_at_str = response_json.get("created_at", "") + if created_at_str: + try: + # Try parsing ISO format + created_at = int( + time.mktime( + time.strptime( + created_at_str.replace("Z", "+00:00")[:19], + "%Y-%m-%dT%H:%M:%S", + ) + ) + ) + except (ValueError, TypeError): + created_at = int(time.time()) + else: + created_at = int(time.time()) + + return OpenAIFileObject( + id=response_json.get("id", ""), + bytes=response_json.get("bytes", 0), + created_at=created_at, + filename=response_json.get("filename", ""), + object="file", + purpose=response_json.get("purpose", "assistants"), + status="uploaded", # After successful upload, status is uploaded + status_details=response_json.get("status_details"), + ) + except Exception as e: + verbose_logger.exception(f"Error parsing Manus file response: {str(e)}") + raise ValueError(f"Error parsing Manus file response: {str(e)}") + + def transform_retrieve_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Get URL and params for retrieving a file.""" + api_base = self.get_complete_url( + api_base=litellm_params.get("api_base"), + api_key=litellm_params.get("api_key"), + model="", + optional_params=optional_params, + litellm_params=litellm_params, + ) + return f"{api_base}/{file_id}", {} + + def transform_retrieve_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> OpenAIFileObject: + """Transform retrieve file response.""" + return self.transform_create_file_response( + model=None, + raw_response=raw_response, + logging_obj=logging_obj, + litellm_params=litellm_params, + ) + + def transform_delete_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Get URL and params for deleting a file.""" + api_base = self.get_complete_url( + api_base=litellm_params.get("api_base"), + api_key=litellm_params.get("api_key"), + model="", + optional_params=optional_params, + litellm_params=litellm_params, + ) + return f"{api_base}/{file_id}", {} + + def transform_delete_file_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> FileDeleted: + """Transform delete file response.""" + response_json = raw_response.json() + return FileDeleted(**response_json) + + def transform_list_files_request( + self, + purpose: Optional[str], + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Get URL and params for listing files.""" + api_base = self.get_complete_url( + api_base=litellm_params.get("api_base"), + api_key=litellm_params.get("api_key"), + model="", + optional_params=optional_params, + litellm_params=litellm_params, + ) + params = {} + if purpose: + params["purpose"] = purpose + return api_base, params + + def transform_list_files_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> List[OpenAIFileObject]: + """Transform list files response.""" + response_json = raw_response.json() + files_data = response_json.get("data", []) + return [self._parse_file_dict(f) for f in files_data] + + def _parse_file_dict(self, file_dict: Dict[str, Any]) -> OpenAIFileObject: + """Parse a file dict into OpenAIFileObject.""" + created_at_str = file_dict.get("created_at", "") + if created_at_str: + try: + created_at = int( + time.mktime( + time.strptime( + created_at_str.replace("Z", "+00:00")[:19], + "%Y-%m-%dT%H:%M:%S", + ) + ) + ) + except (ValueError, TypeError): + created_at = int(time.time()) + else: + created_at = int(time.time()) + + return OpenAIFileObject( + id=file_dict.get("id", ""), + bytes=file_dict.get("bytes", 0), + created_at=created_at, + filename=file_dict.get("filename", ""), + object="file", + purpose=file_dict.get("purpose", "assistants"), + status=file_dict.get("status", "uploaded"), + status_details=file_dict.get("status_details"), + ) + + def transform_file_content_request( + self, + file_content_request: FileContentRequest, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + """Get URL and params for retrieving file content.""" + file_id = file_content_request.get("file_id") + api_base = self.get_complete_url( + api_base=litellm_params.get("api_base"), + api_key=litellm_params.get("api_key"), + model="", + optional_params=optional_params, + litellm_params=litellm_params, + ) + return f"{api_base}/{file_id}/content", {} + + def transform_file_content_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> HttpxBinaryResponseContent: + """Transform file content response.""" + return HttpxBinaryResponseContent(response=raw_response) + diff --git a/litellm/llms/manus/responses/transformation.py b/litellm/llms/manus/responses/transformation.py index 8e0a4a5a30f..fbbed19f8d4 100644 --- a/litellm/llms/manus/responses/transformation.py +++ b/litellm/llms/manus/responses/transformation.py @@ -1,3 +1,4 @@ +import uuid from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple, Union import httpx @@ -227,6 +228,12 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig): total_tokens=0, ) + # Ensure id is present - failed responses may not include it + if "id" not in raw_response_json or raw_response_json.get("id") is None: + # Generate a placeholder id for failed responses + # This allows the response object to be created even when the API doesn't return an id + raw_response_json["id"] = f"unknown-{uuid.uuid4().hex[:8]}" + try: response = ResponsesAPIResponse(**raw_response_json) except Exception: @@ -296,6 +303,28 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig): raw_response_headers = dict(raw_response.headers) processed_headers = process_response_headers(raw_response_headers) + # Ensure reasoning, text, output, and usage are present with defaults + if "reasoning" not in raw_response_json or raw_response_json.get("reasoning") is None: + raw_response_json["reasoning"] = {} + + if "text" not in raw_response_json or raw_response_json.get("text") is None: + raw_response_json["text"] = {} + + if "output" not in raw_response_json or raw_response_json.get("output") is None: + raw_response_json["output"] = [] + + if "usage" not in raw_response_json or raw_response_json.get("usage") is None: + raw_response_json["usage"] = ResponseAPIUsage( + input_tokens=0, + output_tokens=0, + total_tokens=0, + ) + + # Ensure id is present - failed responses may not include it + if "id" not in raw_response_json or raw_response_json.get("id") is None: + # Generate a placeholder id for failed responses + raw_response_json["id"] = f"unknown-{uuid.uuid4().hex[:8]}" + try: response = ResponsesAPIResponse(**raw_response_json) except Exception: diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py index 01f6c86fd4d..b3612113ec2 100644 --- a/litellm/llms/vertex_ai/files/transformation.py +++ b/litellm/llms/vertex_ai/files/transformation.py @@ -1,11 +1,12 @@ import json import os import time -from litellm._uuid import uuid from typing import Any, Dict, List, Optional, Tuple, Union from httpx import Headers, Response +from openai.types.file_deleted import FileDeleted +from litellm._uuid import uuid from litellm.files.utils import FilesAPIUtils from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data from litellm.llms.base_llm.chat.transformation import BaseLLMException @@ -24,6 +25,7 @@ from litellm.types.llms.openai import ( AllMessageValues, CreateFileRequest, FileTypes, + HttpxBinaryResponseContent, OpenAICreateFileRequestOptionalParams, OpenAIFileObject, PathLike, @@ -333,6 +335,70 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): status_code=status_code, message=error_message, headers=headers ) + def transform_retrieve_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("VertexAIFilesConfig does not support file retrieval") + + def transform_retrieve_file_response( + self, + raw_response: Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> OpenAIFileObject: + raise NotImplementedError("VertexAIFilesConfig does not support file retrieval") + + def transform_delete_file_request( + self, + file_id: str, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("VertexAIFilesConfig does not support file deletion") + + def transform_delete_file_response( + self, + raw_response: Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> FileDeleted: + raise NotImplementedError("VertexAIFilesConfig does not support file deletion") + + def transform_list_files_request( + self, + purpose: Optional[str], + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("VertexAIFilesConfig does not support file listing") + + def transform_list_files_response( + self, + raw_response: Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> List[OpenAIFileObject]: + raise NotImplementedError("VertexAIFilesConfig does not support file listing") + + def transform_file_content_request( + self, + file_content_request, + optional_params: dict, + litellm_params: dict, + ) -> tuple[str, dict]: + raise NotImplementedError("VertexAIFilesConfig does not support file content retrieval") + + def transform_file_content_response( + self, + raw_response: Response, + logging_obj: LiteLLMLoggingObj, + litellm_params: dict, + ) -> HttpxBinaryResponseContent: + raise NotImplementedError("VertexAIFilesConfig does not support file content retrieval") + class VertexAIJsonlFilesTransformation(VertexGeminiConfig): """ diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py index a3606ff9deb..826f151df35 100644 --- a/litellm/llms/vertex_ai/vertex_llm_base.py +++ b/litellm/llms/vertex_ai/vertex_llm_base.py @@ -388,10 +388,6 @@ class VertexBase: Internal function. Returns the token and url for the call. Handles logic if it's google ai studio vs. vertex ai. - - For Vertex AI: - - If gemini_api_key is provided, use API key authentication (x-goog-api-key header) - - Otherwise, use service account credentials (OAuth2 Bearer token) Returns token, url @@ -404,7 +400,7 @@ class VertexBase: stream=stream, gemini_api_key=gemini_api_key, ) - auth_header = None # this field is not used for gemini + auth_header = None # this field is not used for gemin else: vertex_location = self.get_vertex_region( vertex_region=vertex_location, @@ -413,32 +409,14 @@ class VertexBase: ### SET RUNTIME ENDPOINT ### version = "v1beta1" if should_use_v1beta1_features is True else "v1" - - # Check if using API key authentication for Vertex AI - if gemini_api_key and not vertex_credentials: - # When using API key with Vertex AI, use the Google AI Studio endpoint - # This is because Vertex AI API keys work with generativelanguage.googleapis.com - verbose_logger.debug( - f"Using Vertex AI API key authentication for model: {model} - routing to Google AI Studio endpoint" - ) - url, endpoint = _get_gemini_url( - mode=mode, - model=model, - stream=stream, - gemini_api_key=gemini_api_key, - ) - # API key is already included in the URL by _get_gemini_url - auth_header = None - else: - # Use OAuth2 Bearer token authentication (traditional Vertex AI) - url, endpoint = _get_vertex_url( - mode=mode, - model=model, - stream=stream, - vertex_project=vertex_project, - vertex_location=vertex_location, - vertex_api_version=version, - ) + url, endpoint = _get_vertex_url( + mode=mode, + model=model, + stream=stream, + vertex_project=vertex_project, + vertex_location=vertex_location, + vertex_api_version=version, + ) return self._check_custom_proxy( api_base=api_base, diff --git a/litellm/main.py b/litellm/main.py index 10e3bcac04b..6905d8f6a87 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -189,7 +189,7 @@ from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from .llms.custom_llm import CustomLLM, custom_chat_llm_router from .llms.databricks.embed.handler import DatabricksEmbeddingHandler from .llms.deprecated_providers import aleph_alpha, palm -from .llms.gemini.common_utils import get_api_key_from_env, get_vertex_api_key_from_env +from .llms.gemini.common_utils import get_api_key_from_env from .llms.groq.chat.handler import GroqChatCompletion from .llms.heroku.chat.transformation import HerokuChatConfig from .llms.huggingface.embedding.handler import HuggingFaceEmbedding @@ -3230,12 +3230,6 @@ def completion( # type: ignore # noqa: PLR0915 or get_secret("VERTEXAI_CREDENTIALS") ) - vertex_api_key = ( - api_key - or get_vertex_api_key_from_env() - or litellm.api_key - ) - api_base = api_base or litellm.api_base or get_secret("VERTEXAI_API_BASE") new_params = safe_deep_copy(optional_params or {}) @@ -3277,7 +3271,7 @@ def completion( # type: ignore # noqa: PLR0915 vertex_location=vertex_ai_location, vertex_project=vertex_ai_project, vertex_credentials=vertex_credentials, - gemini_api_key=vertex_api_key, # Support for Vertex AI API Key + gemini_api_key=None, logging_obj=logging, acompletion=acompletion, timeout=timeout, diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index 26778ece60e..843da0e3a0d 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -1794,11 +1794,20 @@ async def get_org_object( user_api_key_cache: DualCache, parent_otel_span: Optional[Span] = None, proxy_logging_obj: Optional[ProxyLogging] = None, + include_budget_table: bool = False, ) -> Optional[LiteLLM_OrganizationTable]: """ - Check if org id in proxy Org Table - if valid, return LiteLLM_OrganizationTable object - if not, then raise an error + + Args: + org_id: Organization ID to look up + prisma_client: Database client + user_api_key_cache: Cache for storing results + parent_otel_span: Optional OpenTelemetry span + proxy_logging_obj: Optional proxy logging object + include_budget_table: If True, includes litellm_budget_table in the query """ if prisma_client is None: raise Exception( @@ -1807,8 +1816,13 @@ async def get_org_object( if not isinstance(org_id, str): return None + # Use different cache key if budget table is included + cache_key = "org_id:{}".format(org_id) + if include_budget_table: + cache_key = "org_id:{}:with_budget".format(org_id) + # check if in cache - cached_org_obj = user_api_key_cache.async_get_cache(key="org_id:{}".format(org_id)) + cached_org_obj = user_api_key_cache.async_get_cache(key=cache_key) if cached_org_obj is not None: if isinstance(cached_org_obj, dict): return LiteLLM_OrganizationTable(**cached_org_obj) @@ -1816,13 +1830,24 @@ async def get_org_object( return cached_org_obj # else, check db try: + query_kwargs = {"where": {"organization_id": org_id}} + if include_budget_table: + query_kwargs["include"] = {"litellm_budget_table": True} + response = await prisma_client.db.litellm_organizationtable.find_unique( - where={"organization_id": org_id} + **query_kwargs ) if response is None: raise Exception + # Cache the result + await user_api_key_cache.async_set_cache( + key=cache_key, + value=response.model_dump() if hasattr(response, "model_dump") else response, + ttl=DEFAULT_IN_MEMORY_TTL, + ) + return response except Exception: raise Exception( @@ -2344,11 +2369,14 @@ async def _organization_max_budget_check( if org_id is None: return - # Get organization object with budget table to check current spend and max budget + # Get organization object with budget table - use get_org_object so it can be mocked in tests try: - org_table = await prisma_client.db.litellm_organizationtable.find_unique( - where={"organization_id": org_id}, - include={"litellm_budget_table": True}, + org_table = await get_org_object( + org_id=org_id, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + proxy_logging_obj=proxy_logging_obj, + include_budget_table=True, ) except Exception: # If organization lookup fails, skip the check diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 171898b1631..9ea2ea7d5c9 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -4479,21 +4479,35 @@ def validate_model_access( ) -> None: """ Validate that a model is accessible to the user. + Supports batch requests with comma-separated model IDs. Args: - model_id: The model ID to validate + model_id: The model ID to validate (can be comma-separated for batch requests) available_models: List of models available to the user Raises: HTTPException: If the model is not accessible """ - if model_id not in available_models: - raise HTTPException( - status_code=404, - detail="The model `{}` does not exist or is not accessible".format( - model_id - ), - ) + # Handle batch requests with comma-separated models + if "," in model_id: + models = [m.strip() for m in model_id.split(",")] + inaccessible_models = [m for m in models if m not in available_models] + if inaccessible_models: + raise HTTPException( + status_code=404, + detail="The following model(s) do not exist or are not accessible: {}".format( + ", ".join(inaccessible_models) + ), + ) + else: + # Single model validation + if model_id not in available_models: + raise HTTPException( + status_code=404, + detail="The model `{}` does not exist or is not accessible".format( + model_id + ), + ) def _path_matches_pattern(path: str, pattern: str) -> bool: diff --git a/litellm/responses/utils.py b/litellm/responses/utils.py index 7667d1bad84..49b8e123fe2 100644 --- a/litellm/responses/utils.py +++ b/litellm/responses/utils.py @@ -198,11 +198,14 @@ class ResponsesAPIRequestUtils: model_id = model_info.get("id") # access the response id based on the object type - response_id = ( - responses_api_response["id"] - if isinstance(responses_api_response, dict) - else responses_api_response.id - ) + if isinstance(responses_api_response, dict): + response_id = responses_api_response.get("id") + else: + response_id = getattr(responses_api_response, "id", None) + + # If no response_id, return the response as-is (likely an error response) + if response_id is None: + return responses_api_response updated_id = ResponsesAPIRequestUtils._build_responses_api_response_id( model_id=model_id, diff --git a/litellm/router.py b/litellm/router.py index 364e6719300..638df49ac05 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -4486,21 +4486,9 @@ class Router: if hasattr(original_exception, "message"): # add the available fallbacks to the exception - deployment_info = "" - if kwargs is not None: - metadata = kwargs.get('metadata', {}) - if metadata and 'deployment' in metadata: - deployment_info = f"\nUsed Deployment: {metadata['deployment']}" - if 'model_info' in metadata: - model_info = metadata['model_info'] - if isinstance(model_info, dict): - deployment_info += f"\nDeployment ID: {model_info.get('id', 'unknown')}" - - original_exception.message += ( # type: ignore - f". Received Model Group={model_group}" - f"\nAvailable Model Group Fallbacks={fallback_model_group}" - f"{deployment_info}" - f"\n\nšŸ’” Tip: If using wildcard patterns (e.g., 'openai/*'), ensure all matching deployments have credentials with access to this model." + original_exception.message += ". Received Model Group={}\nAvailable Model Group Fallbacks={}".format( # type: ignore + model_group, + fallback_model_group, ) if len(fallback_failure_exception_str) > 0: original_exception.message += ( # type: ignore @@ -7679,10 +7667,6 @@ class Router: ) if pattern_deployments: - verbose_router_logger.debug( - f"Pattern match for model='{model}': Found {len(pattern_deployments)} deployments. " - f"Deployment IDs: {[d.get('model_info', {}).get('id', 'unknown') for d in pattern_deployments]}" - ) return model, pattern_deployments if ( diff --git a/litellm/types/files.py b/litellm/types/files.py index 600ad806e22..8b87b33cd1b 100644 --- a/litellm/types/files.py +++ b/litellm/types/files.py @@ -1,6 +1,8 @@ from enum import Enum from types import MappingProxyType -from typing import List, Set, Mapping +from typing import Any, Dict, List, Literal, Mapping, Set, Union + +from typing_extensions import Required, TypedDict """ Base Enums/Consts @@ -281,3 +283,41 @@ GEMINI_1_5_ACCEPTED_FILE_TYPES: Set[FileType] = { def is_gemini_1_5_accepted_file_type(file_type: FileType) -> bool: return file_type in GEMINI_1_5_ACCEPTED_FILE_TYPES + + +""" +Two-Step File Upload Types +""" + + +class TwoStepFileUploadRequest(TypedDict): + """ + Request structure for two-step file upload process. + + Step 1: Initial request to get upload URL + Step 2: Upload file content to the upload URL + + Used by providers like Manus and Google Cloud Storage. + """ + + method: Required[str] + url: Required[str] + headers: Required[Dict[str, str]] + data: Required[Union[str, bytes, Dict[str, Any]]] + + +class TwoStepFileUploadConfig(TypedDict, total=False): + """ + Configuration for two-step file upload process. + + Properties: + initial_request: Request to create file record and get upload URL + upload_request: Request to upload actual file content + upload_url_location: Where to find upload URL ('headers' or 'body') + upload_url_key: Key name for upload URL in response (default: 'upload_url') + """ + + initial_request: Required[TwoStepFileUploadRequest] + upload_request: Required[TwoStepFileUploadRequest] + upload_url_location: Required[Literal["headers", "body"]] + upload_url_key: str diff --git a/litellm/utils.py b/litellm/utils.py index 42d4a1ac372..1393f1f5cb5 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -48,6 +48,7 @@ from tokenizers import Tokenizer import litellm import litellm.litellm_core_utils + # audio_utils.utils is lazy-loaded - only imported when needed for transcription calls import litellm.litellm_core_utils.json_validation_rule from litellm._lazy_imports import ( @@ -71,8 +72,6 @@ from litellm.constants import ( TOOL_CHOICE_OBJECT_TOKEN_COUNT, ) - - _CachingHandlerResponse = None _LLMCachingHandler = None _CustomGuardrail = None @@ -7959,6 +7958,10 @@ class ProviderConfigManager: from litellm.llms.bedrock.files.transformation import BedrockFilesConfig return BedrockFilesConfig() + elif LlmProviders.MANUS == provider: + from litellm.llms.manus.files.transformation import ManusFilesConfig + + return ManusFilesConfig() return None @staticmethod diff --git a/poetry.lock b/poetry.lock index 0a4ef10d09f..3bafdb157ca 100644 --- a/poetry.lock +++ b/poetry.lock @@ -3081,15 +3081,15 @@ files = [ [[package]] name = "litellm-proxy-extras" -version = "0.4.18" +version = "0.4.21" description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package." optional = true python-versions = "!=2.7.*,!=3.0.*,!=3.1.*,!=3.2.*,!=3.3.*,!=3.4.*,!=3.5.*,!=3.6.*,!=3.7.*,>=3.8" groups = ["main"] markers = "extra == \"proxy\"" files = [ - {file = "litellm_proxy_extras-0.4.18-py3-none-any.whl", hash = "sha256:c3edee68bf8eb073c6158dcf7df05727dfc829e63c03a617fcb48853d11490df"}, - {file = "litellm_proxy_extras-0.4.18.tar.gz", hash = "sha256:898b28e3e74acdc29142906b84787ab05a90e30aa3c0c8aee849915e3a16adb3"}, + {file = "litellm_proxy_extras-0.4.21-py3-none-any.whl", hash = "sha256:83a1734e9773610945230606012e602bbcbfba1c60fde836d51102c1a296f166"}, + {file = "litellm_proxy_extras-0.4.21.tar.gz", hash = "sha256:fa0e012984aa8e5114f88f4bad53d6abb589e5ca3eab445f74f8ddeceb62d848"}, ] [[package]] @@ -7981,4 +7981,4 @@ utils = ["numpydoc"] [metadata] lock-version = "2.1" python-versions = ">=3.9,<4.0" -content-hash = "e9fd12b5ccc703ec156d98877452417083e3ac18b5970cb3a58c3bde09d267bb" +content-hash = "ea62b77c662ab9fc486e421c576f0868bcde16d62a24703ee1f4916a0465ffb2" diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 8432ce4e874..db11ec8a0de 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -2335,6 +2335,7 @@ "audio_speech": false, "moderations": false, "batches": false, + "files": true, "rerank": false, "a2a": true, "interactions": true diff --git a/pyproject.toml b/pyproject.toml index 00e7bff78b6..742f743b54d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -167,7 +167,7 @@ requires = ["poetry-core", "wheel"] build-backend = "poetry.core.masonry.api" [tool.commitizen] -version = "1.80.13" +version = "1.80.14" version_files = [ "pyproject.toml:^version" ] diff --git a/tests/batches_tests/test_manus_files_all_methods.py b/tests/batches_tests/test_manus_files_all_methods.py new file mode 100644 index 00000000000..49322bdcb17 --- /dev/null +++ b/tests/batches_tests/test_manus_files_all_methods.py @@ -0,0 +1,72 @@ +""" +E2E test for all Manus Files API methods. +""" + +import os +import pytest +import litellm + + +@pytest.mark.asyncio +async def test_manus_files_api_e2e_all_methods(): + """ + E2E test for Manus Files API: create, retrieve, list, delete. + """ + litellm._turn_on_debug() + + api_key = os.getenv("MANUS_API_KEY") + if api_key is None: + pytest.skip("MANUS_API_KEY not set") + + # Create a simple test file content + test_content = b"This is a test file for Manus Files API - all methods test." + test_filename = "test_file_all_methods.txt" + + # Step 1: Create file + print("Step 1: Creating file...") + created_file = await litellm.acreate_file( + file=(test_filename, test_content), + purpose="assistants", + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Created file: {created_file}") + assert created_file.filename == test_filename + assert created_file.status == "uploaded" + # Note: Manus doesn't return bytes in initial response + file_id = created_file.id + + # Step 2: Retrieve file + print(f"\nStep 2: Retrieving file {file_id}...") + retrieved_file = await litellm.afile_retrieve( + file_id=file_id, + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Retrieved file: {retrieved_file}") + assert retrieved_file.id == file_id + assert retrieved_file.filename == test_filename + + # Step 3: List files + print("\nStep 3: Listing files...") + files_list = await litellm.afile_list( + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Files list: {files_list}") + assert isinstance(files_list, list) + assert any(f.id == file_id for f in files_list) + + # Step 4: Delete file + print(f"\nStep 4: Deleting file {file_id}...") + deleted_file = await litellm.afile_delete( + file_id=file_id, + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Deleted file: {deleted_file}") + assert deleted_file.id == file_id + assert deleted_file.deleted is True + + print("\nāœ… All Manus Files API methods working!") + diff --git a/tests/code_coverage_tests/recursive_detector.py b/tests/code_coverage_tests/recursive_detector.py index 20e6a381e5b..2a460f621b2 100644 --- a/tests/code_coverage_tests/recursive_detector.py +++ b/tests/code_coverage_tests/recursive_detector.py @@ -35,6 +35,7 @@ IGNORE_FUNCTIONS = [ "_fix_enum_types", # max depth set. "_collect_argument_paths", # max depth set. "_split_text", # max depth set. + "_mask_sequence", # max depth set. "_delete_nested_value_custom", # max depth set (bounded by number of path segments). "filter_exceptions_from_params", # max depth set (default 20) to prevent infinite recursion. "__getattr__", # lazy loading pattern in litellm/__init__.py with proper caching to prevent infinite recursion. diff --git a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py index 2f92afb3824..e8fe4dd3393 100644 --- a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py +++ b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py @@ -1124,124 +1124,6 @@ def test_get_custom_labels_from_metadata_tags(monkeypatch): assert get_custom_labels_from_metadata(metadata) == {} -def test_get_custom_labels_from_top_level_metadata(monkeypatch): - """ - Test that get_custom_labels_from_metadata can extract fields from top-level metadata, - such as requester_ip_address, not just from nested dictionaries like requester_metadata. - """ - monkeypatch.setattr( - "litellm.custom_prometheus_metadata_labels", - ["requester_ip_address", "user_api_key_alias"], - ) - # Simulate metadata structure with top-level fields - metadata = { - "requester_ip_address": "10.48.203.20", # Top-level field - "user_api_key_alias": "TestAlias", # Top-level field - "requester_metadata": {"nested_field": "nested_value"}, # Nested dict (excluded) - "user_api_key_auth_metadata": {"another_nested": "value"}, # Nested dict (excluded) - } - result = get_custom_labels_from_metadata(metadata) - assert result == { - "requester_ip_address": "10.48.203.20", - "user_api_key_alias": "TestAlias", - } - - -def test_get_custom_labels_from_top_level_and_nested_metadata(monkeypatch): - """ - Test that get_custom_labels_from_metadata can extract fields from both top-level - and nested metadata (requester_metadata, user_api_key_auth_metadata). - """ - monkeypatch.setattr( - "litellm.custom_prometheus_metadata_labels", - [ - "requester_ip_address", # Top-level - "metadata.foo", # From requester_metadata - "metadata.bar", # From user_api_key_auth_metadata - ], - ) - # Simulate combined_metadata structure as it would appear after merging - # This is what gets passed to get_custom_labels_from_metadata - combined_metadata = { - "requester_ip_address": "10.48.203.20", # Top-level field - "foo": "bar_value", # From requester_metadata (spread) - "bar": "baz_value", # From user_api_key_auth_metadata (spread) - } - result = get_custom_labels_from_metadata(combined_metadata) - assert result == { - "requester_ip_address": "10.48.203.20", - "metadata_foo": "bar_value", - "metadata_bar": "baz_value", - } - - -async def test_async_log_success_event_with_top_level_metadata(prometheus_logger, monkeypatch): - """ - Test that async_log_success_event correctly extracts custom labels from top-level metadata - fields like requester_ip_address, not just from nested dictionaries. - """ - # Configure custom metadata labels to extract requester_ip_address - monkeypatch.setattr( - "litellm.custom_prometheus_metadata_labels", ["requester_ip_address"] - ) - - # Create standard logging payload with requester_ip_address at top-level metadata - standard_logging_object = create_standard_logging_payload() - standard_logging_object["metadata"]["requester_ip_address"] = "10.48.203.20" - standard_logging_object["metadata"]["requester_metadata"] = {} # Empty nested dict - standard_logging_object["metadata"]["user_api_key_auth_metadata"] = {} # Empty nested dict - - kwargs = { - "model": "gpt-3.5-turbo", - "stream": True, - "litellm_params": { - "metadata": { - "user_api_key": "test_key", - "user_api_key_user_id": "test_user", - "user_api_key_team_id": "test_team", - "user_api_key_end_user_id": "test_end_user", - } - }, - "start_time": datetime.now(), - "completion_start_time": datetime.now(), - "api_call_start_time": datetime.now(), - "end_time": datetime.now() + timedelta(seconds=1), - "standard_logging_object": standard_logging_object, - } - response_obj = MagicMock() - - # Mock the prometheus client methods - prometheus_logger.litellm_requests_metric = MagicMock() - prometheus_logger.litellm_spend_metric = MagicMock() - prometheus_logger.litellm_tokens_metric = MagicMock() - prometheus_logger.litellm_input_tokens_metric = MagicMock() - prometheus_logger.litellm_output_tokens_metric = MagicMock() - prometheus_logger.litellm_remaining_team_budget_metric = MagicMock() - prometheus_logger.litellm_remaining_api_key_budget_metric = MagicMock() - prometheus_logger.litellm_remaining_api_key_requests_for_model = MagicMock() - prometheus_logger.litellm_remaining_api_key_tokens_for_model = MagicMock() - prometheus_logger.litellm_llm_api_time_to_first_token_metric = MagicMock() - prometheus_logger.litellm_llm_api_latency_metric = MagicMock() - prometheus_logger.litellm_request_total_latency_metric = MagicMock() - - await prometheus_logger.async_log_success_event( - kwargs, response_obj, kwargs["start_time"], kwargs["end_time"] - ) - - # Verify that the metrics were called with labels including requester_ip_address - # Check that labels() was called - the actual labels dict should include requester_ip_address - assert prometheus_logger.litellm_requests_metric.labels.called - assert prometheus_logger.litellm_spend_metric.labels.called - - # Get the actual call arguments to verify requester_ip_address is included - # The custom labels should be extracted and included in the label factory - call_args = prometheus_logger.litellm_requests_metric.labels.call_args - assert call_args is not None - # The labels() method receives a dict with label names and values - # We can't easily assert the exact values without checking the internal implementation, - # but we've verified the function is called, which means the extraction happened - - def test_get_custom_labels_from_tags(monkeypatch): from litellm.integrations.prometheus import get_custom_labels_from_tags diff --git a/tests/llm_responses_api_testing/test_manus_files_all_methods.py b/tests/llm_responses_api_testing/test_manus_files_all_methods.py new file mode 100644 index 00000000000..39311441f59 --- /dev/null +++ b/tests/llm_responses_api_testing/test_manus_files_all_methods.py @@ -0,0 +1,71 @@ +""" +E2E test for all Manus Files API methods. +""" + +import os +import pytest +import litellm + + +@pytest.mark.asyncio +async def test_manus_files_api_e2e_all_methods(): + """ + E2E test for Manus Files API: create, retrieve, list, delete. + """ + litellm._turn_on_debug() + + api_key = os.getenv("MANUS_API_KEY") + if api_key is None: + pytest.skip("MANUS_API_KEY not set") + + # Create a simple test file content + test_content = b"This is a test file for Manus Files API - all methods test." + test_filename = "test_file_all_methods.txt" + + # Step 1: Create file + print("Step 1: Creating file...") + created_file = await litellm.acreate_file( + file=(test_filename, test_content), + purpose="assistants", + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Created file: {created_file}") + assert created_file.filename == test_filename + assert created_file.status == "uploaded" + # Note: Manus doesn't return bytes in initial response + file_id = created_file.id + + # Step 2: Retrieve file + print(f"\nStep 2: Retrieving file {file_id}...") + retrieved_file = await litellm.afile_retrieve( + file_id=file_id, + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Retrieved file: {retrieved_file}") + assert retrieved_file.id == file_id + assert retrieved_file.filename == test_filename + + # Step 3: List files + print("\nStep 3: Listing files...") + files_list = await litellm.afile_list( + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Files list: {files_list}") + assert isinstance(files_list, list) + assert any(f.id == file_id for f in files_list) + + # Step 4: Delete file + print(f"\nStep 4: Deleting file {file_id}...") + deleted_file = await litellm.afile_delete( + file_id=file_id, + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Deleted file: {deleted_file}") + assert deleted_file.id == file_id + assert deleted_file.deleted is True + + print("\nāœ… All Manus Files API methods working!") diff --git a/tests/llm_responses_api_testing/test_manus_responses_api.py b/tests/llm_responses_api_testing/test_manus_responses_api.py index 4f64980d261..6256363f407 100644 --- a/tests/llm_responses_api_testing/test_manus_responses_api.py +++ b/tests/llm_responses_api_testing/test_manus_responses_api.py @@ -38,7 +38,11 @@ async def test_manus_responses_api_with_agent_profile(): print("Manus response=", json.dumps(response, indent=4, default=str)) ## Get the status of the response - got_response = await litellm.aget_responses(response_id=response.id) + got_response = await litellm.aget_responses( + response_id=response.id, + custom_llm_provider="manus", + api_key=os.getenv("MANUS_API_KEY"), + ) print("GET API MANUS RESPONSE=", json.dumps(got_response, indent=4, default=str)) if got_response.status == "completed": assert got_response.output is not None @@ -47,6 +51,95 @@ async def test_manus_responses_api_with_agent_profile(): # Manus can return "running" or "pending" status assert got_response.status in ["running", "pending"] assert got_response.id is not None + + +@pytest.mark.asyncio +async def test_manus_responses_api_with_file_upload(): + """ + Test that uploads a file via Files API and then passes it to Responses API. + """ + litellm._turn_on_debug() + + api_key = os.getenv("MANUS_API_KEY") + if api_key is None: + pytest.skip("MANUS_API_KEY not set") + + # Step 1: Upload a file + test_content = b"Warren Buffett's 2023 Letter to Shareholders\n\nKey Points:\n1. Long-term value creation\n2. Capital allocation strategy\n3. Market volatility perspective" + test_filename = "buffett_letter_summary.txt" + + print("Step 1: Uploading file...") + uploaded_file = await litellm.acreate_file( + file=(test_filename, test_content), + purpose="assistants", + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Uploaded file: {uploaded_file}") + assert uploaded_file.id is not None + file_id = uploaded_file.id + + # Step 2: Create a response with the uploaded file + print(f"\nStep 2: Creating response with file {file_id}...") + response = await litellm.aresponses( + model="manus/manus-1.6-lite", + input=[ + { + "role": "user", + "content": [ + { + "type": "input_text", + "text": "Summarize the key points from this letter.", + }, + { + "type": "input_file", + "file_id": file_id, + }, + ], + }, + ], + api_key=api_key, + max_output_tokens=100, + ) + + print(f"Response created: {response}") + print(f"Response type: {type(response)}") + print(f"Response has id: {hasattr(response, 'id')}") + + # Handle both dict and ResponsesAPIResponse object + if isinstance(response, dict): + response_id = response.get("id") + else: + response_id = getattr(response, "id", None) + + assert response_id is not None, f"Response ID is None. Response: {response}" + + # Step 3: Get the response status + print(f"\nStep 3: Getting response status...") + got_response = await litellm.aget_responses( + response_id=response_id, + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Response status: {got_response}") + got_response_id = getattr(got_response, "id", None) + got_response_status = getattr(got_response, "status", None) + + assert got_response_id == response_id + assert got_response_status in ["completed", "running", "pending"] + + # Step 4: Clean up - delete the file + print(f"\nStep 4: Cleaning up - deleting file {file_id}...") + deleted_file = await litellm.afile_delete( + file_id=file_id, + custom_llm_provider="manus", + api_key=api_key, + ) + print(f"Deleted file: {deleted_file}") + assert deleted_file.deleted is True + + print("\nāœ… File upload and responses API integration test passed!") + diff --git a/tests/proxy_unit_tests/test_proxy_server.py b/tests/proxy_unit_tests/test_proxy_server.py index 8d3b0fee48f..0db1240544a 100644 --- a/tests/proxy_unit_tests/test_proxy_server.py +++ b/tests/proxy_unit_tests/test_proxy_server.py @@ -642,8 +642,8 @@ def test_embedding(mock_aembedding, client_no_auth): during_call_kwargs = mock_during_hook.await_args_list[0].kwargs assert ( - during_call_kwargs.get("call_type") == "embeddings" - ), f"expected during_call_hook to receive call_type='embeddings', got {during_call_kwargs.get('call_type')}" + during_call_kwargs.get("call_type") == "embedding" + ), f"expected during_call_hook to receive call_type='embedding', got {during_call_kwargs.get('call_type')}" except Exception as e: pytest.fail(f"LiteLLM Proxy test failed. Exception - {str(e)}") @@ -2185,6 +2185,10 @@ async def test_proxy_server_prisma_setup(): mock_client._set_spend_logs_row_count_in_proxy_state = ( AsyncMock() ) # Mock the _set_spend_logs_row_count_in_proxy_state method + # Mock the db attribute with start_token_refresh_task for RDS IAM token refresh + mock_db = MagicMock() + mock_db.start_token_refresh_task = AsyncMock() + mock_client.db = mock_db await ProxyStartupEvent._setup_prisma_client( database_url=os.getenv("DATABASE_URL"), diff --git a/tests/proxy_unit_tests/test_proxy_utils.py b/tests/proxy_unit_tests/test_proxy_utils.py index 5c3c3948920..64f1ec24234 100644 --- a/tests/proxy_unit_tests/test_proxy_utils.py +++ b/tests/proxy_unit_tests/test_proxy_utils.py @@ -1574,6 +1574,10 @@ async def test_health_check_not_called_when_disabled(monkeypatch): mock_prisma.health_check = AsyncMock() mock_prisma.check_view_exists = AsyncMock() mock_prisma._set_spend_logs_row_count_in_proxy_state = AsyncMock() + # Mock the db attribute with start_token_refresh_task for RDS IAM token refresh + mock_db = MagicMock() + mock_db.start_token_refresh_task = AsyncMock() + mock_prisma.db = mock_db # Mock PrismaClient constructor monkeypatch.setattr( "litellm.proxy.proxy_server.PrismaClient", lambda **kwargs: mock_prisma diff --git a/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py b/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py index 9f0a1ae8ffe..b9f17aa4c7f 100644 --- a/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py +++ b/tests/test_litellm/litellm_core_utils/test_sensitive_data_masker.py @@ -108,14 +108,16 @@ def test_lists_with_sensitive_keys_are_masked(): """ masker = SensitiveDataMasker() data = { - "api_key": ["sk-123", "sk-456"], + "api_key": ["sk-1234567890abcdef", "sk-9876543210fedcba"], "tags": ["prod", "test"], } masked = masker.mask_dict(data) # sensitive key list entries should be masked - assert masked["api_key"][0] != "sk-123" + assert masked["api_key"][0] != "sk-1234567890abcdef" assert "*" in masked["api_key"][0] + assert masked["api_key"][1] != "sk-9876543210fedcba" + assert "*" in masked["api_key"][1] # non-sensitive list should remain unchanged assert masked["tags"] == ["prod", "test"] diff --git a/tests/test_litellm/llms/bedrock/test_bedrock_ssl_verify.py b/tests/test_litellm/llms/bedrock/test_bedrock_ssl_verify.py deleted file mode 100644 index 9142de295ea..00000000000 --- a/tests/test_litellm/llms/bedrock/test_bedrock_ssl_verify.py +++ /dev/null @@ -1,349 +0,0 @@ -""" -Test SSL verification for AWS Bedrock boto3 clients. - -This test ensures that custom CA certificates are properly passed to all boto3 clients -(STS and Bedrock services) to support internal certificate authorities. - -Issue: https://github.com/BerriAI/litellm/issues/XXXX -User reported that SSL_CERT_FILE environment variable and ssl_verify config were not -being applied to boto3 clients, causing "certificate verify failed" errors. -""" - -import os -import sys -import tempfile -from unittest.mock import MagicMock, Mock, patch - -import pytest - -sys.path.insert(0, os.path.abspath("../..")) - -import litellm -from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM -from litellm.llms.bedrock.common_utils import init_bedrock_client - - -class TestBedrockSSLVerify: - """Test suite for SSL verification in Bedrock boto3 clients.""" - - def test_base_aws_llm_get_ssl_verify_default(self): - """Test that _get_ssl_verify returns default value when no custom config is set.""" - base_aws = BaseAWSLLM() - - # Clear any environment variables - os.environ.pop("SSL_VERIFY", None) - os.environ.pop("SSL_CERT_FILE", None) - - # Reset litellm.ssl_verify to default - litellm.ssl_verify = True - - ssl_verify = base_aws._get_ssl_verify() - assert ssl_verify is True - - def test_base_aws_llm_get_ssl_verify_false(self): - """Test that _get_ssl_verify returns False when SSL verification is disabled.""" - base_aws = BaseAWSLLM() - - # Set SSL_VERIFY to False via environment - os.environ["SSL_VERIFY"] = "False" - - ssl_verify = base_aws._get_ssl_verify() - assert ssl_verify is False - - # Clean up - os.environ.pop("SSL_VERIFY", None) - - def test_base_aws_llm_get_ssl_verify_custom_ca_bundle(self): - """Test that _get_ssl_verify returns custom CA bundle path when SSL_CERT_FILE is set.""" - base_aws = BaseAWSLLM() - - # Create a temporary CA bundle file - with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f: - f.write("-----BEGIN CERTIFICATE-----\n") - f.write("FAKE CERTIFICATE FOR TESTING\n") - f.write("-----END CERTIFICATE-----\n") - ca_bundle_path = f.name - - try: - # Set SSL_CERT_FILE environment variable - os.environ["SSL_CERT_FILE"] = ca_bundle_path - os.environ.pop("SSL_VERIFY", None) - litellm.ssl_verify = True - - ssl_verify = base_aws._get_ssl_verify() - assert ssl_verify == ca_bundle_path - finally: - # Clean up - os.environ.pop("SSL_CERT_FILE", None) - os.unlink(ca_bundle_path) - - def test_base_aws_llm_get_ssl_verify_litellm_config(self): - """Test that _get_ssl_verify uses litellm.ssl_verify when set.""" - base_aws = BaseAWSLLM() - - # Clear environment variables - os.environ.pop("SSL_VERIFY", None) - os.environ.pop("SSL_CERT_FILE", None) - - # Create a temporary CA bundle file - with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f: - f.write("-----BEGIN CERTIFICATE-----\n") - f.write("FAKE CERTIFICATE FOR TESTING\n") - f.write("-----END CERTIFICATE-----\n") - ca_bundle_path = f.name - - try: - # Set litellm.ssl_verify to custom CA bundle - litellm.ssl_verify = ca_bundle_path - - ssl_verify = base_aws._get_ssl_verify() - # When ssl_verify is a path, it should be returned directly - assert ssl_verify == ca_bundle_path - finally: - # Clean up - litellm.ssl_verify = True - os.unlink(ca_bundle_path) - - @patch("boto3.client") - def test_init_bedrock_client_passes_ssl_verify_to_sts(self, mock_boto3_client): - """Test that init_bedrock_client passes ssl_verify to STS client.""" - # Create a temporary CA bundle file - with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f: - f.write("-----BEGIN CERTIFICATE-----\n") - f.write("FAKE CERTIFICATE FOR TESTING\n") - f.write("-----END CERTIFICATE-----\n") - ca_bundle_path = f.name - - try: - # Set SSL_CERT_FILE environment variable - os.environ["SSL_CERT_FILE"] = ca_bundle_path - litellm.ssl_verify = True - - # Mock the STS client and Bedrock client - mock_sts_client = MagicMock() - mock_sts_response = { - "Credentials": { - "AccessKeyId": "test_access_key", - "SecretAccessKey": "test_secret_key", - "SessionToken": "test_session_token", - } - } - mock_sts_client.assume_role.return_value = mock_sts_response - - mock_bedrock_client = MagicMock() - - # Configure mock to return different clients based on service name - def side_effect(service_name=None, **kwargs): - if service_name == "sts": - return mock_sts_client - elif service_name == "bedrock-runtime": - return mock_bedrock_client - return MagicMock() - - mock_boto3_client.side_effect = side_effect - - # Call init_bedrock_client with role assumption - client = init_bedrock_client( - aws_region_name="us-west-2", - aws_access_key_id="test_key", - aws_secret_access_key="test_secret", - aws_role_name="arn:aws:iam::123456789012:role/test-role", - aws_session_name="test-session", - ) - - # Verify that boto3.client was called with verify parameter for STS - sts_calls = [ - call for call in mock_boto3_client.call_args_list - if (len(call[0]) > 0 and call[0][0] == "sts") or - ("service_name" not in call[1]) # STS calls don't use service_name kwarg - ] - - assert len(sts_calls) > 0, "STS client should have been created" - - # Check that verify parameter was passed to STS client - sts_call = sts_calls[0] - assert "verify" in sts_call[1], "verify parameter should be passed to STS client" - assert sts_call[1]["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {sts_call[1]['verify']}" - - # Verify that boto3.client was called with verify parameter for Bedrock - bedrock_calls = [ - call for call in mock_boto3_client.call_args_list - if "service_name" in call[1] and call[1]["service_name"] == "bedrock-runtime" - ] - - assert len(bedrock_calls) > 0, "Bedrock client should have been created" - - bedrock_call = bedrock_calls[0] - assert "verify" in bedrock_call[1], "verify parameter should be passed to Bedrock client" - assert bedrock_call[1]["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {bedrock_call[1]['verify']}" - - finally: - # Clean up - os.environ.pop("SSL_CERT_FILE", None) - os.unlink(ca_bundle_path) - - @patch("boto3.client") - def test_base_aws_llm_auth_with_role_passes_ssl_verify(self, mock_boto3_client): - """Test that _auth_with_aws_role passes ssl_verify to STS client.""" - base_aws = BaseAWSLLM() - - # Create a temporary CA bundle file - with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f: - f.write("-----BEGIN CERTIFICATE-----\n") - f.write("FAKE CERTIFICATE FOR TESTING\n") - f.write("-----END CERTIFICATE-----\n") - ca_bundle_path = f.name - - try: - # Set SSL_CERT_FILE environment variable - os.environ["SSL_CERT_FILE"] = ca_bundle_path - litellm.ssl_verify = True - - # Mock the STS client - mock_sts_client = MagicMock() - mock_sts_response = { - "Credentials": { - "AccessKeyId": "test_access_key", - "SecretAccessKey": "test_secret_key", - "SessionToken": "test_session_token", - "Expiration": "2025-01-10T00:00:00Z", - } - } - - # Convert Expiration to datetime - from datetime import datetime, timezone - mock_sts_response["Credentials"]["Expiration"] = datetime.now(timezone.utc) - - mock_sts_client.assume_role.return_value = mock_sts_response - mock_boto3_client.return_value = mock_sts_client - - # Call _auth_with_aws_role - credentials, ttl = base_aws._auth_with_aws_role( - aws_access_key_id="test_key", - aws_secret_access_key="test_secret", - aws_session_token=None, - aws_role_name="arn:aws:iam::123456789012:role/test-role", - aws_session_name="test-session", - ) - - # Verify that boto3.client was called with verify parameter - assert mock_boto3_client.called, "boto3.client should have been called" - - call_kwargs = mock_boto3_client.call_args[1] - assert "verify" in call_kwargs, "verify parameter should be passed to STS client" - assert call_kwargs["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {call_kwargs['verify']}" - - finally: - # Clean up - os.environ.pop("SSL_CERT_FILE", None) - os.unlink(ca_bundle_path) - - @patch("litellm.llms.bedrock.base_aws_llm.get_secret") - @patch("boto3.client") - def test_base_aws_llm_auth_with_web_identity_passes_ssl_verify(self, mock_boto3_client, mock_get_secret): - """Test that _auth_with_web_identity_token passes ssl_verify to STS client.""" - base_aws = BaseAWSLLM() - - # Create a temporary CA bundle file - with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f: - f.write("-----BEGIN CERTIFICATE-----\n") - f.write("FAKE CERTIFICATE FOR TESTING\n") - f.write("-----END CERTIFICATE-----\n") - ca_bundle_path = f.name - - try: - # Set SSL_CERT_FILE environment variable - os.environ["SSL_CERT_FILE"] = ca_bundle_path - litellm.ssl_verify = True - - # Mock get_secret to return the token - mock_get_secret.return_value = "mocked_oidc_token" - - # Mock the STS client - mock_sts_client = MagicMock() - mock_sts_response = { - "Credentials": { - "AccessKeyId": "test_access_key", - "SecretAccessKey": "test_secret_key", - "SessionToken": "test_session_token", - }, - "PackedPolicySize": 100, - } - - mock_sts_client.assume_role_with_web_identity.return_value = mock_sts_response - - # Mock boto3.Session - mock_session = MagicMock() - mock_credentials = MagicMock() - mock_session.get_credentials.return_value = mock_credentials - - mock_boto3_client.return_value = mock_sts_client - - with patch("boto3.Session", return_value=mock_session): - # Call _auth_with_web_identity_token - credentials, ttl = base_aws._auth_with_web_identity_token( - aws_web_identity_token="test_token", - aws_role_name="arn:aws:iam::123456789012:role/test-role", - aws_session_name="test-session", - aws_region_name="us-west-2", - aws_sts_endpoint=None, - ) - - # Verify that boto3.client was called with verify parameter - assert mock_boto3_client.called, "boto3.client should have been called" - - call_kwargs = mock_boto3_client.call_args[1] - assert "verify" in call_kwargs, "verify parameter should be passed to STS client" - assert call_kwargs["verify"] == ca_bundle_path, f"verify should be set to CA bundle path, got {call_kwargs['verify']}" - - finally: - # Clean up - os.environ.pop("SSL_CERT_FILE", None) - os.unlink(ca_bundle_path) - - def test_ssl_verify_priority_env_over_litellm_config(self): - """Test that SSL_VERIFY environment variable takes priority over litellm.ssl_verify.""" - base_aws = BaseAWSLLM() - - # Set litellm.ssl_verify to True - litellm.ssl_verify = True - - # Set SSL_VERIFY environment variable to False - os.environ["SSL_VERIFY"] = "False" - - try: - ssl_verify = base_aws._get_ssl_verify() - assert ssl_verify is False, "Environment variable should take priority" - finally: - # Clean up - os.environ.pop("SSL_VERIFY", None) - litellm.ssl_verify = True - - def test_ssl_cert_file_priority_over_default(self): - """Test that SSL_CERT_FILE takes priority when ssl_verify is True.""" - base_aws = BaseAWSLLM() - - # Create a temporary CA bundle file - with tempfile.NamedTemporaryFile(mode="w", suffix=".pem", delete=False) as f: - f.write("-----BEGIN CERTIFICATE-----\n") - f.write("FAKE CERTIFICATE FOR TESTING\n") - f.write("-----END CERTIFICATE-----\n") - ca_bundle_path = f.name - - try: - # Set SSL_CERT_FILE environment variable - os.environ["SSL_CERT_FILE"] = ca_bundle_path - os.environ.pop("SSL_VERIFY", None) - litellm.ssl_verify = True - - ssl_verify = base_aws._get_ssl_verify() - assert ssl_verify == ca_bundle_path, "SSL_CERT_FILE should be used when ssl_verify is True" - finally: - # Clean up - os.environ.pop("SSL_CERT_FILE", None) - os.unlink(ca_bundle_path) - - -if __name__ == "__main__": - # Run tests - pytest.main([__file__, "-v", "-s"]) diff --git a/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py b/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py index 80d65991acb..389c8446135 100644 --- a/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py +++ b/tests/test_litellm/llms/vertex_ai/test_vertex_llm_base.py @@ -13,7 +13,6 @@ sys.path.insert( import litellm from litellm.llms.vertex_ai.vertex_llm_base import VertexBase -from litellm.llms.vertex_ai.common_utils import _get_gemini_url def run_sync(coro): @@ -1049,139 +1048,3 @@ class TestVertexBase: MockCredentials.from_info.assert_called_once_with(json_obj) mock_creds.with_scopes.assert_called_once_with(scopes) assert result == "scoped_creds" - - def test_get_token_and_url_with_api_key(self): - """Test that API key authentication routes to Google AI Studio endpoint""" - vertex_base = VertexBase() - - # Test with API key and no credentials - should use Google AI Studio endpoint - auth_header, url = vertex_base._get_token_and_url( - model="gemini-2.0-flash-exp", - auth_header=None, - gemini_api_key="test-api-key-123", - vertex_project="test-project", - vertex_location="us-central1", - vertex_credentials=None, # No service account credentials - stream=False, - custom_llm_provider="vertex_ai", - api_base=None, - should_use_v1beta1_features=False, - mode="chat", - ) - - # Should route to Google AI Studio endpoint - assert "generativelanguage.googleapis.com" in url - assert "gemini-2.0-flash-exp" in url - assert "key=test-api-key-123" in url - assert auth_header is None # API key is in URL, not header - - def test_get_token_and_url_with_credentials(self): - """Test that service account credentials route to Vertex AI endpoint""" - vertex_base = VertexBase() - - mock_creds = MagicMock() - mock_creds.token = "mock-bearer-token" - mock_creds.expired = False - - with patch.object( - vertex_base, "_ensure_access_token", return_value=("mock-bearer-token", "test-project") - ): - # Test with credentials - should use Vertex AI endpoint - auth_header, url = vertex_base._get_token_and_url( - model="gemini-2.0-flash-exp", - auth_header="mock-bearer-token", - gemini_api_key=None, - vertex_project="test-project", - vertex_location="us-central1", - vertex_credentials={"type": "service_account"}, - stream=False, - custom_llm_provider="vertex_ai", - api_base=None, - should_use_v1beta1_features=False, - mode="chat", - ) - - # Should route to Vertex AI endpoint - assert "aiplatform.googleapis.com" in url - assert "projects/test-project" in url - assert "locations/us-central1" in url - assert auth_header == "mock-bearer-token" - - def test_get_token_and_url_api_key_with_streaming(self): - """Test API key authentication with streaming enabled""" - vertex_base = VertexBase() - - auth_header, url = vertex_base._get_token_and_url( - model="gemini-2.0-flash-exp", - auth_header=None, - gemini_api_key="test-api-key-456", - vertex_project="test-project", - vertex_location="us-central1", - vertex_credentials=None, - stream=True, # Streaming enabled - custom_llm_provider="vertex_ai", - api_base=None, - should_use_v1beta1_features=False, - mode="chat", - ) - - # Should route to Google AI Studio endpoint with streaming - assert "generativelanguage.googleapis.com" in url - assert "streamGenerateContent" in url - assert "key=test-api-key-456" in url - assert "alt=sse" in url - assert auth_header is None - - def test_get_token_and_url_api_key_priority(self): - """Test that credentials take priority over API key when both are provided""" - vertex_base = VertexBase() - - # When both API key and credentials are provided, credentials take priority - mock_creds = MagicMock() - mock_creds.token = "mock-bearer-token" - mock_creds.expired = False - - with patch.object( - vertex_base, "_ensure_access_token", return_value=("mock-bearer-token", "test-project") - ): - auth_header, url = vertex_base._get_token_and_url( - model="gemini-2.0-flash-exp", - auth_header="mock-bearer-token", - gemini_api_key="test-api-key-789", - vertex_project="test-project", - vertex_location="us-central1", - vertex_credentials={"type": "service_account"}, # Credentials provided - stream=False, - custom_llm_provider="vertex_ai", - api_base=None, - should_use_v1beta1_features=False, - mode="chat", - ) - - # Should use Vertex AI endpoint with Bearer token (credentials take priority) - assert "aiplatform.googleapis.com" in url - assert auth_header == "mock-bearer-token" - - def test_get_token_and_url_with_embedding_mode(self): - """Test API key authentication with embedding mode""" - vertex_base = VertexBase() - - auth_header, url = vertex_base._get_token_and_url( - model="text-embedding-004", - auth_header=None, - gemini_api_key="test-embedding-key", - vertex_project="test-project", - vertex_location="us-central1", - vertex_credentials=None, - stream=False, - custom_llm_provider="vertex_ai", - api_base=None, - should_use_v1beta1_features=False, - mode="embedding", - ) - - # Should route to Google AI Studio endpoint for embeddings - assert "generativelanguage.googleapis.com" in url - assert "embedContent" in url - assert "key=test-embedding-key" in url - assert auth_header is None \ No newline at end of file